Compare commits
107
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7e4071a065 | ||
|
|
6e261ccda1 | ||
|
|
ded555961c | ||
|
|
802d51ba74 | ||
|
|
570f6f545c | ||
|
|
da5f404041 | ||
|
|
ab732dc455 | ||
|
|
4ea4a040d2 | ||
|
|
0b10a637ae | ||
|
|
10956c6660 | ||
|
|
18408b0666 | ||
|
|
27f87cb5ba | ||
|
|
7ee6033fa4 | ||
|
|
e52ae242fc | ||
|
|
9da1a9899c | ||
|
|
92b03b8d07 | ||
|
|
983b4ed05a | ||
|
|
0ac07dadc9 | ||
|
|
4091ea6912 | ||
|
|
d8cf533b73 | ||
|
|
1638845a4b | ||
|
|
7082699f5f | ||
|
|
f0611ef8ac | ||
|
|
eb6e8edafe | ||
|
|
59ba95a315 | ||
|
|
f587e7e05e | ||
|
|
b541921218 | ||
|
|
a91365b3d9 | ||
|
|
ffa45edc8b | ||
|
|
d446ddd2ed | ||
|
|
0a4388c3bc | ||
|
|
0045fd87ba | ||
|
|
ad0fd52cb8 | ||
|
|
65a44a5568 | ||
|
|
e186858315 | ||
|
|
7c5645fe48 | ||
|
|
127ea2dad9 | ||
|
|
30d6ea622a | ||
|
|
d0c1b3e45b | ||
|
|
f480c5d790 | ||
|
|
9f18d8340e | ||
|
|
0a84e52bf8 | ||
|
|
1cf9985da6 | ||
|
|
2d0858caf6 | ||
|
|
ec6e888076 | ||
|
|
4f02f75602 | ||
|
|
16618d2cdc | ||
|
|
15107f54be | ||
|
|
23f915c593 | ||
|
|
981a4af7e0 | ||
|
|
5ab7263c9c | ||
|
|
7f415e724f | ||
|
|
cf140eb772 | ||
|
|
e2dddc941f | ||
|
|
28b3635979 | ||
|
|
6101e429ba | ||
|
|
a4e44e8f31 | ||
|
|
c7e9b5a4f6 | ||
|
|
6bc329456a | ||
|
|
ed3b3f1c45 | ||
|
|
8259678f0a | ||
|
|
f4813c8e99 | ||
|
|
8b7f1d009c | ||
|
|
def34e71fc | ||
|
|
1b33f48acd | ||
|
|
b3a8147bd7 | ||
|
|
0730e77530 | ||
|
|
73df864fd2 | ||
|
|
11e363896f | ||
|
|
6e8b02d771 | ||
|
|
d26515706e | ||
|
|
acf8ff2c33 | ||
|
|
e5dcc9790b | ||
|
|
f8ad4ac971 | ||
|
|
914af52b94 | ||
|
|
bdd8a48476 | ||
|
|
f309ce04f4 | ||
|
|
9989ebbec7 | ||
|
|
626e3c5e9b | ||
|
|
2c63c76288 | ||
|
|
05fc1764de | ||
|
|
65bb04d890 | ||
|
|
24c49f56e1 | ||
|
|
0217662808 | ||
|
|
4231301896 | ||
|
|
58927ed7e5 | ||
|
|
6e0e0a62c6 | ||
|
|
88ad432758 | ||
|
|
9333d0572f | ||
|
|
f3342118f5 | ||
|
|
2723b6f778 | ||
|
|
a01a4f3b3d | ||
|
|
e1605e3235 | ||
|
|
1e80c57484 | ||
|
|
c3e9c59086 | ||
|
|
88ed3c5417 | ||
|
|
3fc2d5b083 | ||
|
|
105203b447 | ||
|
|
f9cf0007c5 | ||
|
|
69b018cc32 | ||
|
|
cd812cc00e | ||
|
|
f157a93c9c | ||
|
|
88644e57d6 | ||
|
|
10c11d1806 | ||
|
|
c4595700ba | ||
|
|
28b4dabbaa | ||
|
|
4cb4f2a80f |
+2
-1
@@ -6,4 +6,5 @@ zig-out/
|
||||
.idea/
|
||||
|
||||
.claude/
|
||||
.github/
|
||||
.github/
|
||||
/var/log/
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
0.16.0
|
||||
@@ -1,8 +1,8 @@
|
||||
# DanOS
|
||||
Codename: Shodan
|
||||
Version: 1
|
||||
|
||||
A small resilient operating system, written from scratch in Zig.
|
||||
**Codename: Shodan**
|
||||
|
||||
A very small resilient operating system.
|
||||
|
||||
## Zen of DanOS:
|
||||
|
||||
@@ -18,12 +18,12 @@ A small resilient operating system, written from scratch in Zig.
|
||||
- Useful during driver development.
|
||||
- Drivers can claim MMIO / ports
|
||||
- Driver resources (e.g. IRQ/Port/MMIO) claims are automatically cleaned up if the driver dies or is killed
|
||||
- Drivers can also hook into the process lifecyle to clean up or reset hardware
|
||||
- Drivers can also hook into the process lifecycle to clean up or reset hardware
|
||||
- No legacy to deal with
|
||||
- Zig code uses a clean coding style (Zen of Zig)
|
||||
- Favor reading code over writing code.
|
||||
- No magic numbers.
|
||||
- No shortend names unless its for ABI compatibility or acronyms
|
||||
- No shortened names unless its for ABI compatibility or acronyms
|
||||
- Inter-Process Communication (IPC)
|
||||
- Publish and subscribe to Asynchronous Messages
|
||||
- Talk to services and processes synchronously
|
||||
@@ -54,6 +54,17 @@ the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||
|
||||
## Release media
|
||||
|
||||
```sh
|
||||
zig build release-x86-64
|
||||
```
|
||||
|
||||
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
|
||||
USB stick (balenaEtcher, dd) or burned to optical media — see
|
||||
[docs/release-iso.md](docs/release-iso.md). `zig build check-iso-image`
|
||||
validates it without booting.
|
||||
|
||||
## Run
|
||||
|
||||
Boot it in QEMU with OVMF (opens a display window):
|
||||
|
||||
+262
-50
@@ -2,6 +2,8 @@ const std = @import("std");
|
||||
const uefi = std.os.uefi;
|
||||
const elf = std.elf;
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||
const EdidActive = uefi.protocol.edid.Active;
|
||||
@@ -14,11 +16,11 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
/// The kernel image: /system/kernel.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||
|
||||
/// The init program: /system/services/init.
|
||||
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||
|
||||
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||
/// The user binaries: everything under /system except the kernel itself. The
|
||||
/// loader walks this tree and packs it into the in-RAM initial_ramdisk image —
|
||||
/// the volume's file structure is the single source of truth (no packed image
|
||||
/// artifact on disk).
|
||||
const system_directory_name = std.unicode.utf8ToUtf16LeStringLiteral("system");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
@@ -63,20 +65,14 @@ fn boot() !noreturn {
|
||||
|
||||
const entry = try loadKernel(bs, &boot_information);
|
||||
|
||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_information) catch |err| {
|
||||
log("EFI: no /system/services/init (");
|
||||
// Best effort: a volume without a /system tree of user binaries still boots
|
||||
// (kernel-only). The tree — init included — becomes the initial_ramdisk.
|
||||
loadSystemTree(bs, &boot_information) catch |err| {
|
||||
log("EFI: no /system binaries (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||
log("EFI: no initial_ramdisk (");
|
||||
logBytes(@errorName(err));
|
||||
log(")\r\n");
|
||||
};
|
||||
|
||||
// Build the page tables the kernel starts life on: identity + a physmap of
|
||||
// low RAM, plus the higher-half kernel image once it links high. Allocated
|
||||
// now, while boot services (and the memory map) are still stable — nothing
|
||||
@@ -84,7 +80,7 @@ fn boot() !noreturn {
|
||||
// the map and exiting would invalidate the map key.
|
||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||
|
||||
log("EFI: kernel loaded, exiting boot services\r\n");
|
||||
progress("EFI: kernel loaded, exiting boot services\r\n");
|
||||
boot_information.memory_map = try exitBootServices(bs);
|
||||
|
||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||
@@ -96,7 +92,7 @@ fn boot() !noreturn {
|
||||
}
|
||||
|
||||
/// A display resolution in pixels.
|
||||
const Resolution = struct { width: u32, height: u32 };
|
||||
const Resolution = struct { width: u32, height: u32, refresh_hz: u32 };
|
||||
|
||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||
/// and read the resulting graphics mode into our own framebuffer description.
|
||||
@@ -127,6 +123,10 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||
// Each pixel is 32 bits, so the byte pitch is 4 * pixels-per-row.
|
||||
.pitch = info.pixels_per_scan_line * 4,
|
||||
.format = try pixelFormat(info.pixel_format),
|
||||
// The refresh rate rides the EDID preferred timing. If the firmware kept a
|
||||
// non-native mode it may not describe that mode exactly — but it is the panel's
|
||||
// own clock, a far better frame-clock seed than a hardcoded 60 Hz.
|
||||
.refresh_hz = if (native) |n| n.refresh_hz else 0,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -175,10 +175,12 @@ fn nativeResolution(bs: *uefi.tables.BootServices, handles: []uefi.Handle) ?Reso
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Parse the native resolution from a raw EDID block. The first Detailed Timing
|
||||
/// Descriptor (at byte 54) is the preferred — i.e. native — mode by convention;
|
||||
/// its active pixel counts are split across low bytes and the high nibbles of
|
||||
/// later bytes.
|
||||
/// Parse the native resolution and refresh rate from a raw EDID block. The first
|
||||
/// Detailed Timing Descriptor (at byte 54) is the preferred — i.e. native — mode by
|
||||
/// convention; its active pixel counts are split across low bytes and the high nibbles
|
||||
/// of later bytes. The refresh rate is derived, not stored: the descriptor carries the
|
||||
/// pixel clock (10 kHz units) and the active+blanking extents, and
|
||||
/// refresh = clock / (horizontal total × vertical total).
|
||||
fn edidNative(edid: []const u8) ?Resolution {
|
||||
if (edid.len < 128) return null;
|
||||
// Every EDID begins with this fixed 8-byte header.
|
||||
@@ -192,7 +194,13 @@ fn edidNative(edid: []const u8) ?Resolution {
|
||||
const w = @as(u32, dtd[2]) | (@as(u32, dtd[4] & 0xf0) << 4);
|
||||
const h = @as(u32, dtd[5]) | (@as(u32, dtd[7] & 0xf0) << 4);
|
||||
if (w == 0 or h == 0) return null;
|
||||
return .{ .width = w, .height = h };
|
||||
|
||||
const clock_hz = (@as(u64, dtd[0]) | (@as(u64, dtd[1]) << 8)) * 10_000;
|
||||
const h_blank = @as(u64, dtd[3]) | (@as(u64, dtd[4] & 0x0f) << 8);
|
||||
const v_blank = @as(u64, dtd[6]) | (@as(u64, dtd[7] & 0x0f) << 8);
|
||||
const total = (@as(u64, w) + h_blank) * (@as(u64, h) + v_blank);
|
||||
const refresh: u32 = if (total == 0) 0 else @intCast((clock_hz + total / 2) / total);
|
||||
return .{ .width = w, .height = h, .refresh_hz = refresh };
|
||||
}
|
||||
|
||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||
@@ -356,11 +364,36 @@ fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) nor
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Read a whole file off the boot volume into a pool buffer that outlives the
|
||||
/// loader. The buffer is deliberately NOT freed: it's LoaderData, which the
|
||||
/// memory-map conversion classifies as reserved, so the kernel identity-maps it
|
||||
/// and reads from there. Returns the buffer (pointer + length).
|
||||
fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
// --- the /system tree -> initial_ramdisk ------------------------------------
|
||||
|
||||
/// Cap on bundled binaries. Generous: the tree carries ~30 today.
|
||||
const maximum_bundled = 64;
|
||||
|
||||
/// How deep the walk goes below /system ("/system/services/x" is depth 1).
|
||||
const maximum_tree_depth = 3;
|
||||
|
||||
/// One binary discovered under /system: its FHS path (UTF-8, '/'-separated,
|
||||
/// NUL-free) and its contents in a transient pool buffer.
|
||||
const Bundled = struct {
|
||||
path: [initial_ramdisk.maximum_name]u8,
|
||||
path_len: usize,
|
||||
data: []align(8) u8,
|
||||
};
|
||||
|
||||
/// Gather the boot volume's user binaries into an in-RAM v2 initial_ramdisk
|
||||
/// image, entries named by full FHS path — the volume's file structure is the
|
||||
/// single source of truth (no packed ramdisk artifact; init travels in the
|
||||
/// table like everything else).
|
||||
///
|
||||
/// Two strategies, most portable first:
|
||||
/// 1. /system/manifest (written by the build): each listed path is opened BY
|
||||
/// NAME — the case-insensitive lookup every firmware FAT driver gets
|
||||
/// right, and the only file access the pre-tree loader ever used.
|
||||
/// 2. No manifest: ENUMERATE the /system tree. Portable in principle, but
|
||||
/// firmware differs in what names enumeration returns (bare 8.3 entries
|
||||
/// come back uppercase on some drivers), so this is the fallback for
|
||||
/// hand-assembled sticks, not the primary path.
|
||||
fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
@@ -370,40 +403,211 @@ fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
const root = try fs.openVolume();
|
||||
defer _ = root.close() catch {};
|
||||
|
||||
const file = try root.open(name, .read, .{});
|
||||
defer _ = file.close() catch {};
|
||||
// Unconditional breadcrumb (con_out, independent of -Dserial): this phase
|
||||
// is where a slow firmware stalls, and a silent black screen here already
|
||||
// cost a real-hardware debugging session.
|
||||
log("EFI: loading the system...\r\n");
|
||||
|
||||
// The capsule (boot\system.img) first: one open + one sequential read is
|
||||
// the only firmware file I/O shape that is fast everywhere. It is already
|
||||
// the kernel's wire format — hand it over as-is.
|
||||
if (loadCapsule(bs, root, boot_information)) {
|
||||
log("EFI: system image loaded, starting the kernel\r\n");
|
||||
return;
|
||||
}
|
||||
|
||||
var list: [maximum_bundled]Bundled = undefined;
|
||||
var count: usize = 0;
|
||||
|
||||
loadByManifest(bs, root, &list, &count) catch {
|
||||
count = 0; // a torn manifest read leaves partial entries; start over
|
||||
};
|
||||
if (count == 0) {
|
||||
const system_directory = try root.open(system_directory_name, .read, .{});
|
||||
defer _ = system_directory.close() catch {};
|
||||
try walkDirectory(bs, system_directory, "/system", 0, &list, &count);
|
||||
}
|
||||
if (count == 0) return error.NoBinaries;
|
||||
|
||||
// Assemble the v2 image: header, entry table, then the blobs.
|
||||
const table_end = @sizeOf(initial_ramdisk.Header) + count * @sizeOf(initial_ramdisk.Entry);
|
||||
var total: usize = table_end;
|
||||
for (list[0..count]) |e| total += e.data.len;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, total); // survives the handoff
|
||||
std.mem.bytesAsValue(initial_ramdisk.Header, image[0..@sizeOf(initial_ramdisk.Header)]).* = .{
|
||||
.magic = initial_ramdisk.magic,
|
||||
.count = @intCast(count),
|
||||
};
|
||||
var offset: usize = table_end;
|
||||
for (list[0..count], 0..) |e, i| {
|
||||
var record = initial_ramdisk.Entry{ .name = @splat(0), .offset = offset, .len = e.data.len };
|
||||
@memcpy(record.name[0..e.path_len], e.path[0..e.path_len]);
|
||||
const slot = image[@sizeOf(initial_ramdisk.Header) + i * @sizeOf(initial_ramdisk.Entry) ..][0..@sizeOf(initial_ramdisk.Entry)];
|
||||
std.mem.bytesAsValue(initial_ramdisk.Entry, slot).* = record;
|
||||
@memcpy(image[offset..][0..e.data.len], e.data);
|
||||
offset += e.data.len;
|
||||
_ = bs.freePool(e.data.ptr) catch {};
|
||||
}
|
||||
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = total;
|
||||
log("EFI: /system tree loaded, starting the kernel\r\n");
|
||||
}
|
||||
|
||||
/// The boot capsule: the bundled binaries as one v2 initial_ramdisk image.
|
||||
const capsule_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\system.img");
|
||||
|
||||
/// Load boot\system.img whole and hand it to the kernel unmodified — it is
|
||||
/// already the initial_ramdisk wire format. Returns false (capsule absent or
|
||||
/// unreadable or wrong magic) to let the caller fall back to per-file loading.
|
||||
fn loadCapsule(bs: *uefi.tables.BootServices, root: *uefi.protocol.File, boot_information: *BootInformation) bool {
|
||||
const file = root.open(capsule_file_name, .read, .{}) catch return false;
|
||||
defer _ = file.close() catch {};
|
||||
const image = readWholeFile(bs, file) catch return false;
|
||||
if (image.len < @sizeOf(initial_ramdisk.Header) or
|
||||
std.mem.bytesToValue(initial_ramdisk.Header, image[0..@sizeOf(initial_ramdisk.Header)]).magic != initial_ramdisk.magic)
|
||||
{
|
||||
_ = bs.freePool(image.ptr) catch {};
|
||||
return false;
|
||||
}
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The manifest path, and a scratch limit for its UTF-16 conversion.
|
||||
const manifest_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\manifest");
|
||||
|
||||
/// Load every binary the manifest lists, opening each path by name from the
|
||||
/// volume root. A listed-but-unopenable file is skipped (the kernel reports the
|
||||
/// absence); a missing manifest errors so the caller falls back to the walk.
|
||||
fn loadByManifest(bs: *uefi.tables.BootServices, root: *uefi.protocol.File, list: *[maximum_bundled]Bundled, count: *usize) !void {
|
||||
const manifest_handle = try root.open(manifest_file_name, .read, .{});
|
||||
var manifest_open = true;
|
||||
defer if (manifest_open) {
|
||||
_ = manifest_handle.close() catch {};
|
||||
};
|
||||
const manifest = try readWholeFile(bs, manifest_handle);
|
||||
_ = manifest_handle.close() catch {};
|
||||
manifest_open = false;
|
||||
defer _ = bs.freePool(manifest.ptr) catch {};
|
||||
|
||||
var lines = std.mem.tokenizeAny(u8, manifest, "\r\n");
|
||||
while (lines.next()) |line| {
|
||||
if (line.len < 2 or line[0] != '/') continue;
|
||||
if (line.len >= initial_ramdisk.maximum_name) continue;
|
||||
if (count.* == maximum_bundled) return;
|
||||
|
||||
// "/system/services/init" -> UTF-16 "system\services\init".
|
||||
var name16: [initial_ramdisk.maximum_name]u16 = undefined;
|
||||
var i: usize = 0;
|
||||
for (line[1..]) |c| {
|
||||
name16[i] = if (c == '/') '\\' else c;
|
||||
i += 1;
|
||||
}
|
||||
name16[i] = 0;
|
||||
|
||||
const file = root.open(@ptrCast(name16[0..i :0]), .read, .{}) catch continue;
|
||||
defer _ = file.close() catch {};
|
||||
const data = readWholeFile(bs, file) catch continue;
|
||||
|
||||
var entry: *Bundled = &list[count.*];
|
||||
@memcpy(entry.path[0..line.len], line);
|
||||
entry.path_len = line.len;
|
||||
entry.data = data;
|
||||
count.* += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Recursively collect the regular files below `directory` into `list`. Top-level
|
||||
/// files (depth 0) are skipped: the only one is /system/kernel, which loadKernel
|
||||
/// has already consumed and which is not a spawnable user binary.
|
||||
fn walkDirectory(
|
||||
bs: *uefi.tables.BootServices,
|
||||
directory: *uefi.protocol.File,
|
||||
prefix: []const u8,
|
||||
depth: usize,
|
||||
list: *[maximum_bundled]Bundled,
|
||||
count: *usize,
|
||||
) !void {
|
||||
// Each read() on a directory yields one EFI_FILE_INFO; zero bytes means done.
|
||||
var info_buffer: [1024]u8 align(8) = undefined;
|
||||
while (true) {
|
||||
const n = try directory.read(&info_buffer);
|
||||
if (n == 0) return;
|
||||
const info: *const uefi.protocol.File.Info.File = @ptrCast(@alignCast(&info_buffer));
|
||||
const name16 = info.getFileName();
|
||||
|
||||
// Convert the (ASCII in practice) UTF-16 name. A hostile-shaped entry
|
||||
// (too long, non-ASCII) is SKIPPED, never fatal — one odd file on a
|
||||
// hand-written stick must not cost the whole boot. Names are lowered:
|
||||
// the danos tree is canonically lowercase and FAT lookups are
|
||||
// case-insensitive, but firmware ENUMERATION returns whatever the
|
||||
// directory stores — an 8.3 short entry comes back uppercase ("INIT"),
|
||||
// which would otherwise poison every path comparison downstream.
|
||||
var name_buffer: [initial_ramdisk.maximum_name]u8 = undefined;
|
||||
var name_length: usize = 0;
|
||||
var name_ok = true;
|
||||
while (name16[name_length] != 0) : (name_length += 1) {
|
||||
if (name_length == name_buffer.len) {
|
||||
name_ok = false;
|
||||
break;
|
||||
}
|
||||
const c = name16[name_length];
|
||||
if (c > 0x7F) {
|
||||
name_ok = false;
|
||||
break;
|
||||
}
|
||||
name_buffer[name_length] = std.ascii.toLower(@intCast(c));
|
||||
}
|
||||
if (!name_ok) continue;
|
||||
const name = name_buffer[0..name_length];
|
||||
// Skip dot entries: "." / ".." and host-OS litter (macOS "._*" AppleDouble
|
||||
// resource forks, ".fseventsd", ".Spotlight-V100") a copied-onto stick
|
||||
// accumulates — none of it is a danos binary.
|
||||
if (name.len == 0 or name[0] == '.') continue;
|
||||
|
||||
if (info.attribute.directory) {
|
||||
if (depth == maximum_tree_depth) continue;
|
||||
var child_prefix: [initial_ramdisk.maximum_name]u8 = undefined;
|
||||
const child = try std.fmt.bufPrint(&child_prefix, "{s}/{s}", .{ prefix, name });
|
||||
const child_directory = try directory.open(name16, .read, .{});
|
||||
defer _ = child_directory.close() catch {};
|
||||
try walkDirectory(bs, child_directory, child, depth + 1, list, count);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (depth == 0) continue; // /system/kernel — already loaded, not bundled
|
||||
if (count.* == maximum_bundled) return error.TooManyBinaries;
|
||||
|
||||
var entry: *Bundled = &list[count.*];
|
||||
const path = std.fmt.bufPrint(&entry.path, "{s}/{s}", .{ prefix, name }) catch continue; // path too long: skip the file, keep the boot
|
||||
entry.path_len = path.len;
|
||||
|
||||
const file = directory.open(name16, .read, .{}) catch continue;
|
||||
defer _ = file.close() catch {};
|
||||
entry.data = readWholeFile(bs, file) catch continue; // unreadable/empty: skip
|
||||
count.* += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Read an open file completely into a fresh pool buffer that survives the
|
||||
/// handoff (LoaderData is classified reserved, so the kernel identity-maps it).
|
||||
fn readWholeFile(bs: *uefi.tables.BootServices, file: *uefi.protocol.File) ![]align(8) u8 {
|
||||
try file.setPosition(seek_end);
|
||||
const size: usize = @intCast(try file.getPosition());
|
||||
try file.setPosition(0);
|
||||
if (size == 0) return error.EmptyFile;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||
|
||||
const buffer = try bs.allocatePool(.loader_data, size);
|
||||
var read_total: usize = 0;
|
||||
while (read_total < size) {
|
||||
const n = try file.read(image[read_total..]);
|
||||
const n = try file.read(buffer[read_total..]);
|
||||
if (n == 0) return error.UnexpectedEof;
|
||||
read_total += n;
|
||||
}
|
||||
return image[0..size];
|
||||
}
|
||||
|
||||
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
log("EFI: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
log("EFI: initial_ramdisk loaded\r\n");
|
||||
return buffer[0..size];
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
@@ -561,6 +765,14 @@ fn log(comptime message: []const u8) void {
|
||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||
}
|
||||
|
||||
/// A boot-progress breadcrumb: like `log`, but compiled out unless `-Dserial`
|
||||
/// (off by default), so a real-hardware boot stays silent. Fatal errors use
|
||||
/// `log` directly and always show, so a failed boot still explains itself.
|
||||
fn progress(comptime message: []const u8) void {
|
||||
if (!build_options.serial) return;
|
||||
log(message);
|
||||
}
|
||||
|
||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||
fn logBytes(bytes: []const u8) void {
|
||||
const out = uefi.system_table.con_out orelse return;
|
||||
|
||||
@@ -54,6 +54,12 @@ fn timestamp(b: *std.Build) []const u8 {
|
||||
/// can't reach), linked against the `runtime` runtime library with the shared user
|
||||
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
||||
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||
///
|
||||
/// The compilation root is not the program's own file but the shared shim
|
||||
/// library/runtime/root.zig, which supplies the root declarations (`main`
|
||||
/// re-export, panic handler, `_start` pull) so a program only defines
|
||||
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||
/// imports; reach it through `programModule` to add per-binary imports.
|
||||
fn addUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
@@ -64,27 +70,66 @@ fn addUserBinary(
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, false);
|
||||
}
|
||||
|
||||
/// As `addUserBinary`, but built multi-threaded (`single_threaded = false`) so real
|
||||
/// atomics/TLS work — required before a binary may call `runtime.Thread.spawn`
|
||||
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||
fn addThreadedUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, true);
|
||||
}
|
||||
|
||||
fn addUserBinaryImpl(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
threaded: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Settings (target, optimize, code model, ...) live on the root module only;
|
||||
// the program and runtime modules leave theirs null and inherit them.
|
||||
const program_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
// to any program that wants it. See library/xkeyboard-config/.
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
},
|
||||
});
|
||||
const exe = b.addExecutable(.{
|
||||
.name = name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.root_source_file = b.path("library/runtime/root.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
.single_threaded = true,
|
||||
.single_threaded = !threaded, // a threaded binary needs real atomics/TLS
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
// to any program that wants it. See library/xkeyboard-config/.
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
.{ .name = "program", .module = program_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
@@ -96,6 +141,123 @@ fn addUserBinary(
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// The `program` module of a binary built by `addUserBinary` — the module rooted
|
||||
/// at the program's own source file. Per-binary imports (protocol modules, bus
|
||||
/// ABIs) go here, not on the root shim: module imports are not transitive, so an
|
||||
/// import added to the root would be invisible to the program's code.
|
||||
fn programModule(exe: *std.Build.Step.Compile) *std.Build.Module {
|
||||
return exe.root_module.import_table.get("program").?;
|
||||
}
|
||||
|
||||
/// The modules the kernel imports, gathered once so both kernel variants (the
|
||||
/// installed one and the serial-enabled one `run-x86-64` boots) are built from
|
||||
/// the same set. `build_options` is *not* here — it carries `serial`/`test_case`,
|
||||
/// which differ per variant, so `addKernel` builds it fresh each time.
|
||||
const KernelModules = struct {
|
||||
boot_handoff: *std.Build.Module,
|
||||
abi: *std.Build.Module,
|
||||
device_abi: *std.Build.Module,
|
||||
architecture: *std.Build.Module,
|
||||
platform: *std.Build.Module,
|
||||
parameters: *std.Build.Module,
|
||||
initial_ramdisk: *std.Build.Module,
|
||||
};
|
||||
|
||||
/// Build the freestanding x86_64 kernel ELF. Factored so we can build it twice
|
||||
/// from one recipe: the installed/flashable image (serial off by default) and the
|
||||
/// serial-enabled variant `run-x86-64` boots — they differ only in the `serial`
|
||||
/// build option baked into `build_options`.
|
||||
fn addKernel(
|
||||
b: *std.Build,
|
||||
kernel_target: std.Build.ResolvedTarget,
|
||||
optimize: std.builtin.OptimizeMode,
|
||||
modules: KernelModules,
|
||||
test_case: ?[]const u8,
|
||||
serial: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`:
|
||||
// the QEMU harness's -Dtest-case, and whether the serial log sink is compiled
|
||||
// in (see the -Dserial option). Built per variant since `serial` differs.
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
build_options.addOption(bool, "serial", serial);
|
||||
const build_options_module = build_options.createModule();
|
||||
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||
.{ .name = "abi", .module = modules.abi },
|
||||
.{ .name = "device-abi", .module = modules.device_abi },
|
||||
.{ .name = "architecture", .module = modules.architecture },
|
||||
.{ .name = "platform", .module = modules.platform },
|
||||
.{ .name = "parameters", .module = modules.parameters },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initial-ramdisk", .module = modules.initial_ramdisk },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||
// linker's AT() clauses give each segment a low physical load address
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// One user binary and its FHS home on the boot volume (and in zig-out).
|
||||
const BundledBinary = struct { path: []const u8, binary: std.Build.LazyPath };
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding the
|
||||
/// EFI stub, the kernel, and every user binary at its FHS path — the volume's
|
||||
/// /system tree IS the system image; the EFI loader walks it at boot and builds
|
||||
/// the in-RAM initial_ramdisk from it. Factored so the serial-enabled
|
||||
/// `run-x86-64` variant can bundle its own serial kernel while sharing the
|
||||
/// loader and user tree (the loader's boot breadcrumbs and init's heartbeat both
|
||||
/// follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
manifest: std.Build.LazyPath,
|
||||
capsule: std.Build.LazyPath,
|
||||
bundled: []const BundledBinary,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/manifest");
|
||||
mk_fat.addFileArg(manifest);
|
||||
mk_fat.addArg("boot/system.img");
|
||||
mk_fat.addFileArg(capsule);
|
||||
for (bundled) |item| {
|
||||
mk_fat.addArg(item.path);
|
||||
mk_fat.addFileArg(item.binary);
|
||||
}
|
||||
return fat_image;
|
||||
}
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
ensureZigVersion();
|
||||
|
||||
@@ -208,7 +370,7 @@ pub fn build(b: *std.Build) void {
|
||||
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||
// expose theirs the same way.
|
||||
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||
.root_source_file = b.path("system/vfs-protocol.zig"),
|
||||
});
|
||||
|
||||
// The input wire protocol: the input service's public interface, exposed as its own
|
||||
@@ -249,6 +411,20 @@ pub fn build(b: *std.Build) void {
|
||||
// The block protocol, so runtime.block (the block-device client) can speak it.
|
||||
runtime_module.addImport("block-protocol", block_protocol_module);
|
||||
|
||||
// The display protocol, so runtime.display (the compositor client) and the display
|
||||
// service both speak it through the runtime, like the other protocol modules.
|
||||
const display_protocol_module = b.addModule("display-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("display-protocol", display_protocol_module);
|
||||
|
||||
// The scanout protocol: the compositor's outbound present channel to a native scanout
|
||||
// driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md).
|
||||
const scanout_protocol_module = b.addModule("scanout-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/scanout-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("scanout-protocol", scanout_protocol_module);
|
||||
|
||||
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||
const power_protocol_module = b.addModule("power-protocol", .{
|
||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||
@@ -276,8 +452,9 @@ pub fn build(b: *std.Build) void {
|
||||
},
|
||||
});
|
||||
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and
|
||||
// the EFI loader (packs it in RAM from the boot volume's /system tree). No
|
||||
// dependencies.
|
||||
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||
});
|
||||
@@ -285,9 +462,16 @@ pub fn build(b: *std.Build) void {
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
const build_options_module = build_options.createModule();
|
||||
// The serial-console log sink. Off by default: a real machine often has no
|
||||
// working legacy COM1, and the boot log is kept in RAM (klog) and flushed to
|
||||
// disk instead — serial is now only a QEMU convenience. `run-x86-64` and the
|
||||
// QEMU test harness (test/qemu_test.py, which asserts on serial markers) turn
|
||||
// it on; a flashable `zig build` image leaves it out. See serial.zig.
|
||||
const serial = b.option(bool, "serial", "Compile the serial-console log sink into the kernel (default: off; run-x86-64 and the test harness enable it)") orelse false;
|
||||
// The diagnose boot: init skips the display service (and demo), so the
|
||||
// on-screen boot transcript is never suppressed — the full timestamped
|
||||
// timeline stays on the screen for real-hardware debugging by eye.
|
||||
const diagnose = b.option(bool, "diagnose", "Boot without the display service so the timestamped boot transcript stays on screen (real-hardware debugging)") orelse false;
|
||||
|
||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||
@@ -299,41 +483,17 @@ pub fn build(b: *std.Build) void {
|
||||
.abi = .none,
|
||||
});
|
||||
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "architecture", .module = architecture_module },
|
||||
.{ .name = "platform", .module = platform_module },
|
||||
.{ .name = "parameters", .module = parameters_module },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||
// linker's AT() clauses give each segment a low physical load address
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
const kernel_modules = KernelModules{
|
||||
.boot_handoff = boot_handoff_module,
|
||||
.abi = abi_module,
|
||||
.device_abi = device_abi_module,
|
||||
.architecture = architecture_module,
|
||||
.platform = platform_module,
|
||||
.parameters = parameters_module,
|
||||
.initial_ramdisk = initial_ramdisk_module,
|
||||
};
|
||||
// The installed/flashable kernel: serial follows -Dserial (off by default).
|
||||
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
||||
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
@@ -348,15 +508,21 @@ pub fn build(b: *std.Build) void {
|
||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||
b.getInstallStep().dependOn(&init_install.step);
|
||||
// init reads the same `serial` flag the kernel does: its liveness heartbeat is a
|
||||
// serial/test-build diagnostic (the QEMU harness's init tests assert on it, and
|
||||
// -Dserial images emit it), so a flashable image runs a purely event-driven PID 1
|
||||
// that wakes only for real work. The test harness builds with -Dserial=true, so
|
||||
// the heartbeat stays present under test.
|
||||
const init_options = b.addOptions();
|
||||
init_options.addOption(bool, "serial", serial);
|
||||
init_options.addOption(bool, "diagnose", diagnose);
|
||||
programModule(init_exe).addImport("build_options", init_options.createModule());
|
||||
|
||||
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||
// Each is built by the same user-binary recipe, then packed into one image by
|
||||
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
// --- the rest of the /system tree: services, drivers, test fixtures ---
|
||||
// Each is built by the same user-binary recipe and laid out at its FHS path on
|
||||
// the boot volume (see `bundled` below). The EFI loader walks the tree at boot
|
||||
// and hands the kernel an in-RAM initial_ramdisk of it (system/initial-ramdisk.zig).
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs-test/vfs-test.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
@@ -364,28 +530,35 @@ pub fn build(b: *std.Build) void {
|
||||
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||
// usb-ids.packTriple.
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-ids", usb_ids_module);
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
// The USB HID class drivers: keyboard and mouse. They own no hardware — each
|
||||
// opens its device through runtime.usb (the transfer protocol) and publishes to
|
||||
// the input service. They build chapter-9 class requests from usb-abi.
|
||||
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||
usb_hid_keyboard_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_hid_keyboard_exe).addImport("usb-abi", usb_abi_module);
|
||||
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||
usb_hid_mouse_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_hid_mouse_exe).addImport("usb-abi", usb_abi_module);
|
||||
// The USB mass-storage class driver: opens its device via runtime.usb, drives it
|
||||
// with Bulk-Only Transport + SCSI, and serves the block protocol under `.block`.
|
||||
const usb_storage_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||
usb_storage_exe.root_module.addImport("block-protocol", block_protocol_module);
|
||||
programModule(usb_storage_exe).addImport("block-protocol", block_protocol_module);
|
||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||
// Threaded: the display runs a mouse-listener thread alongside its compositor loop
|
||||
// (docs/threading.md, docs/display.md), so it opts into real atomics/TLS.
|
||||
const display_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||
const shared_memory_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shared-memory-server", "system/services/shared-memory-server/shared-memory-server.zig");
|
||||
const shared_memory_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shared-memory-client", "system/services/shared-memory-client/shared-memory-client.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
// The PCI bus driver decodes each function's class triple to human names in its
|
||||
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||
pci_bus_exe.root_module.addImport("pci-class", pci_class_module);
|
||||
programModule(pci_bus_exe).addImport("pci-class", pci_class_module);
|
||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||
// what the driver-restart scenario drives the crash-loop cap with.
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
@@ -404,13 +577,13 @@ pub fn build(b: *std.Build) void {
|
||||
.fdt => "system/services/fdt/fdt.zig",
|
||||
};
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("aml", aml_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||
device_manager_exe.root_module.addImport("pci-class", pci_class_module);
|
||||
programModule(device_manager_exe).addImport("pci-class", pci_class_module);
|
||||
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||
device_manager_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||
programModule(device_manager_exe).addImport("usb-ids", usb_ids_module);
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
@@ -418,86 +591,99 @@ pub fn build(b: *std.Build) void {
|
||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||
const logger_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "logger", "system/services/logger/logger.zig");
|
||||
// The first multi-threaded binary: exercises runtime.Thread over the thread ABI
|
||||
// (docs/threading.md). Built threaded so its shared-memory poll is real.
|
||||
const thread_test_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "thread-test", "system/services/thread-test/thread-test.zig");
|
||||
|
||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||
mk_run.addArg("vfs");
|
||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||
mk_run.addArg("vfs-test");
|
||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-bus");
|
||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-keyboard");
|
||||
mk_run.addFileArg(ps2_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-mouse");
|
||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-xhci-bus");
|
||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-keyboard");
|
||||
mk_run.addFileArg(usb_hid_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-mouse");
|
||||
mk_run.addFileArg(usb_hid_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-storage");
|
||||
mk_run.addFileArg(usb_storage_exe.getEmittedBin());
|
||||
mk_run.addArg("fat");
|
||||
mk_run.addFileArg(fat_exe.getEmittedBin());
|
||||
mk_run.addArg("fat-test");
|
||||
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
||||
mk_run.addArg("pci-bus");
|
||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("crash-test");
|
||||
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||
mk_run.addArg("device-list");
|
||||
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||
mk_run.addArg("discovery");
|
||||
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||
mk_run.addArg("device-manager");
|
||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||
mk_run.addArg("input");
|
||||
mk_run.addFileArg(input_exe.getEmittedBin());
|
||||
mk_run.addArg("input-source");
|
||||
mk_run.addFileArg(input_source_exe.getEmittedBin());
|
||||
mk_run.addArg("input-test");
|
||||
mk_run.addFileArg(input_test_exe.getEmittedBin());
|
||||
mk_run.addArg("args-echo");
|
||||
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||
mk_run.addArg("process-test");
|
||||
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||
mk_run.addArg("log-flush");
|
||||
mk_run.addFileArg(log_flush_exe.getEmittedBin());
|
||||
// Every user binary and its FHS home on the boot volume. There is no packed
|
||||
// ramdisk artifact any more: make-fat-image.py lays each binary out at this
|
||||
// path on the image, and the EFI loader walks /system at boot and builds the
|
||||
// in-RAM initial_ramdisk table from the tree — the volume's file structure is
|
||||
// the single source of truth. Entry names (and hence argv[0] and task names)
|
||||
// are these paths with a leading slash.
|
||||
const bundled = [_]BundledBinary{
|
||||
.{ .path = "system/services/init", .binary = init_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/fat", .binary = fat_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/display", .binary = display_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/display-demo", .binary = display_demo_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/device-manager", .binary = device_manager_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/input", .binary = input_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/discovery", .binary = discovery_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/logger", .binary = logger_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-bus", .binary = ps2_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-keyboard", .binary = ps2_keyboard_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-mouse", .binary = ps2_mouse_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-xhci-bus", .binary = usb_xhci_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-hid-keyboard", .binary = usb_hid_keyboard_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-hid-mouse", .binary = usb_hid_mouse_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-storage", .binary = usb_storage_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/virtio-gpu", .binary = virtio_gpu_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/pci-bus", .binary = pci_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/vfs-test", .binary = vfstest_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/fat-test", .binary = fat_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/shared-memory-server", .binary = shared_memory_server_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/shared-memory-client", .binary = shared_memory_client_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/crash-test", .binary = crash_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/device-list", .binary = device_list_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/input-source", .binary = input_source_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/input-test", .binary = input_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/args-echo", .binary = args_echo_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/process-test", .binary = process_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/thread-test", .binary = thread_test_exe.getEmittedBin() },
|
||||
};
|
||||
|
||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||
.{ vfs_exe, "system/services" },
|
||||
.{ device_manager_exe, "system/services" },
|
||||
.{ input_exe, "system/services" },
|
||||
.{ ps2_bus_exe, "system/drivers" },
|
||||
.{ ps2_keyboard_exe, "system/drivers" },
|
||||
.{ ps2_mouse_exe, "system/drivers" },
|
||||
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||
.{ usb_hid_keyboard_exe, "system/drivers" },
|
||||
.{ usb_hid_mouse_exe, "system/drivers" },
|
||||
.{ usb_storage_exe, "system/drivers" },
|
||||
.{ fat_exe, "system/services" },
|
||||
.{ log_flush_exe, "system/services" },
|
||||
}) |entry| {
|
||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||
b.getInstallStep().dependOn(&step.step);
|
||||
// The boot manifest: the FHS path of every bundled binary, one per line. The
|
||||
// EFI loader reads THIS by name and opens each listed path by name — FAT
|
||||
// name lookup is case-insensitive and firmware-portable, unlike directory
|
||||
// ENUMERATION, whose returned names vary by firmware (bare 8.3 entries come
|
||||
// back uppercase on some FAT drivers). The tree walk remains only as the
|
||||
// loader's fallback for hand-assembled sticks without a manifest.
|
||||
var manifest_text: std.ArrayListUnmanaged(u8) = .empty;
|
||||
for (bundled) |item| {
|
||||
manifest_text.append(b.allocator, '/') catch @panic("OOM");
|
||||
manifest_text.appendSlice(b.allocator, item.path) catch @panic("OOM");
|
||||
manifest_text.append(b.allocator, '\n') catch @panic("OOM");
|
||||
}
|
||||
const manifest_files = b.addWriteFiles();
|
||||
const manifest_file = manifest_files.add("manifest", manifest_text.items);
|
||||
const manifest_install = b.addInstallFileWithDir(manifest_file, .prefix, "system/manifest");
|
||||
b.getInstallStep().dependOn(&manifest_install.step);
|
||||
|
||||
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||
// The boot capsule: the same bundled list packed into ONE file (v2
|
||||
// initial_ramdisk format), because a single open + sequential read is the
|
||||
// only firmware file I/O shape that is fast everywhere — a per-file tree
|
||||
// walk measured MINUTES on real firmware. The loader tries this first,
|
||||
// then the manifest, then the walk; the running system cannot tell the
|
||||
// difference (it always receives the same in-RAM table). Derived from the
|
||||
// tree in the same build graph, so the two cannot drift.
|
||||
const mk_capsule = b.addSystemCommand(&.{"python3"});
|
||||
mk_capsule.addFileArg(b.path("tools/pack-system-image.py"));
|
||||
const capsule_img = mk_capsule.addOutputFileArg("system.img");
|
||||
for (bundled) |item| {
|
||||
mk_capsule.addArg(item.path);
|
||||
mk_capsule.addFileArg(item.binary);
|
||||
}
|
||||
const capsule_install = b.addInstallFile(capsule_img, "boot/system.img");
|
||||
b.getInstallStep().dependOn(&capsule_install.step);
|
||||
|
||||
// Install every bundled binary to its FHS home, so zig-out is a true image of
|
||||
// the filesystem — the same tree make-fat-image.py lays out on the boot volume.
|
||||
for (bundled) |item| {
|
||||
const install = b.addInstallFileWithDir(item.binary, .prefix, item.path);
|
||||
b.getInstallStep().dependOn(&install.step);
|
||||
}
|
||||
|
||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||
// The loader reads -Dserial too, so its boot-progress breadcrumbs (con_out,
|
||||
// which firmware may mirror to a serial console) are silenced by default — a
|
||||
// real-hardware boot stays quiet. Fatal-error messages ignore this and always
|
||||
// show, so a failed boot still explains itself on screen. See boot/efi.zig.
|
||||
const loader_options = b.addOptions();
|
||||
loader_options.addOption(bool, "serial", serial);
|
||||
const loader_options_module = loader_options.createModule();
|
||||
const efiexe = b.addExecutable(.{
|
||||
.name = "BOOTX64",
|
||||
.root_module = b.createModule(.{
|
||||
@@ -508,8 +694,11 @@ pub fn build(b: *std.Build) void {
|
||||
}),
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||
// The bootloader speaks the handoff contract and the ramdisk
|
||||
// container it packs the /system tree into — never the user ABI.
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||
.{ .name = "build_options", .module = loader_options_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
@@ -521,25 +710,21 @@ pub fn build(b: *std.Build) void {
|
||||
|
||||
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||
// holding exactly what the firmware and bootloader need off the ESP: the EFI
|
||||
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efiexe.getEmittedBin());
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(exe.getEmittedBin());
|
||||
mk_fat.addArg("system/services/init");
|
||||
mk_fat.addFileArg(init_exe.getEmittedBin());
|
||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||
mk_fat.addFileArg(initial_ramdisk_img);
|
||||
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||
// danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), manifest_file, capsule_img, &bundled);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), manifest_file, capsule_img, &bundled);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||
@@ -549,6 +734,31 @@ pub fn build(b: *std.Build) void {
|
||||
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||
check_fat_step.dependOn(&check_fat.step);
|
||||
|
||||
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
|
||||
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
|
||||
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
|
||||
// and MBR ESP partition entry both point at the embedded FAT image. One
|
||||
// file then boots every way release media is consumed — flashed raw to a
|
||||
// USB stick with Etcher or dd, or burned to optical media — while
|
||||
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
|
||||
const mk_iso = b.addSystemCommand(&.{"python3"});
|
||||
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
|
||||
mk_iso.addFileArg(fat_image);
|
||||
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
|
||||
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
|
||||
release_step.dependOn(&iso_install.step);
|
||||
|
||||
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
|
||||
// check-fat-image): the MBR partition, the El Torito catalog, and the
|
||||
// embedded FAT32 image must all agree.
|
||||
const check_iso = b.addSystemCommand(&.{"python3"});
|
||||
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
check_iso.addArg("--verify");
|
||||
check_iso.addFileArg(iso_image);
|
||||
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
|
||||
check_iso_step.dependOn(&check_iso.step);
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
@@ -611,8 +821,9 @@ pub fn build(b: *std.Build) void {
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image);
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_efi.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
@@ -634,13 +845,62 @@ pub fn build(b: *std.Build) void {
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
||||
run_efi.step.dependOn(b.getInstallStep());
|
||||
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||
// scratch dir first.
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// --- run-x86-64-gpu: the same boot plus a virtio-gpu adapter ---
|
||||
// The VGA device still supplies the boot (GOP) framebuffer the compositor starts
|
||||
// on; the virtio-gpu function is discovered by the device-manager stack, its
|
||||
// driver announces a shared scanout, and the compositor upgrades off the GOP
|
||||
// floor to fenced, tear-free native presents (docs/display-v2.md).
|
||||
// This is the interactive twin of the `display-native` test case, and 512M
|
||||
// matches it (the whole driver stack + the compositor's surfaces at once).
|
||||
// QEMU shows one head per adapter: pick the virtio-gpu head in the View menu
|
||||
// to watch the native output.
|
||||
const run_gpu = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"512M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_gpu.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
"-device",
|
||||
"virtio-gpu-pci",
|
||||
});
|
||||
const gpu_serial_log = b.fmt("{s}/run-x86-64-gpu-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_gpu.addArgs(&.{ "-serial", b.fmt("file:{s}", .{gpu_serial_log}) });
|
||||
run_gpu.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_gpu_step = b.step("run-x86-64-gpu", "Boot in QEMU with a virtio-gpu adapter: the compositor upgrades to fenced (tear-free) native presents; watch the virtio-gpu head in QEMU's View menu");
|
||||
run_gpu_step.dependOn(&run_gpu.step);
|
||||
|
||||
// const run_cmd = b.addRunArtifact(exe);
|
||||
// const run_step = b.step("run", "Run the app");
|
||||
// run_step.dependOn(&run_cmd.step);
|
||||
@@ -658,6 +918,7 @@ pub fn build(b: *std.Build) void {
|
||||
for ([_][]const u8{
|
||||
"system/boot-handoff.zig",
|
||||
"system/abi.zig",
|
||||
"system/initial-ramdisk.zig", // v2 path-named entries: find/basename/magic
|
||||
"system/devices/device-abi.zig",
|
||||
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||
@@ -670,10 +931,13 @@ pub fn build(b: *std.Build) void {
|
||||
"system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||
"system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||
"system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian)
|
||||
"system/services/vfs/path.zig", // mount-prefix path matching
|
||||
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"system/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||
"system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile
|
||||
"system/services/display/protocol.zig", // pack(): native pixel encoding per format
|
||||
"system/drivers/virtio-gpu/virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||
"system/drivers/virtio-gpu/virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||
}) |root| {
|
||||
const mod_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
@@ -700,6 +964,21 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
|
||||
// The tagged kernel log ring: append/wrap/reclaim/sequence-gap behavior over
|
||||
// a RAM buffer. Needs the `abi` module (record header layout), so it doesn't
|
||||
// fit the plain loop above.
|
||||
const log_ring_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/log-ring.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(log_ring_tests).step);
|
||||
|
||||
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||
// loop above.
|
||||
@@ -715,6 +994,22 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||
|
||||
// runtime.Thread's lock/condvar state machines (Mutex/Condition/RwLock/WaitGroup). Its
|
||||
// Futex seam falls back to std.Thread.Futex off the danos target, so the tests exercise
|
||||
// them with real host threads (docs/threading-plan.md M11). Like time.zig it pulls in
|
||||
// system.zig (syscall wrappers), which needs the `abi` module.
|
||||
const thread_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/thread.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(thread_tests).step);
|
||||
|
||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||
|
||||
+43
-9
@@ -39,37 +39,53 @@ rather than restate it. Roughly in the order things happen at runtime:
|
||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||
deliberately tiny.
|
||||
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs
|
||||
the public boundary that will hide them.
|
||||
13. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||
the operation table, mount routing, and the append-only evolution rules — the
|
||||
first IPC protocol documented as public ABI.
|
||||
14. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
15. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
15. **[process-management.md](process-management.md) — process management.** The
|
||||
16. **[process-management.md](process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
17. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
18. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
[resilience.md](resilience.md)'s restart goal into increments.
|
||||
18. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
19. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||
service layered on top.
|
||||
19. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
20. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||
that tear-free doesn't. Plan: [display-plan.md](display-plan.md). **v2** (complete) makes
|
||||
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||
[display-v2.md](display-v2.md), plan [display-v2-plan.md](display-v2-plan.md). Looking
|
||||
further out, two research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
|
||||
21. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||
|
||||
Start with the north star:
|
||||
@@ -88,6 +104,19 @@ Start with the north star:
|
||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||
- **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
`runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||
a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the
|
||||
native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](threading-plan.md).
|
||||
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
|
||||
stable boundary without danos growing a dynamic linker. danos's public ABI = the
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](vfs-protocol.md) first).
|
||||
|
||||
Cutting across all of these:
|
||||
|
||||
@@ -95,6 +124,11 @@ Cutting across all of these:
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[release-iso.md](release-iso.md) — the release ISO.** The flashable boot
|
||||
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
|
||||
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
|
||||
flash to USB or a burner writes to disc — built by an in-repo pure-Python
|
||||
tool, like the FAT image itself.
|
||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
@@ -239,5 +273,5 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
# Display service — build plan (v1: the dumb-framebuffer compositor)
|
||||
|
||||
The ordered, checkpointable build-out for [display.md](display.md). Each milestone is
|
||||
small, lands on its own, and ends in a **verifiable gate** — shaped for a `/loop` run.
|
||||
Read [display.md](display.md) first for the *why*; this is the *what* and the *order*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **Handoff = device node + write-combining `mmio_map`.** The kernel seeds a synthetic
|
||||
`display0` node from `BootInformation.framebuffer`; the service claims + WC-maps it.
|
||||
(Not a bespoke `framebuffer_map` syscall — the device route inherits ownership,
|
||||
release-on-death, and re-claim-on-restart.)
|
||||
- **v1 = the full compositor pipeline on the dumb framebuffer.** One `display` service
|
||||
owns the LFB + a cacheable back buffer + a layer stack; double-buffer + damage-driven
|
||||
present; clients draw via server-side commands. **No** runtime mode-setting, **no**
|
||||
shared-memory surfaces — both deferred (see display.md, "What v1 does not do").
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations in
|
||||
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||
go through `addUserBinary` in [build.zig](../build.zig) and get packed into the
|
||||
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||
`runtime` module.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
- `zig build test` — host unit tests (compositor math: layer clipping, damage merge,
|
||||
pitch/format blits are all host-testable with a fake framebuffer).
|
||||
- `python3 test/qemu_test.py <case>` — boots the real kernel in QEMU; assert on the
|
||||
serial log ([tests.zig](../system/kernel/tests.zig) is the registry).
|
||||
- The `run-efi` target renders to QEMU's display (`-device VGA,edid=on,xres=1280,yres=720`)
|
||||
— a screenshot confirms pixels for the milestones whose gate is visual.
|
||||
|
||||
---
|
||||
|
||||
## D1 — The handoff primitive (kernel) ✅
|
||||
|
||||
Make the boot framebuffer reachable and mappable **write-combining** from user space.
|
||||
|
||||
- [x] [device-abi.zig](../system/devices/device-abi.zig): added `DeviceClass.display`; a
|
||||
`DisplayInfo{ width, height, pitch, format }` carried on the descriptor; a
|
||||
`flags` field on `ResourceDescriptor` + `resource_flag_write_combining`.
|
||||
- [x] [devices-broker.zig](../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||
pitch, format)` publishes a root-level `display` node with one WC-flagged `memory`
|
||||
resource `[base, height*pitch]` + the `DisplayInfo`; `displayDevice()` /
|
||||
`displayClaimed()`. Seeded from `kmain` after `devices_broker.init`.
|
||||
- [x] [process.zig](../system/kernel/process.zig) `systemMmioMap` + paging
|
||||
(`mapUserDeviceInto` gains a `write_combining` bool): a resource's WC flag maps it
|
||||
through the WC PAT slot (`setupPat`) instead of strong-uncacheable.
|
||||
- [x] [console.zig](../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||
the display device is claimed (driven from `systemDeviceClaim` / release); the
|
||||
terminal panic + exception paths clear it first so a dying machine still draws.
|
||||
|
||||
**Gate (met, automated):** the `display` kernel test (`python3 test/qemu_test.py display`,
|
||||
`displayTest` in [tests.zig](../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||
and geometry, then walks the real claim + `mmio_map` path into a throwaway address space
|
||||
and verifies the leaf is **write-combining** (PAT entry 4: PAT bit set, PCD/PWT clear) —
|
||||
with an uncacheable-still-uncacheable regression guard. Chosen over the original
|
||||
screenshot-of-a-fill gate because it proves the *actual* WC property headlessly; the
|
||||
visible fill folds into D2's gate (the service clears the screen through the back buffer).
|
||||
Regression-checked: `discovery`, `ioport`, `claim-release`, `supervision`, `device-list`,
|
||||
`device-manager` all still pass with the +1 device in the table.
|
||||
|
||||
## D2 — Service skeleton, protocol, runtime module ✅
|
||||
|
||||
Stand up the named service and the double-buffer, no layers yet.
|
||||
|
||||
- [x] `system/services/display/protocol.zig`: `Operation{ info, create_layer,
|
||||
configure_layer, destroy_layer, fill_rect, blit_tile, damage, present }`; `extern`
|
||||
`Request`/`Reply`; size + `maximum_payload` consts. (Model: block/protocol.zig.)
|
||||
- [x] [abi.zig](../system/abi.zig): `ServiceId.display = 9`.
|
||||
- [x] `system/services/display/display.zig`: `main` → enumerate + claim + WC-map the LFB
|
||||
(front) → `mmap` a cacheable back buffer of `height*pitch` → `runtime.service.run`.
|
||||
`info` and a whole-screen `present` (back → front) are live; layer ops fail-stub
|
||||
until D3. Init clears the back buffer and presents it — the double-buffer path.
|
||||
- [x] [library/runtime/display.zig](../library/runtime/runtime.zig) (+ barrel export of
|
||||
`display` and `display_protocol`): `info()` and `present()`, cached `.display`
|
||||
lookup with retry (model: block.zig).
|
||||
- [x] [init.zig](../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||
- [x] [build.zig](../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||
via `addUserBinary`; packed into the initial-ramdisk; installed to
|
||||
`/system/services/display`.
|
||||
- [x] **Kernel fix the back buffer surfaced:** `mmap` was capped at 256 pages (1 MiB) by
|
||||
a fixed kernel-stack `frames` array. Rewrote `systemMmap` to map page-by-page with
|
||||
rollback (no scratch array) and raised the cap to 8192 pages (32 MiB) — enough for a
|
||||
4K back buffer. A real limitation met, exactly the kind this project chases.
|
||||
|
||||
**Gate (met, automated):** `python3 test/qemu_test.py display-service` spawns the
|
||||
compositor and matches its own serial heartbeats — `display: online {w}x{h} pitch …`
|
||||
followed by `display: presented frame 0` — which it prints only after the whole
|
||||
claim → WC-map → back-buffer → clear → present chain succeeds (matched on serial like the
|
||||
fault cases, since a lone blocking service can't reschedule the in-kernel test context to
|
||||
poll). Regression-checked: `usermem`, `heap` (the `mmap` rewrite), `init` (the boot-list
|
||||
addition), and D1's `display` all still pass.
|
||||
|
||||
## D3 — Layer stack + compositor + damage present ✅
|
||||
|
||||
The heart: composite an ordered layer stack, present only what changed.
|
||||
|
||||
- [x] A layer table (16 slots): each `Layer` = position, z, visible, a server-owned
|
||||
`mmap`'d surface (freed on `destroy_layer`). `damage` accumulates the dirty screen
|
||||
region since the last present.
|
||||
- [x] `create_layer` / `configure_layer` (damages old + new footprints) / `destroy_layer`,
|
||||
`fill_rect`, `blit_tile` (reads the inline tile from the IPC payload, unaligned-safe),
|
||||
`damage`, `present`.
|
||||
- [x] Pure, host-tested [compositor.zig](../system/services/display/compositor.zig): `Rect`
|
||||
(intersect/unite), `Surface`, `fillRect`, `composite` (opaque, clipped to a damage
|
||||
rect), `blitTile`. `present` clears the damaged region to the wallpaper, paints the
|
||||
visible layers bottom-to-top (z-sorted), and flushes just that rect back → front (WC).
|
||||
Colour packing (rgbx/bgrx) is `protocol.pack`, also host-tested.
|
||||
- [x] Host tests (`zig build test`, green): rect intersect/unite, `fillRect` clipping +
|
||||
`stride > width` padding, `composite` overlap-shows-top + damage clipping, `blitTile`
|
||||
unaligned read + clipping, and `pack` for both pixel formats.
|
||||
|
||||
**Gate (met):** `zig build test` green for the compositor + pack unit tests, **and** the
|
||||
`display-service` case's startup self-check composites two overlapping layers on the real
|
||||
framebuffer and reads back the composited pixels — overlap = top layer, outside = bottom
|
||||
layer — logging `display: compositor self-check ok` (matched by the harness).
|
||||
|
||||
## D4 — Client API + the demo client ✅
|
||||
|
||||
Prove the pipeline end-to-end from a separate process.
|
||||
|
||||
- [x] Finished [runtime/display.zig](../library/runtime/runtime.zig): a `Layer` handle with
|
||||
`fill` / `blitTile` (inline tile) / `configure` (move/restack/show) / `damage` /
|
||||
`destroy`, `createLayer`, and a `color(r,g,b)` helper (caches the mode, packs via
|
||||
`protocol.pack`). Coordinates are signed over the wire (`@bitCast` both ways).
|
||||
- [x] `system/services/display-demo/`: a hardware-free client (the `input-source` analog)
|
||||
— a full-screen wallpaper layer, a rectangle that slides back and forth (moved by
|
||||
`configure` each frame, so the compositor repaints old + new), and a cursor layer;
|
||||
presents in a loop paced by `runtime.time`. Wired into build + initial-ramdisk.
|
||||
- [x] **Bug this surfaced:** `protocol.message_maximum` was 4096, but the kernel caps
|
||||
every IPC message at `MESSAGE_MAXIMUM` = 256 — so `replyWait` rejected the oversized
|
||||
receive buffer with `-E2BIG` and the serve loop had been *spinning* since D2 (unseen,
|
||||
as D2/D3 matched init-time heartbeats). Set it to 256; `blit_tile` is now explicitly
|
||||
a small-tile path (≤ 54 px inline), larger bitmaps being the deferred shared-memory surface.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display-demo` spawns the service + `display-demo`;
|
||||
the demo drives a run of frames of motion through the layer client API and logs
|
||||
`display-demo: ok` (the visible motion is a screenshot via `zig build run-x86-64`).
|
||||
Regression-checked: `zig build test`, `display` (D1), and `display-service` (D2/D3) all
|
||||
still pass, and the default `zig build` is clean.
|
||||
|
||||
## D5 — Test cases + docs ✅
|
||||
|
||||
- [x] The three integration cases exist and pass: `display` (D1 handoff, kernel),
|
||||
`display-service` (D2/D3 compositor + self-check), and `display-demo` (D4 full
|
||||
pipeline: spawn `display` + `display-demo`, match `display-demo: ok`) —
|
||||
[tests.zig](../system/kernel/tests.zig) + [qemu_test.py](../test/qemu_test.py). Plus
|
||||
the pure host tests (`zig build test`).
|
||||
- [x] [display.md](display.md) updated to the built state (the "Verifying it" section names
|
||||
the real cases); [README index](README.md) entry present (#19); the `display-track`
|
||||
memory marked DONE with the commits.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display display-service display-demo` all pass,
|
||||
`zig build test` is green, and the default `zig build` is clean.
|
||||
|
||||
---
|
||||
|
||||
## v1 status: complete
|
||||
|
||||
D1–D5 done. The display service is a working framebuffer compositor: it owns the
|
||||
framebuffer (write-combining), composites a z-ordered layer stack into a cacheable back
|
||||
buffer, presents only the damaged region, and is driven over IPC by the `runtime.display`
|
||||
client — proven end-to-end by a separate demo process. Two limitations are deliberate and
|
||||
documented (docs/display.md): no runtime mode-setting (native backend) and no true vsync
|
||||
(no vblank on a dumb framebuffer). Next steps are the Deferred items below.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Shared-memory surfaces** — generalize M13 capability passing to memory objects
|
||||
(`shared_memory_create`/`shared_memory_map`), so bitmap clients hand the compositor a rendered surface
|
||||
instead of drawing commands. The compositor's layer model already anticipates it.
|
||||
- **Native backend (Bochs DISPI, then virtio-gpu)** — behind the same internal backend
|
||||
interface as the dumb framebuffer: EDID mode list + runtime resolution/bpp change +
|
||||
(eventually) a vblank/flip path for true vsync.
|
||||
- **Driver/compositor process split** — only when a second backend or a second head makes
|
||||
the abstraction pay for itself.
|
||||
@@ -0,0 +1,175 @@
|
||||
# Display v2 — build plan (pluggable scanout: GOP floor + virtio-gpu native)
|
||||
|
||||
The ordered, checkpointable build-out for [display-v2.md](display-v2.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-plan.md](display-plan.md). Read display-v2.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **First native backend = virtio-gpu** (VM standard: mode-set + fenced present/flush).
|
||||
- **Dynamic hot-attach**: boot on GOP, upgrade to native when the driver **announces**
|
||||
(push, not polling); re-attach across driver restarts; GOP is the floor for "no driver
|
||||
ever," not a live fall-back after a reprogram.
|
||||
- **v2 builds the shared-memory capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
- The compositor's layers/back-buffer/damage are **unchanged**; only scanout is pluggable.
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||
[abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
**Every gate is serial-checkable — no screenshots** (this plan is built to run unattended).
|
||||
Where "does it actually display" would otherwise need a human eyeball, the code **reads its
|
||||
own pixels back**: the scanout resource is CPU-visible RAM (shared-memory-backed) and the back buffer
|
||||
is cacheable, so a driver/compositor can write a known value, read it back, and log a
|
||||
pass/fail — and a virtio `resource_flush` is confirmed by the device **acking it on the
|
||||
used ring**. Those two together (pixel-readback + flush-ack) are the automated stand-in for
|
||||
"it's on screen."
|
||||
|
||||
- `zig build test` — host unit tests (backend selection, virtio struct sizes/encodings,
|
||||
pixel-check helpers).
|
||||
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial markers.
|
||||
The virtio cases boot with `-device virtio-gpu` (a per-case `qemu_extra`).
|
||||
- `run-x86-64` renders to a window — for the human's own satisfaction, **not** a gate.
|
||||
|
||||
---
|
||||
|
||||
## V1 — The scanout backend seam (refactor, no behaviour change) ✅
|
||||
|
||||
Extract scanout from the compositor so today's path becomes one backend among future ones.
|
||||
|
||||
- [x] `system/services/display/backend.zig`: a `Backend` tagged union with `info()`,
|
||||
`surface()` (the cacheable compose target), `present(damage)`, and capability flags
|
||||
(`canModeSet`/`hasFencedPresent`, both false for GOP).
|
||||
- [x] The v1 GOP path is now `backend.Gop` (claims the `display` node, WC-maps the LFB,
|
||||
keeps the cacheable back buffer, `present` = the damage-rect WC copy). display.zig
|
||||
composes into `backend.surface()` and calls `backend.present(damage)` — no LFB or
|
||||
framebuffer geometry left in the compositor core.
|
||||
- [x] The selection decision is the pure `chooseKind(native_available)` (gop unless a
|
||||
native driver announced), split from the syscall-bound `select()`/`Gop.init()`.
|
||||
|
||||
**Gate (met):** `display-service` + `display-demo` pass **unchanged** (pure refactor; GOP
|
||||
is the only backend), and `zig build test` stays green.
|
||||
|
||||
## V2 — The shared-memory cross-process capability (kernel) ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `shared_memory_create` (34) / `shared_memory_map` (35) syscalls + a
|
||||
`shared_memory_test` service id. Handlers in process.zig: `shared_memory_create(len)` allocates contiguous,
|
||||
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||
handle, maps them into the caller's shared-memory arena → returns virtual_address + handle; `shared_memory_map(cap)`
|
||||
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
||||
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
||||
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
||||
dispatch by kind, so a `SharedMemoryObject` rides an `ipc_call` `send_cap` exactly like an
|
||||
endpoint and frees only when its last capability drops. `mapUserSharedInto` (paging)
|
||||
maps WB-cacheable + `device_grant`, so a sharer's teardown never frees the shared
|
||||
frames — the object owns them.
|
||||
- [x] `library/runtime/shared-memory.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||
`map(handle) -> ptr`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py shared-memory` — `shared-memory-client` creates a region, writes a
|
||||
pattern, and passes its capability to `shared-memory-server` as an `ipc_call` send_cap; the server
|
||||
`shared_memory_map`s it and reads the **same bytes** back → `shared-memory: shared 4096 bytes ok`. Guardrail:
|
||||
`ipc`/`ipc-call`/`ipc-cap`, `supervision`, `dma`, `usermem`, `display-service`, and host
|
||||
tests all still pass — the handle-table change broke no existing IPC.
|
||||
|
||||
## V3 — The virtio-gpu driver: bring-up + a frame on screen ✅
|
||||
|
||||
- [x] `system/drivers/virtio-gpu/`: claim the virtio-gpu PCI function (device-manager
|
||||
match on the display/other class triple, driver self-confirms vendor 0x1AF4/device
|
||||
0x1050 from config space), enable memory-space + bus-master, walk the vendor
|
||||
capabilities in config space to find common-config + notify, map the BAR, negotiate
|
||||
VERSION_1, and stand up the control virtqueue in coherent DMA. `virtio-gpu-protocol.zig`
|
||||
+ `virtio-pci.zig` for the control/transport structs (host-tested sizes).
|
||||
- [x] Create a 2D scanout resource backed by a coherent DMA region (V4 swaps this for the
|
||||
shared-memory surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||
+ `resource_flush` of a test pattern, and wait on the used ring.
|
||||
- [x] Register a `scanout` service (`ServiceId.scanout` = 11).
|
||||
|
||||
**Gate (met):** the `virtio-gpu` case (QEMU `-device virtio-gpu-pci`) boots the
|
||||
device-manager stack, which discovers the function and spawns the driver; the driver writes
|
||||
a known test pattern into the scanout backing, `transfer_to_host_2d` + `resource_flush`es
|
||||
it, and **waits for the device's used-ring ack**, then reads the backing back and checks the
|
||||
pattern — logging `virtio-gpu: scanout 640x480 online` and `virtio-gpu: flush acked, pixel
|
||||
check ok`. That proves virtqueue + resource + attach + set_scanout + transfer + flush end to
|
||||
end without a screenshot (the used-ring ack is the device confirming it consumed the frame).
|
||||
|
||||
## V4 — The native backend + hot-attach ✅
|
||||
|
||||
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared-memory scanout surface
|
||||
(the compositor composes straight into the device's resource backing; x86 DMA is
|
||||
coherent, so the cacheable shared pages need no flush), `present(damage)` = a `present`
|
||||
request over the driver's `.scanout` endpoint (→ transfer-to-host + resource flush).
|
||||
- [x] The driver **announces** to `.display` after bring-up (looks it up with a bounded retry,
|
||||
sends `attach_scanout` with the geometry + the shared surface as an `ipc_call` send_cap).
|
||||
The compositor maps it, looks up `.scanout` itself (no need to pass the endpoint — the
|
||||
driver registered it), switches backend, and re-composites the current frame full-screen.
|
||||
The present is deferred to a one-shot timer so it runs *after* the reply unblocks the
|
||||
driver and it serves `.scanout` — presenting inline would deadlock.
|
||||
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shared_memory_physical` (a
|
||||
new syscall) gives the driver the guest-physical of the shared surface for `attach_backing`.
|
||||
|
||||
**Gate (met):** the `display-native` case (QEMU `-device virtio-gpu-pci`, `mem` bumped since it
|
||||
boots the whole system) starts the compositor + `display-demo` + device-manager; the driver
|
||||
announces, the compositor logs `display: scanout upgraded to virtio-gpu`, drives frames through
|
||||
the native backend, and **reads a pixel back** from the shared surface after a present to
|
||||
confirm the composited frame landed (`display: native present verified`), while `display-demo:
|
||||
ok` still fires — checked order-independently. Without `-device virtio-gpu-pci` nothing is
|
||||
announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pass unchanged.
|
||||
|
||||
## V5 — Mode-setting, EDID, and fenced presents ✅
|
||||
|
||||
- [x] The driver negotiates `VIRTIO_GPU_F_EDID` (when offered) and reads the monitor's EDID,
|
||||
logging its preferred mode; it offers a small mode list over `.scanout` `get_modes`. The
|
||||
resource + shared surface are sized to the largest mode, so `set_mode` just re-points the
|
||||
scanout rectangle (no resource/surface churn) — a runtime resolution change. `runtime.display`
|
||||
gains `modes()` / `setMode()` (display-protocol `get_modes`/`set_mode`, forwarded to the backend).
|
||||
- [x] Every `resource_flush` is issued fenced (`VIRTIO_GPU_FLAG_FENCE`); the device signals the
|
||||
fence when it has consumed the frame, which the used-ring ack the synchronous present waits
|
||||
on already gates — a tear-free present. (Completion feedback, **not vblank**: base
|
||||
virtio-gpu 2D has no display-refresh event, so nothing paces presents to the monitor —
|
||||
see the "Fenced is not vsync" note in [display-v2.md](display-v2.md).)
|
||||
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasFencedPresent` = true.
|
||||
|
||||
**Gate (met):** the `display-modeset` case (reusing the display-native boot) upgrades to
|
||||
virtio-gpu, queries the driver's modes, `setMode`s to a different resolution, and confirms the
|
||||
change by reading the backend's geometry back (`display: mode set to {w}x{h}, verified`); the
|
||||
fenced present path is exercised and confirmed (`display: fenced present ok`) — both from serial,
|
||||
passing 3/3. The driver also logs the EDID preferred mode (`virtio-gpu: EDID preferred mode …`).
|
||||
|
||||
## V6 — Resilience (restart + re-attach) + tests + docs ✅
|
||||
|
||||
- [x] The virtio-gpu driver now **hellos** the device manager (role: bus) so it is properly
|
||||
supervised — no longer stopped at the hello deadline — and is restarted on death. On
|
||||
driver loss the compositor keeps the last frame (its `.scanout` calls now return
|
||||
`-EPEER` instead of hanging — a kernel fix: an endpoint is marked dead when its owner
|
||||
dies) and **re-attaches** when the restarted driver re-announces. A permanent give-up
|
||||
(crash-loop cap) leaves the frozen frame; GOP is not re-taken.
|
||||
- [x] `test/qemu_test.py`: the `virtio-gpu`, `display-native` (hot-attach), `display-modeset`,
|
||||
and `display-reattach` (driver-kill/re-attach) cases. display-v2.md status updated.
|
||||
|
||||
**Gate (met):** the `display-reattach` case — device-manager (in `test-scanout-restart` mode)
|
||||
kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
|
||||
re-announces, and the compositor logs `display: scanout re-attached` after the initial
|
||||
`display: scanout upgraded to virtio-gpu`, with no CPU exception / panic (the compositor
|
||||
survives) — passing 3/3. All v1 + v2 cases (host tests, `ipc`/`ipc-call`/`ipc-cap`,
|
||||
`supervision`, `shared-memory`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||
`display-modeset`) pass; default `zig build` is clean.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Client-rendered surfaces** — now unblocked by the shared-memory capability (V2): an app renders
|
||||
its own bitmap and hands the compositor a reference. A natural follow-on.
|
||||
- **Bochs DISPI backend** — a simpler second native backend (mode-set only, dumb scanout);
|
||||
slots behind the same interface if wanted.
|
||||
- **Real-GPU (NVIDIA/AMD/Intel) drivers** — out of scope; those devices stay on the GOP
|
||||
floor by design.
|
||||
- **Hardware-accelerated compositing / multiple heads** — future.
|
||||
@@ -0,0 +1,139 @@
|
||||
# The display service v2: a pluggable scanout backend
|
||||
|
||||
**Status: complete (V1–V6).** The compositor boots on the GOP framebuffer and, when a
|
||||
virtio-gpu driver announces itself, hot-attaches a native backend over the shared-memory
|
||||
scanout surface — with runtime mode-setting, EDID, and fenced presents, and it
|
||||
re-attaches across driver restarts. All serial-gated (see [display-v2-plan.md](display-v2-plan.md)).
|
||||
|
||||
v1 ([display.md](display.md)) is a compositor that owns the **GOP framebuffer** — it
|
||||
composites a layer stack into a cacheable back buffer and streams damage to the linear
|
||||
framebuffer the firmware handed over. That path is portable and good: it drives any GPU,
|
||||
including a real NVIDIA card at an ultrawide's native resolution, with zero GPU-specific
|
||||
code. v2 keeps it as the **floor** and makes *scanout* — how a finished frame reaches the
|
||||
panel — a **pluggable backend**, so the compositor can **upgrade to a real GPU driver when
|
||||
one is present** and fall back to the framebuffer when it isn't.
|
||||
|
||||
The compositor itself (layers, back buffer, damage) does not change. Only the last step —
|
||||
"put this frame on screen" — becomes swappable.
|
||||
|
||||
## The shape
|
||||
|
||||
```
|
||||
compositor (display service) ── layer stack + back buffer + damage (unchanged)
|
||||
│ composites a frame, then: backend.present(damage)
|
||||
▼
|
||||
scanout backend (selected at runtime — GOP by default, native when it appears)
|
||||
│
|
||||
├─ GopBackend the v1 path: WC copy back→front to the firmware LFB.
|
||||
│ Always available. No mode-set, no present fence. THE FLOOR.
|
||||
│
|
||||
└─ VirtioGpuBackend talks to a virtio-gpu driver process over a `scanout`
|
||||
service: present via a shared resource + fenced flush,
|
||||
EDID mode list, runtime mode-set.
|
||||
```
|
||||
|
||||
A **backend** is a small interface the compositor calls:
|
||||
|
||||
- `surface()` → the pixels to compose into and their geometry `{ptr, pitch, format, w, h}`
|
||||
(the LFB for GOP; a shared scanout resource for virtio-gpu),
|
||||
- `present(damage: Rect)` → make the damaged region visible (a no-op-ish WC copy for GOP;
|
||||
a fenced virtio flush for the native path),
|
||||
- capability queries — `canModeSet`, `hasFencedPresent` — and, when supported, `modes()` /
|
||||
`setMode(m)`.
|
||||
|
||||
The compositor composes into `surface()` and calls `present(damage)` exactly as it does
|
||||
today; everything device-specific lives behind the interface.
|
||||
|
||||
## Selection and hot-attach
|
||||
|
||||
The choice is **dynamic**, because a GPU driver is spawned asynchronously (the device
|
||||
manager brings it up after boot), and because danos is meant to be resilient:
|
||||
|
||||
1. **Boot on GOP.** The compositor starts on `GopBackend` immediately, so there is never a
|
||||
blank screen while drivers load — the exact v1 behaviour.
|
||||
2. **Upgrade on announce.** When the virtio-gpu driver has claimed its device and set up a
|
||||
scanout, it **announces itself to the display service** (a `push`: the driver looks up
|
||||
`.display` and sends an *attach-scanout* message carrying its `scanout` endpoint as a
|
||||
capability). The compositor switches to `VirtioGpuBackend` and re-presents the current
|
||||
frame full-screen. Push beats polling — the compositor doesn't know a priori which
|
||||
driver, if any, exists, and danos has no service-registration pub/sub.
|
||||
3. **Native is restartable, not fallback-on-crash.** Once a native driver has reprogrammed
|
||||
the device, the firmware's GOP framebuffer is **stale** — "native → GOP" is not a clean
|
||||
fall-back. So a native driver that **crashes** is *restarted* by its supervisor (the
|
||||
resilience work already merged), re-announces, and the compositor **re-attaches**
|
||||
(native → native). The screen freezes on the last frame during the gap — acceptable.
|
||||
4. **GOP is the floor for "no driver was ever there."** On a real GPU (NVIDIA/AMD/Intel)
|
||||
the class-0x03 device matches nothing in the driver table, no `scanout` is ever
|
||||
announced, and the compositor stays on GOP forever — no special-casing. Only if a
|
||||
native driver *permanently* gives up (crash-loop cap) does the compositor attempt GOP
|
||||
again, and even then only if the LFB is still mappable.
|
||||
|
||||
## The shared-memory primitive this needs
|
||||
|
||||
virtio-gpu's scanout resource is **guest RAM** — the driver allocates it and attaches it
|
||||
to a virtio resource, and the compositor composes into it. That means the compositor
|
||||
writing into the driver's buffer is **cross-process memory sharing**, the primitive v1
|
||||
deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural generalization
|
||||
of M13 capability-passing from *endpoints* to *memory objects* —
|
||||
|
||||
```
|
||||
shared_memory_create(len) -> {handle, virtual_address} // a shareable, page-aligned RAM region
|
||||
… pass `handle` as the send_cap on an ipc_call …
|
||||
shared_memory_map(cap) -> virtual_address // the receiver maps the same physical pages
|
||||
```
|
||||
|
||||
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
||||
client-rendered surfaces (an app composing its own bitmap and handing the compositor a
|
||||
reference instead of drawing by command). One piece of kernel work, two features.
|
||||
|
||||
## The virtio-gpu driver
|
||||
|
||||
A new ring-3 driver process (the topology v1 anticipated — "split the driver from the
|
||||
compositor when a second backend arrives"). It claims the virtio-gpu PCI function, and:
|
||||
|
||||
- sets up the **virtqueues** (control + cursor) and the device's config space,
|
||||
- creates a **2D scanout resource** backed by a shared-memory region, `attach_backing`s it,
|
||||
`set_scanout`s it to a CRTC, and `resource_flush`es damaged rectangles,
|
||||
- reads **EDID** (the `GET_EDID` control command) for the mode list, and `set_scanout`
|
||||
at a chosen mode for **runtime mode-setting**,
|
||||
- registers a `scanout` service and announces to the display service.
|
||||
|
||||
Its `resource_flush` is the real **present** — and gives a **fenced, tear-free** path a
|
||||
dumb GOP framebuffer can't.
|
||||
|
||||
**Fenced is not vsync.** The fence completes when the device has *consumed* the frame:
|
||||
real completion feedback, and tear-freedom by snapshot semantics (the host displays
|
||||
discrete transferred frames, never a half-written surface). It is **not** a vblank —
|
||||
base virtio-gpu 2D has no display-refresh event at all (Linux's driver for this device
|
||||
fakes one with a software timer), so nothing paces presents to the monitor's refresh.
|
||||
Refresh-paced presents need either a native driver's vblank interrupt (delivered over
|
||||
the existing IRQ-as-IPC path) or the compositor's own frame clock.
|
||||
|
||||
## What v2 unlocks — and its honest scope
|
||||
|
||||
Behind the abstraction, a native backend gives runtime **mode-setting** (resolution /
|
||||
refresh / bpp), **EDID** enumeration, and **fenced presents**. But only on devices we have a driver
|
||||
for — realistically **VMs** (virtio-gpu, and later maybe Bochs DISPI). Real discrete GPUs
|
||||
need per-vendor KMS-class drivers that aren't getting written, so they **stay on GOP** —
|
||||
which is genuinely fine (v1 on the NVIDIA box is smooth). So v2's real value is twofold:
|
||||
the **pluggable architecture** (a driver slots in when one exists) and a **rich, fenced
|
||||
path in VMs**, where danos development happens. The framebuffer floor never goes away.
|
||||
|
||||
## Locked decisions
|
||||
|
||||
- **First native backend: virtio-gpu** — the VM standard; gives mode-set + a real
|
||||
present/flush (fenced), and exercises the whole pluggable design. Tested with QEMU
|
||||
`-device virtio-gpu`.
|
||||
- **Dynamic hot-attach** — boot on GOP, upgrade to native on the driver's announce,
|
||||
re-attach across driver restarts; GOP is the floor for "no driver ever," not a live
|
||||
fall-back after a reprogram.
|
||||
- **Detection = push** (the driver announces to `.display`), not compositor polling.
|
||||
- **v2 builds the shared-memory capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
|
||||
## See also
|
||||
|
||||
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||
- [resilience.md](resilience.md) — the restart machinery the hot-attach leans on.
|
||||
+313
@@ -0,0 +1,313 @@
|
||||
# The display service: a framebuffer compositor
|
||||
|
||||
The [framebuffer](framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../system/kernel/console.zig) draws text
|
||||
into it directly. That console is a stop-gap. The **display service**
|
||||
(`system/services/display/`) is the real thing: an ordinary ring-3 process that *owns*
|
||||
the framebuffer, composes a stack of **layers** into an off-screen back buffer, and
|
||||
**presents** finished frames to the screen — the display half of the GUI track
|
||||
([vision.md](vision.md)), the sibling of the [input service](input.md).
|
||||
|
||||
This note is the architecture and the reasoning behind it. The concrete build order
|
||||
lives in [display-plan.md](display-plan.md).
|
||||
|
||||
## First, a distinction that shapes everything: GOP vs. the PCI device
|
||||
|
||||
It is tempting to think "the GOP framebuffer" and "the VGA-compatible display
|
||||
controller in the PCIe tree" are two different things. They are not — they are **two
|
||||
interfaces to the same silicon, at different times and different levels**, and knowing
|
||||
which one you're holding decides what you can do.
|
||||
|
||||
- **GOP is firmware's *temporary* driver** for the display controller. It gives you a
|
||||
linear framebuffer pointer and can set video modes — but only until
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../boot/efi.zig)
|
||||
reads the monitor's EDID, picks the native mode, and calls `set_mode` **before**
|
||||
exiting ([gop.md](gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
mode list, no EDID. What survives is the frozen snapshot in
|
||||
[`BootInformation.framebuffer`](../system/boot-handoff.zig): `{base, width, height,
|
||||
pitch, format}`, and nothing more.
|
||||
|
||||
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||
([`-device VGA,edid=on`](../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||
VRAM BAR. danos already decodes this device
|
||||
([pci-class.zig](../system/devices/pci-class.zig) has the full `display` namespace, and
|
||||
`pci-bus` already reports it to the [device manager](device-manager.md) with its class
|
||||
triple) — but nothing binds it yet.
|
||||
|
||||
What that difference costs you, concretely:
|
||||
|
||||
| You want to… | Dumb GOP framebuffer (boot handoff) | Native device driver (PCI 0x03) |
|
||||
|-------------------------------------------|-------------------------------------|------------------------------------------|
|
||||
| **Report** the current mode | ✅ from the handoff | ✅ |
|
||||
| **Change resolution / bpp at runtime** | ❌ GOP is gone | ✅ program DISPI regs / virtio-gpu queue |
|
||||
| **Re-read EDID, enumerate monitor modes** | ❌ | ✅ the device exposes an EDID block |
|
||||
| **Refresh rate** | ❌ (virtual anyway) | only a real KMS driver — far future |
|
||||
| **vblank / tear-free present** | ❌ no vblank signal | ✅ vblank IRQ + page-flip (real GPUs) |
|
||||
| **Works on the Pi (no PCI VGA)** | ✅ VideoCore hands a simple FB | ✗ per-device |
|
||||
|
||||
The lesson: the **portable base for the whole GUI stack is the GOP / boot-handoff linear
|
||||
framebuffer**. Runtime mode-setting is a *per-device upgrade* layered on top — and on
|
||||
the Raspberry Pis there is no PCI VGA at all, so the neutral framebuffer is the only
|
||||
thing all three target machines share. That is why the display service is built on the
|
||||
dumb framebuffer first, with the native backend as an optional module behind the same
|
||||
interface.
|
||||
|
||||
## Two constraints this service exists to meet
|
||||
|
||||
Like the input service — which existed partly to motivate the asynchronous
|
||||
[`ipc_send`](ipc.md) primitive — the display service runs straight into two limits the
|
||||
rest of the system hasn't had to face:
|
||||
|
||||
1. **The framebuffer is kernel-only today.** It arrives through the boot handoff, is
|
||||
mapped into the kernel's physmap, and is touched only by
|
||||
[`console.zig`](../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../system/kernel/devices-broker.zig) node, so
|
||||
`device.claim`/`mmio_map` cannot reach it, and there is no framebuffer
|
||||
[syscall](syscall.md). A user-space display service needs a **new mechanism just to
|
||||
touch the pixels**. (See "The handoff" below — this is built.)
|
||||
|
||||
2. **danos has no cross-process shared memory.** The memory syscalls are `mmap`
|
||||
(private, zeroed), `mmio_map` (a *claimed device's* MMIO), and `dma_alloc` (new
|
||||
pinned physical). The block driver's "pass a buffer by physical address" trick
|
||||
([block/protocol.zig](../system/services/block/protocol.zig)) works *only because its
|
||||
consumer is DMA hardware*. A compositor that CPU-reads and blends client layers can't
|
||||
use it — it would have to *map* another process's memory, which nothing allows. This
|
||||
is deferred (see "What v1 does not do"), because v1 sidesteps it entirely.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
kernel ── owns the boot framebuffer; bootstrap console only
|
||||
│ seeds a "display0" device node from BootInformation.framebuffer
|
||||
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
||||
│ plus DisplayInfo{width, height, pitch, format})
|
||||
▼
|
||||
display service (system/services/display/, ServiceId.display) ← the compositor
|
||||
│ device.claim(display0) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||
│ mmap(cacheable) a BACK buffer of the same geometry
|
||||
│ owns: an ordered LAYER STACK + a per-frame DAMAGE list
|
||||
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
||||
│ backend is an INTERNAL interface: {gop-fb} today; {bochs-dispi, virtio-gpu} later
|
||||
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
||||
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||
drawing clients (v1) surface clients (deferred)
|
||||
runtime.display commands: runtime.display surfaces:
|
||||
create_layer / configure_layer shared_memory_create → pass as a capability →
|
||||
fill_rect / blit_tile / damage the compositor maps & composites the
|
||||
present client-rendered bitmap directly
|
||||
```
|
||||
|
||||
The bring-up sequence mirrors a hardware driver's — it is the
|
||||
[`usb-xhci-bus` `initialise`](../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
(claim → `mmio_map` → run loop) — and the request/reply service shell is the
|
||||
[FAT](../system/services/fat/fat.zig) / [input](../system/services/input/input.zig) shape
|
||||
([`runtime.service.run`](../library/runtime/service.zig) with a `protocol.zig` of
|
||||
`extern struct` messages and an `Operation` tag).
|
||||
|
||||
**One process, for now.** v1 is a *single* service that both owns the framebuffer and
|
||||
composites — it does not split a "framebuffer driver" from a "compositor" the way input
|
||||
splits `ps2-bus` from the input service. The backend (dumb FB vs. a native GPU) is an
|
||||
*internal* interface, not a process boundary. That boundary earns its keep only when a
|
||||
second backend or a second monitor appears; until then it is complexity with no payoff.
|
||||
|
||||
## The handoff: a device node + a write-combining map
|
||||
|
||||
The framebuffer crosses into user space through the machinery that already exists for
|
||||
every other device, rather than a bespoke syscall — so it inherits ownership,
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](resilience.md)
|
||||
story: a crashed display service returns the LFB to the kernel, and its restart
|
||||
re-claims it).
|
||||
|
||||
- The kernel seeds a synthetic **`display0`** node into the
|
||||
[devices-broker](../system/kernel/devices-broker.zig) at init, from
|
||||
`BootInformation.framebuffer`: one `ResourceKind.memory` resource spanning
|
||||
`[base, height*pitch]`, tagged **write-combining**, plus a small
|
||||
`DisplayInfo{width, height, pitch, format}` (the memory resource says *where* and *how
|
||||
big*; `DisplayInfo` says how to *interpret* the bytes).
|
||||
- The service `device.claim`s it and `mmio_map`s the resource. The map is
|
||||
**write-combining**, not the strong-uncacheable that `mmio_map` uses for register
|
||||
MMIO. The kernel already programs a WC PAT slot for its own console
|
||||
([`setupPat`](../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||
the user mapping path. **This matters:** an uncacheable framebuffer makes the
|
||||
back→front blit unusably slow.
|
||||
- On `claim`, the kernel's bootstrap console goes quiet, so the two never fight over the
|
||||
LFB. A panic is the one exception — by then the service is likely dead anyway, and a
|
||||
panic on screen wins.
|
||||
|
||||
The display service is a **named boot service**: `init` spawns it by name alongside
|
||||
`vfs`/`input`/`device-manager` ([init.zig](../system/services/init/init.zig)), and it
|
||||
self-discovers `display0` with `device.enumerate`. The [device manager](device-manager.md)
|
||||
matching path (PCI class 0x03 → a driver) is reserved for the future *native* backend, not
|
||||
this singleton synthetic node.
|
||||
|
||||
## Double buffering and the write-combining discipline
|
||||
|
||||
Two buffers, with deliberately different memory types:
|
||||
|
||||
- The **front buffer** is the LFB — **write-combining**: fast to *write*, slow to
|
||||
*read*. The rule is therefore **never read the front buffer**. Only ever stream into
|
||||
it, sequentially.
|
||||
- The **back buffer** is ordinary **cacheable** RAM (`mmap`), the same geometry. All
|
||||
compositing happens here, where reads and read-modify-write blends are cheap.
|
||||
|
||||
So a frame is: compose every dirty layer into the cacheable back buffer, then **present**
|
||||
— copy the changed regions back→front in sequential, WC-friendly writes. Two details the
|
||||
[framebuffer](framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](gop.md).
|
||||
|
||||
## Flicker vs. tearing — what double buffering does and doesn't buy
|
||||
|
||||
These are two different artifacts, and the dumb framebuffer fixes exactly one of them:
|
||||
|
||||
- **Flicker** is the user seeing intermediate, half-drawn states (a clear-then-redraw
|
||||
flash). Double buffering **eliminates it completely** — the screen only ever receives
|
||||
whole, finished frames.
|
||||
- **Tearing** is a present landing while the display's scanout beam is mid-frame, so the
|
||||
top of the screen shows the new frame and the bottom the old. Avoiding it requires
|
||||
presenting during the vertical blank (**vsync**) — which needs a vblank signal. **A
|
||||
dumb GOP framebuffer has no vblank.**
|
||||
|
||||
So v1 is **flicker-free**, and it *minimizes* the tear window by presenting only damaged
|
||||
rectangles (less to copy → a smaller window in which the beam can catch a half-updated
|
||||
frame), but it is **not tear-free**. Genuine vsync waits for a backend with a vblank IRQ
|
||||
or a flush/flip path — a native-device capability, not something the firmware
|
||||
framebuffer can offer. Stated plainly here so the limitation is understood, not
|
||||
discovered.
|
||||
|
||||
## Layers and the client protocol
|
||||
|
||||
The compositor holds an **ordered stack of layers**. Each layer has a rectangle, a
|
||||
z-order, a visibility flag, and a surface. Presenting walks the stack bottom-to-top,
|
||||
painting each dirty layer into the back buffer, then flushes the damage to the front.
|
||||
|
||||
In v1 the surfaces are **server-owned**, and clients draw into them with a small
|
||||
immediate-mode command protocol — essentially the model early X used, and enough for a
|
||||
shell, a terminal, a cursor, and a wallpaper:
|
||||
|
||||
| Operation | Meaning |
|
||||
|--------------------|---------------------------------------------------------------|
|
||||
| `info` | report `{width, height, pitch, format}` of the display |
|
||||
| `create_layer` | allocate a server-owned surface, return a layer handle |
|
||||
| `configure_layer` | set a layer's rect, z-order, visibility |
|
||||
| `destroy_layer` | release a layer |
|
||||
| `fill_rect` | fill a rectangle of a layer with a colour |
|
||||
| `blit_tile` | copy a small client-supplied pixel tile into a layer (inline) |
|
||||
| `damage` | mark a region of a layer dirty |
|
||||
| `present` | request a repaint: composited at the next frame-clock tick |
|
||||
|
||||
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||
(the [PSF font](../system/kernel/font.psf) path the console already uses can move into a
|
||||
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||
policy in the client.
|
||||
|
||||
`present` is a *request*, not an immediate flush: the compositor runs a ~60 Hz **frame
|
||||
clock** (a one-shot kernel timer re-armed on demand), and each tick composites all the
|
||||
damage accumulated since the last one. Any number of client presents and cursor moves
|
||||
inside one interval coalesce into a single repaint — the software stand-in for vblank
|
||||
pacing on backends that have none (all of them today; see
|
||||
[display-v2.md](display-v2.md), "Fenced is not vsync"). Bring-up paths that must put
|
||||
pixels on screen synchronously (initialisation, the self-checks) bypass the clock.
|
||||
|
||||
## `runtime.display`
|
||||
|
||||
Clients speak the protocol through a new [`library/runtime/display.zig`](../library/runtime/runtime.zig),
|
||||
the [`runtime.block`](../library/runtime/block.zig) shape (a cached `.display` lookup
|
||||
with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitTile` /
|
||||
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||
runtime, as with every other danos service.
|
||||
|
||||
## The cursor: a mouse-listener thread feeding the compositor
|
||||
|
||||
The compositor is the single owner of the framebuffer — only the main `service.run` loop
|
||||
touches the backend and the layer stack. Tracking the mouse without breaking that
|
||||
ownership is the display's first use of [threads](threading.md): the service is built
|
||||
multi-threaded (`addThreadedUserBinary`) and, at startup, spawns a **mouse-listener
|
||||
thread** beside the compositor loop.
|
||||
|
||||
- **Listener thread.** Blocks on the input service's mouse stream
|
||||
(`input.subscribeMouse()`), accumulates the relative `dx`/`dy` motion into an absolute
|
||||
cursor position clamped to the screen, and hands it to the compositor. It never touches
|
||||
the compositor — so no lock guards the framebuffer. A parked `next()` leaves its core
|
||||
free to halt ([halting.md](halting.md)).
|
||||
- **The channel.** A single-slot *latest-value* cell (`CursorChannel`) guarded by a
|
||||
`runtime.Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
every delta, so a new position overwrites the old. The listener also **pokes** the
|
||||
compositor awake — the main loop is parked in `replyWait`, so the listener posts a
|
||||
zero-payload `ipc.send` to the compositor's endpoint, which arrives as a
|
||||
message-notification ([ipc.md](ipc.md)). The poke is *coalesced*: at most one is queued
|
||||
while the main loop has not drained the last, so a fast mouse cannot flood the endpoint.
|
||||
- **Render.** On the poke, the main loop takes the latest position and moves the cursor —
|
||||
which is just a top-z compositor layer — with the existing `configure` + `present` path
|
||||
(it damages the old and new footprints, so only those two rectangles repaint).
|
||||
|
||||
Two threading facts shape this (both in [threading.md](threading.md)). IPC **handles do
|
||||
not cross threads**, so the listener can't reuse the main loop's endpoint handle — it
|
||||
`ipc.lookup(.display)`s its *own* handle to the same endpoint to poke through. And a
|
||||
multi-threaded service doing concurrent IPC is why the kernel's endpoint-create / register
|
||||
/ lookup syscalls now serialize under the big kernel lock. Shared fate applies: a fault in
|
||||
the listener takes the whole display down, and the supervisor restarts the process
|
||||
([resilience.md](resilience.md)).
|
||||
|
||||
## What v1 does not do (and why that's fine)
|
||||
|
||||
Two capabilities are deliberately out of the first cut. Neither reshapes anything above;
|
||||
both are clean additions behind the interfaces v1 establishes.
|
||||
|
||||
- **Client-rendered surfaces (shared memory).** The fast path for a bitmap-heavy app is
|
||||
to render into its *own* buffer and hand the compositor a *reference*, not a stream of
|
||||
commands. That needs the missing cross-process shared-memory primitive — best built as
|
||||
the natural generalization of the existing M13 [capability passing](driver-model.md)
|
||||
from *endpoints* to *memory objects* (`shared_memory_create(len) → {cap, virtual_address}`, pass `cap` on
|
||||
an `ipc_call`, receiver `shared_memory_map(cap) → virtual_address`). v1 avoids it because server-owned
|
||||
surfaces already prove the whole pipeline.
|
||||
|
||||
- **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing
|
||||
resolution / bpp at runtime needs the raw PCI device. The first native backend is
|
||||
Bochs DISPI — the register interface QEMU's `-device VGA` exposes — behind the same
|
||||
internal backend interface the dumb framebuffer sits behind. Refresh-rate and colour
|
||||
management (a gamma LUT) are real-GPU-KMS territory, far beyond this.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Four QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||
test/qemu_test.py <case>`), each layering on the last:
|
||||
|
||||
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||
the claim → `mmio_map` leaf is genuinely **write-combining** (PAT entry 4), asserted at
|
||||
the page-table level.
|
||||
- **`display-service`** — the compositor comes up: it claims the framebuffer, allocates
|
||||
the cacheable back buffer, presents a cleared frame through the double-buffer path
|
||||
(`display: online … / presented frame 0`), and a startup **self-check** composites two
|
||||
overlapping layers on the real framebuffer and reads them back — overlap = the top
|
||||
layer — logging `display: compositor self-check ok`.
|
||||
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||
[`display-demo`](../system/services/display-demo/) client (the
|
||||
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper and
|
||||
a sliding rectangle — through the layer client API and heartbeats
|
||||
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||
the [input test](input.md) proves an event travels source → service → subscriber. It draws
|
||||
no cursor and reads no input — the cursor is the service's own (below), and the demo
|
||||
animates on its own frame timer, independent of the mouse (the test spawns `input`
|
||||
alongside it to keep that independence honest). The visible motion itself is a screenshot
|
||||
away via `zig build run-x86-64`.
|
||||
- **`display-cursor`** — the mouse-listener thread end to end: with the `input` service up,
|
||||
`input-source mouse` publishes pure motion, and the display's listener thread accumulates
|
||||
it into a cursor position handed to the render loop over the `CursorChannel`. Once the
|
||||
cursor has tracked a run of that motion, the service logs
|
||||
`display: cursor tracking mouse ok`. Runs `smp: 4` — the compositor and listener threads
|
||||
execute on different cores, which is what surfaced the IPC-under-lock requirement above.
|
||||
|
||||
The compositor's pixel math (rectangle clipping, fill, composite, tile blit) and colour
|
||||
packing are additionally covered by pure host unit tests under `zig build test`.
|
||||
|
||||
## See also
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [input.md](input.md) — the sibling service; the async `ipc_send` fan-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map`, capability passing, the trust model.
|
||||
- [device-manager.md](device-manager.md) — matching and supervision (the native backend's route).
|
||||
- [display-plan.md](display-plan.md) — the ordered build-out.
|
||||
@@ -240,8 +240,8 @@ once per page, maps writeback-cached, and never reveals a physical address.
|
||||
**The fix.**
|
||||
|
||||
```
|
||||
dma_alloc(len, flags) -> vaddr (rax), paddr (rdx)
|
||||
dma_free(vaddr, len) -> 0
|
||||
dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx)
|
||||
dma_free(virtual_address, len) -> 0
|
||||
|
||||
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||
|
||||
+1
-1
@@ -74,7 +74,7 @@ The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||
|---|------|---------|
|
||||
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||
| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window |
|
||||
| 13 | `mmio_map(id, res_idx) -> virtual_address` | Map a claimed device's register window |
|
||||
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||
|
||||
@@ -0,0 +1,556 @@
|
||||
# Native Intel iGPU display support — feasibility and roadmap
|
||||
|
||||
**Status: research snapshot, not implemented.** This records what a *minimal, display-only*
|
||||
native driver for an **Intel integrated GPU** — EDID read + mode-set + framebuffer scanout, with
|
||||
**no** 3D/media/compute — would take, and how it slots into danos's pluggable scanout
|
||||
architecture. It is a survey of primary sources (Intel's open-source
|
||||
[Programmer's Reference Manuals](https://www.intel.com/content/www/us/en/docs/graphics-for-linux/developer-reference/1-0/overview.html),
|
||||
coreboot's [libgfxinit](https://doc.coreboot.org/gfx/libgfxinit.html), the Linux
|
||||
[i915 display](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/i915/display) driver,
|
||||
and Haiku's [intel_extreme](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/intel_extreme/)),
|
||||
not an implementation. It is the companion to [nvidia-gpus.md](nvidia-gpus.md) and should be read
|
||||
against it — the two answer the same question for opposite silicon.
|
||||
|
||||
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the v2
|
||||
model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- **Intel is a materially easier, lower-tier target than the NVIDIA RTX 3060 — and the reason is
|
||||
documentation, not silicon.** Intel publishes official, register-level, per-platform **Display
|
||||
Engine** PRMs with named registers, bitfields, and numbered enable sequences; NVIDIA publishes
|
||||
no display PRM and forces reverse-engineering against GPL nouveau. A minimal Intel display-only
|
||||
driver is roughly **tier 2 to low-tier 3** for well-covered generations (Skylake / Kaby Lake /
|
||||
Coffee Lake), versus NVIDIA's **tier 4** for GA106. This is the load-bearing conclusion.
|
||||
- **The display block is a genuinely separable register domain.** Mode-set + scanout touch only
|
||||
display registers (pipes, planes, transcoders, DDI buffers, PLLs, power wells, GMBUS/AUX) — **no
|
||||
render engine, no command streamer, no GEM/3D, no signed microcode.** Two small carve-outs, both
|
||||
trivial pokes that do *not* pull in the render engine: a real CDCLK frequency change writes the
|
||||
shared GT PCODE mailbox, and the plane's surface register is a GGTT (memory-interface) address.
|
||||
- **There is no firmware wall on the display path.** The only display microcontroller (DMC / "CSR",
|
||||
Skylake+) is **optional** — its sole job is saving/restoring display state across DC5/DC6
|
||||
low-power idle. Without it, i915 prints "Disabling runtime power management" and mode-sets and
|
||||
scans out normally. GuC/HuC are render/media coprocessors, never touched by a display driver.
|
||||
Pre-Skylake parts have no display microcontroller at all yet mode-set fine. There is **nothing
|
||||
analogous to NVIDIA's GSP**.
|
||||
- **The scanout memory model is dramatically simpler than a discrete GPU.** Intel iGPUs have **no
|
||||
VRAM**: the display scans out of ordinary system RAM addressed through the Global GTT (GGTT), a
|
||||
flat single-level page table. Linear (untiled) framebuffers are first-class. You need **no
|
||||
GEM/TTM, no VMM, no VRAM allocator, no BAR1 aperture juggling** — the exact machinery the NVIDIA
|
||||
path forces on you.
|
||||
- **coreboot libgfxinit is a compact, complete, display-only reference** doing precisely this scope
|
||||
(EDID + PLL/mode-set + scanout, zero 3D) in ~22k lines of formally-analysed SPARK/Ada — versus
|
||||
i915's ~400k lines. It is a *read-and-reimplement* reference, not drop-in code (GPL-2.0-or-later,
|
||||
and Ada, not Zig).
|
||||
- **The clean-room, permissively-licensed path is real** — you can implement from the PRM without
|
||||
reading GPL code, and Haiku's MIT `intel_extreme` is a permissive precedent. This is the decisive
|
||||
contrast with NVIDIA, where no vendor register spec exists.
|
||||
- **The practical catch is hardware, not software.** On a desktop with an RTX 3060, the monitor is
|
||||
almost certainly cabled to the *card*, so an iGPU driver would light a dark motherboard port; the
|
||||
CPU may be an **F-SKU with the iGPU fused off entirely**; and every clean-room reference targets
|
||||
*older* Intel. Intel is the right target to **learn** display bring-up — "run it on my machine"
|
||||
is a separate, machine-dependent question that may not resolve in the reader's favour.
|
||||
- **Recommendation:** as with the NVIDIA doc, GOP already gives native-resolution scanout with zero
|
||||
GPU code. A native Intel driver buys runtime mode changes, hardware vsync, and multihead — and it
|
||||
reaches "first pixel" far faster than the NVIDIA path *if* the target machine actually has a
|
||||
usable, cable-attached iGPU of a documented generation.
|
||||
|
||||
## Display engine architecture, and why it's separable
|
||||
|
||||
For the common single-display path (SST DisplayPort / HDMI / eDP), the Intel display data flow is a
|
||||
small, fully documented, essentially fixed sequence:
|
||||
|
||||
```
|
||||
memory surface → PLANE(s) → PIPE → TRANSCODER → DDI (drives IO/PHY) → connector
|
||||
```
|
||||
|
||||
The Tiger Lake PRM Vol 12 states it verbatim: *"The front end of the display contains the pipes.
|
||||
The pipes connect to the transcoders. The transcoders, except for wireless, connect to the DDIs to
|
||||
drive the IO/PHY."* A **pipe** blends planes (primary/sprite/cursor) into one raster stream; the
|
||||
**transcoder** wraps it in port-protocol timing (DP/HDMI/eDP/DSI); the **DDI** is the physical port
|
||||
and PHY. Pipe, Planes, Transcoder, and Digital Display Interface are each first-class PRM chapters
|
||||
with per-object files in libgfxinit
|
||||
([TGL PRM Vol 12](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)).
|
||||
|
||||
**Two honest qualifications** the raw research overstated (per verification):
|
||||
|
||||
- The pipeline is *not* strictly linear in all cases — the same PRM pages document optional branches
|
||||
a minimal driver simply ignores (wireless writeback to memory, MIPI DSI, DisplayPort multistream
|
||||
many-to-one, DSC/tiled pipe-joining). Ignoring them does not weaken feasibility.
|
||||
- The four-object model *as named* is **Haswell-onward** (DDI introduced ~2013), not "every gen."
|
||||
Pre-Haswell used FDI + PCH transcoders + port-specific encoders. Within the modern iGPU range
|
||||
danos would realistically target (Skylake → Meteor/Lunar Lake) the model is stable.
|
||||
|
||||
**The DPLL/clock block is a separate, per-port programmable clock source** and is one of the harder,
|
||||
most gen-specific pieces: pick/enable a PLL, route its output to the DDI, then bring up the port.
|
||||
The register layout and divider math change substantially per generation — pre-SKL SPLL/WRPLL/LCPLL,
|
||||
Skylake+ shared DPLL0–3, Gen11+ combo-PHY plus Type-C MG/DKL PLLs. Pixel-clock computation is a
|
||||
classic per-gen rewrite.
|
||||
|
||||
### Separable from render — the single most important enabler
|
||||
|
||||
The display is a distinct register domain from render/media, and this is confirmed at the primary
|
||||
level: the TGL PRM ships display as its own volume (Vol 12), separate from Render Engine (Vol 9) and
|
||||
Media (Vol 11); Linux's KMS "is provided by Intel Display Driver, and **shared with drm/xe**"
|
||||
([kernel.org i915](https://docs.kernel.org/gpu/i915.html)) — i.e. the display module is
|
||||
reused across two different GPU drivers. A full mode-set lights a display end-to-end using only power
|
||||
wells, PLL/port-clock, DDI-buffer/PHY, transcoder and pipe registers — **zero render commands, zero
|
||||
GEM objects, zero command-streamer.** libgfxinit is decisive proof: complete EDID + modeset +
|
||||
framebuffer with no render/3D code at all.
|
||||
|
||||
Two carve-outs the "touches ONLY display registers" phrasing needs (per verification), **neither of
|
||||
which drags in the render engine**:
|
||||
|
||||
1. A mode-set that changes the **Core Display Clock (CDCLK)** frequency/voltage pokes the shared **GT
|
||||
Driver Mailbox** (PCODE/PCU power-controller interface), per Vol 12's own "Display Voltage
|
||||
Frequency Switching" step. A trivial register handshake, documented alongside the display sequence.
|
||||
2. The primary plane's surface register (`PLANE_SURF`) holds a **GGTT graphics address** (a
|
||||
memory-interface concept, not covered in Vol 12). Using pre-mapped stolen memory — as libgfxinit
|
||||
does — sidesteps any active GGTT programming. See [Memory and scanout](#memory-and-scanout).
|
||||
|
||||
### Per-gen churn: what's stable, what you rewrite
|
||||
|
||||
The **object model** (pipes/planes/transcoders/DDIs, GMBUS-for-EDID, double-buffered plane registers
|
||||
armed atomically) is conceptually stable from Ironlake/Haswell through Tiger Lake. What you rewrite
|
||||
per generation is:
|
||||
|
||||
1. the **CPU-vs-PCH split and interconnect**,
|
||||
2. the **port/PHY + DPLL** programming,
|
||||
3. **register offsets + power-well / CDCLK topology**, and
|
||||
4. the **mode-set enable sequence itself** (power-well ordering, PLL lock, DDI-buffer enable,
|
||||
transcoder clock-select) — an effective fourth axis the raw research folded into (1)/(2).
|
||||
|
||||
Interconnect eras, with the timeline **corrected** (the cited Haiku doc was chronologically loose):
|
||||
|
||||
- **Gen5 Ironlake (2010) → Ivy Bridge:** FDI (Flexible Display Interface) links the CPU display
|
||||
engine to PCH-resident ports. The FDI/PCH-split era begins at **Ironlake**, not Gen7.
|
||||
- **Haswell (Gen7.5):** the main digital outputs come **back onto the CPU die as DDIs** (DDI A = eDP)
|
||||
— the *opposite* of "moving output to the PCH," and it collapses the FDI/PCH dance **for the
|
||||
digital ports only**. FDI is **retained** for the legacy VGA/CRT path (DDI E → PCH CRT DAC), so a
|
||||
driver gets the single DDI code path only by omitting analog VGA (which a minimal driver does).
|
||||
- **Skylake (Gen9):** reworks clock/PLL, CDCLK, and the power-well model; introduces the optional DMC.
|
||||
- **Gen11 Ice Lake / Gen12 Tiger Lake:** add combo-PHY + USB-Type-C/Thunderbolt MG/DKL PHYs — the
|
||||
single biggest cost increase, and the reason "newest silicon" is *not* the easiest target. (DSC is
|
||||
documented per-**pipe**; MSO is an eDP feature — not "per-transcoder" as the raw research said.)
|
||||
|
||||
### The tractable sweet spot
|
||||
|
||||
The documented, tractable sweet spot for a from-scratch display-only driver is the
|
||||
**Haswell (Gen7.5) / Broadwell (Gen8) DDI family, with Skylake (Gen9) as the modern-hardware pick**
|
||||
since it shares the same DDI object model. Rationale:
|
||||
|
||||
- Broadwell has a complete, freely downloadable
|
||||
[PRM Vol 11 Display](https://cdrdv2-public.intel.com/690828/intel-gfx-prm-osrc-bdw-vol-11-display.pdf);
|
||||
its engine (3 pipes A/B/C, 4 transcoders incl. transcoder-EDP that floats onto any pipe, DDI A–E,
|
||||
WRPLL/SPLL/LCPLL) is the classic "DDI + transcoder + WRPLL" model.
|
||||
- It predates the combo-PHY / Type-C / MG-DKL complexity of Ice Lake / Tiger Lake.
|
||||
- libgfxinit's DDI **connector/EDID/DP layer is uniform from Haswell through Coffee Lake**, so the
|
||||
hardest-to-get-right port logic generalises widely.
|
||||
|
||||
Two supporting claims from the raw research are **wrong and corrected here (verification):**
|
||||
|
||||
- **The BDW and SKL PRMs are NOT 0BSD-licensed.** Both carry a Creative Commons
|
||||
**Attribution-NoDerivatives** notice. Only the *newer* OSRC PRMs (Tiger Lake 2021 onward) put their
|
||||
embedded code samples under **Zero-Clause BSD**. So for the recommended Haswell/Broadwell/Skylake
|
||||
generations there are no "copy-pasteable 0BSD code samples" — the legal basis is *reimplementation
|
||||
from a CC-BY-ND spec* (register facts are not copyrightable), not copying.
|
||||
- **FDI+PCH is not fully eliminated on Haswell/Broadwell.** The BDW PRM keeps FDI for the DDI E → PCH
|
||||
CRT DAC. The "one DDI code path" holds only for the digital outputs a minimal driver targets.
|
||||
|
||||
Sandy/Ivy Bridge (Gen6/7) is where the hobby-doc walkthroughs concentrate (the OSDev GMBUS/EDID
|
||||
material) but carries the FDI+PCH split cost. *(Low confidence on the OSDev specifics — the wiki
|
||||
returns 403 to automated fetches and its "guaranteed to work" phrasing is a hobby assertion, not a
|
||||
silicon guarantee.)*
|
||||
|
||||
## Documentation — and the clean-room question
|
||||
|
||||
This is the crux of the whole comparison. **Intel hands you the register spec that NVIDIA withholds.**
|
||||
|
||||
- The Tiger Lake **"Vol 12: Display Engine"** PRM is a real, first-party, open-source document —
|
||||
**433 pages, verified by direct download** — with named registers + addresses + bitfield tables
|
||||
(`TRANS_DDI_FUNC_CTL`, `DDI_BUF_CTL`, `DP_TP_CTL`, `PLANE_STRIDE`, `DPLL_CFGCR0/1`, `CDCLK_CTL`,
|
||||
`PWR_WELL_CTL_DDI`, …) and **numbered, step-by-step enable sequences** with explicit writes, wait
|
||||
conditions, and microsecond timeouts. It even includes the "magic value" tables older PRMs deferred
|
||||
to the driver (DisplayPort PLL DCO/divider values; voltage-swing/de-emphasis in mV). *"A spec you
|
||||
could write a driver from directly"* is well-supported, not hyperbole
|
||||
([TGL Vol 12](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)).
|
||||
- **Clean-room, permissively-licensed implementation is legally and practically feasible from the
|
||||
PRM alone.** CC-BY-ND governs redistribution of the *document*; register addresses and bit
|
||||
definitions are functional facts, and original code implementing a described hardware interface is
|
||||
not a derivative of the PDF. *(This is standard copyright reasoning, not adjudicated case law —
|
||||
treat it as well-grounded, not settled.)* Two independent implementations already exist built
|
||||
essentially from these docs (libgfxinit, Haiku), so the spec is demonstrably sufficient.
|
||||
|
||||
**The documentation ceiling — corrected.** The raw research said public PRMs stop "roughly at Ice
|
||||
Lake / Tiger Lake." Verification refuted this: full public **"Vol 12 Display Engine"** PRMs exist for
|
||||
Ice Lake, Lakefield, Tiger Lake, Rocket Lake, DG1, **and DG2/Arc "Alchemist" (Gen12.5, 2022)** —
|
||||
[the ACM display PRM is public](https://www.x.org/docs/intel/ACM/intel-gfx-prm-osrc-acm-vol12-displayengine.pdf).
|
||||
The genuine cliff is **Meteor Lake (2023) and newer**: those have only a high-level architecture
|
||||
overview, no register-level display PRM, and i915 references their display registers by opaque
|
||||
internal **Bspec numeric IDs**. Alder Lake and Raptor Lake iGPUs are Gen12 Xe-LP display — the same
|
||||
IP as Tiger Lake — so despite lacking a dedicated PRM they are effectively covered by the TGL PRM.
|
||||
|
||||
Net: a from-docs driver can confidently target **Skylake through DG2/Arc**, which is essentially the
|
||||
entire current laptop/NUC installed base; only Meteor Lake and later slide back toward the NVIDIA
|
||||
situation (reverse-engineering or reading GPL i915). The PRMs also survived 01.org's shutdown and are
|
||||
mirrored in several stable places (Intel's cdrdv2 host, the
|
||||
[Igalia CC-BY-ND archive](https://github.com/Igalia/intel-osrc-gfx-prm) for Gen4–Gen9.5,
|
||||
[kiwitree](https://kiwitree.net/~lina/intel-gfx-docs/prm/), x.org) — not a single point of failure.
|
||||
*(Note: the Igalia archive stops at Kaby Lake and contains no Display Engine volume; the TGL/DG2
|
||||
display PRMs are separate Intel/x.org downloads.)*
|
||||
|
||||
## coreboot libgfxinit — the native reference
|
||||
|
||||
[libgfxinit](https://doc.coreboot.org/gfx/libgfxinit.html) is the closest thing to a template danos
|
||||
could ask for: a self-contained **native modeset library** (no VBIOS/int10, no firmware blobs) that
|
||||
probes displays via EDID over DDC/I²C and DP AUX, and drives LVDS, eDP, DP1–3, HDMI1–3, analog VGA,
|
||||
plus USB-C DP/HDMI alt-mode on Tiger Lake. It sets up pipes (Primary/Secondary/Tertiary), planes,
|
||||
transcoders, PLLs, panel power/backlight, the GTT, and framebuffer scanout — **display-only, zero
|
||||
3D/media/compute**, which is exactly danos's scope. Its public entry is essentially
|
||||
`Initialize()` then `Update_Outputs(Pipe_Configs)`, where each `Pipe_Config` carries
|
||||
`{Port, Framebuffer, Cursor, Mode}` — a near-perfect fit for a pluggable scanout backend.
|
||||
|
||||
Why it beats i915 as a reference (**verified by measurement**): **131 Ada source files, ~818 KB,
|
||||
~22k code lines** across *all* generations, factored precisely along the axes you care about (`edid`,
|
||||
`dp_aux`, `dp_training`, `pipe_setup`, `transcoder`, `plls`, `connectors`, `port_detect`), with
|
||||
**none** of the DRM/KMS/GEM/TTM, GT/3D, RC6/RPS, or GuC/HuC machinery that makes
|
||||
`drivers/gpu/drm/i915` **~419k lines / 900 files / 12 MB**. (A grep confirms *zero* gem/ttm/guc/huc/
|
||||
execbuf identifiers in the tree.) It depends only on a small HW-access shim, `libhwbase`
|
||||
(`HW.PCI`, `HW.Port_IO`, `HW.MMIO`, `HW.Time`), which maps naturally onto danos's MMIO-grant + IPC
|
||||
primitives — you provide Zig equivalents and the modeset logic sits on top. *(Correction to the raw
|
||||
research: the widely-quoted "~13–14k LOC" is only the generic `common/` layer; the eight
|
||||
per-generation subdirs roughly double it.)*
|
||||
|
||||
**It is a read-and-reimplement reference, not drop-in code.** Two hard constraints:
|
||||
|
||||
- **License is GPL-2.0-or-later** (the COPYING file is GPLv2; per-file headers add "or any later
|
||||
version"). The CC-BY-4.0 on the docs *site* is a footer, not the source license. Copyleft applies
|
||||
to ported code.
|
||||
- **It is SPARK/Ada, and designed to run as coreboot boot-firmware**, not a runtime OS driver. A
|
||||
danos port means either an Ada/GNAT toolchain in the build or hand-transliteration into Zig; the
|
||||
SPARK "absence of runtime errors" proof does **not** carry over to your reimplementation (and note
|
||||
it proves absence of runtime errors, **not** functional modeset correctness).
|
||||
|
||||
Two more caveats worth knowing: its **error handling is limited** — "only the case that no display
|
||||
could be found counts as failure"; a later DP link-training failure is *not* propagated. And its
|
||||
**verified-in-coreboot** hardware list stops at **Coffee Lake + Apollo Lake**, even though the tree
|
||||
contains a `tigerlake/` directory (Ice Lake has no directory at all, and Alder Lake support is only
|
||||
"begun"). So treat Haswell..Coffee Lake as the trustworthy transliteration window and TGL as
|
||||
present-but-less-proven.
|
||||
|
||||
The orchestration reads as a clean state machine (`hw-gfx-gma.adb` `Enable_Output`):
|
||||
`Fill_Port_Config → Preferred_Link_Setting → PLLs.Alloc → [retry] Connectors.Pre_On →
|
||||
Display_Controller.On → Connectors.Post_On`, with a literal *"try each DP-lane configuration twice"*
|
||||
inner retry and an outer link-setting step-down. `hw-gfx-dp_training.adb` (398 lines) is a complete,
|
||||
generic DP link-training implementation (TP1/TP2/TP3, CR + EQ loops, swing/pre-emphasis adjust from
|
||||
sink status). Per-generation buffer translations plug in underneath via
|
||||
`Program_Buffer_Translations`, gated on `Config.Has_DDI_Buffer_Trans`. All of this was confirmed
|
||||
against the source line-by-line.
|
||||
|
||||
## The EDID + mode-set path (Haswell/Broadwell target)
|
||||
|
||||
The whole path is memory-mapped register programming with polled status bits — no command ring, no
|
||||
microcode, no DMA channel.
|
||||
|
||||
**EDID over DDC (GMBUS).** Pure MMIO poking of the GMBUS I²C controller (`GMBUS0`–`GMBUS5`): `GMBUS0`
|
||||
selects pin-pair/port + clock; `GMBUS1` carries slave address (`0x50` for EDID), byte count,
|
||||
direction, SW-ready; `GMBUS2` exposes HW-ready/NAK/ACTIVE to poll; `GMBUS3` is a 4-byte data FIFO;
|
||||
`GMBUS5` gives the 2-byte segment index for E-DDC. A read is: write `GMBUS0`, write `GMBUS1`
|
||||
(`CYCLE_WAIT | count | SLAVE_READ | SW_RDY | slave<<addr`), loop {poll `HW_RDY`, read 4 bytes}, then
|
||||
STOP ([i915 intel_gmbus.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_gmbus.c)).
|
||||
|
||||
**EDID + DPCD over DP AUX.** For DisplayPort/eDP, EDID (as I²C-over-AUX to `0x50`) and all DPCD
|
||||
capability/link-status registers are read over the AUX channel: per-DDI `DDI_AUX_CTL` + 5×
|
||||
`DDI_AUX_DATA`. Build a 3–5 byte header + payload, set SEND_BUSY, poll it clear, read
|
||||
DONE/TIMEOUT/RECEIVE_ERROR. Message size 1–20 bytes; spec requires ≥3 retries. On Haswell/BDW the AUX
|
||||
clock divider is programmed explicitly; SKL+ derive it automatically
|
||||
([i915 intel_dp_aux.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_dp_aux.c)).
|
||||
Both GMBUS and DP-AUX live in libgfxinit's shared `common/` — cheap and nearly gen-invariant.
|
||||
|
||||
**The mode-set is a fixed, documented register sequence.** The Broadwell DisplayPort enable order
|
||||
(verbatim from BDW PRM Vol 11, pp.98–99): (1) DDI lane capability; (2) panel power sequencing if
|
||||
needed; (3) enable the CPU display PLL (WRPLL/SPLL) and wait ~20 µs; (4) Port Clock Select → DDI,
|
||||
enable `DP_TP_CTL` with training pattern 1, configure `DDI_BUF_TRANS`, enable `DDI_BUF_CTL`, wait
|
||||
>518 µs, run link training, set `DP_TP_CTL` to Normal (Idle first for eDP); (5) Transcoder Clock
|
||||
Select, enable the plane, panel fitter if needed, program transcoder timings + M/N/TU, enable
|
||||
`TRANS_DDI_FUNC_CTL`, enable `TRANS_CONF`, then backlight. Disable is the exact reverse — a bounded
|
||||
checklist.
|
||||
|
||||
**DisplayPort/eDP link training is driver-driven in software over AUX** — the CPU runs the
|
||||
clock-recovery and channel-equalization state machines by hand; it is **not** offloaded to a hardware
|
||||
sequencer or firmware. The source side exposes only primitives: `DP_TP_CTL` selects the training
|
||||
pattern the port emits; `DDI_BUF_CTL`/`DDI_BUF_TRANS` set voltage-swing/pre-emphasis. The driver
|
||||
loops: emit pattern + set source levels → write `TRAINING_PATTERN_SET` (DPCD 0x102) + `TRAINING_LANEx_SET`
|
||||
(0x103) over AUX → delay (100 µs CR / 400 µs EQ) → read `LANE_STATUS` → on failure adjust to the
|
||||
sink's `ADJUST_REQUEST` values and retry. A few hundred lines of ordinary CPU/AUX code (libgfxinit
|
||||
`Train_DP`: CR loop 1..32, EQ loop 1..6). **This is the single fiddliest, most fragile piece** — a
|
||||
TMDS/HDMI panel avoids it entirely, and targeting an already-lit eDP panel avoids most of it.
|
||||
|
||||
**The clock (WRPLL) is documented divider math, not a magic table.** On Haswell/BDW the WRPLL derives
|
||||
the symbol clock from a 2700 MHz LCPLL reference through R2/N2/P dividers with VCO 2400–4800 MHz —
|
||||
small integer arithmetic. DP is *easier* than HDMI because it runs at a few fixed link rates (1.62 /
|
||||
2.7 / 5.4 GHz), so a DP/eDP-only minimal driver can often use fixed rates and skip most of the search.
|
||||
|
||||
**Plane/scanout programming is trivial for a compositor.** The primary plane is `PRI_CTL`
|
||||
(enable + pixel format), `PRI_STRIDE`, `PRI_SURF` (surface base — writing it triggers the atomic
|
||||
update), `PRI_OFFSET`; formats include 32-bit BGRX 8:8:8 and 16-bit BGRX 5:6:5 — a direct match for a
|
||||
linear XRGB compositor buffer. Plane registers are double-buffered and latch at vblank via an
|
||||
**arming** write — so a page-flip is "write base + stride + size, then the arming write." This is
|
||||
*exactly* the primitive danos's damage-driven compositor already expresses over GOP/virtio-gpu; the
|
||||
incremental work is "program these display-domain registers," not a new scanout model. The panel
|
||||
fitter (`PF_WIN_POS`/`PF_WIN_SZ`/`PF_CTRL`) can be left disabled for native-resolution scanout;
|
||||
Skylake+ replaces it with a shared pipe-scaler (`PS_CTRL`).
|
||||
|
||||
**Smallest useful target:** eDP (DDI A / transcoder-EDP) or a single DP output at native resolution,
|
||||
panel fitter off, plane in 32bpp XRGB. That is: GMBUS + I²C-over-AUX EDID/DPCD, one fixed-rate or
|
||||
WRPLL config, the ~20-step enable sequence, the software CR/EQ loop, and `PRI_*` plane setup with
|
||||
`PRI_SURF`-write flips. Out of scope: 3D, media, tiling, RC6/power-gating, PSR, audio.
|
||||
|
||||
## Memory and scanout
|
||||
|
||||
This is where Intel's *architecture* — not just its docs — makes the job smaller, and it is the
|
||||
biggest single simplification versus a discrete GPU.
|
||||
|
||||
- **No VRAM.** Intel iGPUs have a unified memory architecture; the display scans out of ordinary
|
||||
**system RAM** addressed through the **Global GTT (GGTT)**. The only way to give the GPU memory is
|
||||
to bind system pages into the GGTT
|
||||
([i915/GEM crashcourse](https://blog.ffwll.ch/2012/10/i915gem-crashcourse.html)).
|
||||
- **The plane surface register is a GGTT offset**, not a raw physical address — the display walks the
|
||||
GGTT to fetch pixels, so a scanout buffer must be GGTT-mapped (global, not per-process). libgfxinit
|
||||
writes the framebuffer offset straight into `DSPSURF`/`PLANE_SURF` masked to 4 KB.
|
||||
- **Linear (untiled) scanout is a first-class supported mode** — the plane's tiling field value 0 is
|
||||
Linear. No X/Y/Yf tiling engine is needed for a display-only driver. (UEFI GOP itself hands off a
|
||||
linear framebuffer the plane is already scanning.)
|
||||
- **No memory manager.** You need only (1) some contiguous-ish system pages and (2) GGTT PTEs
|
||||
pointing at them (`physical_addr | valid_bit` — the GGTT is a flat single-level array of PTEs in
|
||||
the `GTTMMADR` MMIO BAR), then program the plane. **No GEM/TTM/PPGTT/GuC.** coreboot's native-init
|
||||
literally does `for(i…) WRITE32(base + i*inc | 1, (i*4) | 1)`.
|
||||
- **"Stolen memory"** (GSM/DSM) is firmware-reserved system RAM where the firmware places the GGTT
|
||||
itself and the boot framebuffer. A driver is not obligated to keep scanout there — it can rebind
|
||||
GGTT entries to its own pages. Stolen memory matters mainly for *inheriting* the GOP framebuffer at
|
||||
handoff.
|
||||
|
||||
**The contrast with NVIDIA is stark.** On a discrete GPU the scanout surface must live in **VRAM**
|
||||
(nouveau always pins scanout to VRAM), CPU access goes through the **BAR1** aperture (which on
|
||||
consumer cards can be far smaller than total VRAM unless Resizable BAR is on), and you need a
|
||||
contiguous aligned VRAM allocator plus a BAR1 mapping. The Intel iGPU path **eliminates all of that**
|
||||
— scanout is plain system RAM, and a userspace compositor can write the framebuffer pages directly
|
||||
(as danos already does with the GOP WC framebuffer).
|
||||
|
||||
Because danos boots via GOP, an Intel driver attaches to a display whose **GGTT is already populated
|
||||
and whose plane is already scanning a linear framebuffer at native resolution.** A minimal driver can
|
||||
reuse that live mapping and reprogram the running plane rather than come up from cold — the same
|
||||
"attach to a live display" advantage the NVIDIA doc identifies, but with a far smaller register
|
||||
surface and no firmware wall. *(Low-confidence, per-target details to pin from the specific gen's
|
||||
PRM: GGTT PTE size — 4-byte pre-gen8 vs 8-byte gen8+ — the `GTTMMADR`/aperture BAR layout, surface
|
||||
alignment — 4 KB floor but some gens/tilings want 256 KB — and whether the display's GGTT-mediated
|
||||
DMA sits before or after danos's M16 IOMMU on the target platform.)*
|
||||
|
||||
## Firmware
|
||||
|
||||
A minimal display-only Intel driver is **effectively firmware-free — more so than NVIDIA.**
|
||||
|
||||
- **DMC (Display Microcontroller, "CSR", Skylake+) is NOT required for mode-set or scanout.** Its
|
||||
sole job is saving/restoring display-engine registers across DC5/DC6 low-power idle. Absent, i915
|
||||
prints *"Failed to load DMC firmware … Disabling runtime power management"* and the display
|
||||
mode-sets and scans out normally — you lose only the deep display idle states, not output
|
||||
([intel_dmc.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_dmc.c);
|
||||
corroborated by multiple distro bug threads). *(A source-level `HAS_DMC` early-return citation would
|
||||
strengthen this beyond distro testimony, but the conclusion is well-supported.)*
|
||||
- **Pre-Skylake parts have no display microcontroller at all** yet perform full mode-set (and even
|
||||
Panel Self Refresh). This confirms the display engine is fundamentally CPU/MMIO-driven; the
|
||||
microcontroller is an add-on for autonomous idling, not a prerequisite for lighting a panel.
|
||||
Targeting a pre-Skylake or DMC-optional generation sidesteps the question entirely.
|
||||
- **GuC and HuC are render/media microcontrollers on the GT side** — GuC schedules the render engines,
|
||||
HuC assists HEVC/H.265 codec (plus later HDCP/PXP/GSC). Neither is in the scanout path; a
|
||||
display-only driver never loads them
|
||||
([kernel.org microcontrollers](https://docs.kernel.org/gpu/i915.html)).
|
||||
- **PSR firmware lives on the panel**, not in the OS — a minimal driver simply doesn't enable PSR.
|
||||
- **Type-C/TCSS (Ice Lake+) firmware** (PMC/IOM/PHY) is part of platform BIOS/coreboot init and the
|
||||
hardware, *not* a signed blob the display driver loads at runtime. A driver attaching to an
|
||||
already-lit GOP connector, or targeting classic DDI ports, avoids it. *(Cold DP-alt-mode changes
|
||||
from a userspace driver on modern TCSS platforms were not traced to primary source — flagged.)*
|
||||
|
||||
There is **no signed-firmware wall over the Intel GPU at all** on the display path. This is the
|
||||
architectural opposite of NVIDIA's mandatory, unsignable, ABI-unstable GSP — which even on the
|
||||
near-side "direct" display path is a permanent maintenance liability for anything beyond scanout.
|
||||
|
||||
## Licensing
|
||||
|
||||
The situation is *better* than NVIDIA's but still nuanced.
|
||||
|
||||
- **The two best code references are both GPL** — Linux i915 (GPL-2.0) and coreboot libgfxinit
|
||||
(GPL-2.0-or-later). You cannot copy either into a permissively-licensed danos. libgfxinit's WRPLL
|
||||
divider math is itself copied from i915, so it carries the same encumbrance.
|
||||
- **But you don't need to copy code.** The Intel PRM is a *specification*, and a clean-room Zig
|
||||
implementation written from the PRM (using libgfxinit/i915 only to understand behaviour, never to
|
||||
copy) is legitimate — register numbers and bit definitions are functional facts, not copyrightable
|
||||
expression. This is the exact inverse of the NVIDIA case, where no such spec exists and the only
|
||||
guide is the GPL/RE'd code itself.
|
||||
- **A permissive precedent exists: Haiku's `intel_extreme` is MIT-licensed** and was built from
|
||||
Intel's public docs. So if danos wants a permissive license, the model is: implement from the PRM,
|
||||
optionally read MIT Haiku for structure, treat GPL libgfxinit/i915 as documentation-of-last-resort.
|
||||
- **A licensing nuance on the recommended generations:** the "copy the 0BSD PRM code samples" shortcut
|
||||
only applies to Tiger-Lake-era (2021+) PRMs. The Haswell/Broadwell/Skylake PRMs are CC-BY-ND, so
|
||||
their register *facts* are free to implement but there are no code samples to lift.
|
||||
|
||||
As with the NVIDIA doc: danos's userspace-driver-over-IPC model (a driver is a separate process behind
|
||||
a defined protocol) is the cleanest possible license boundary if the project ever chooses to ship a
|
||||
GPL display-driver binary and keep the rest of danos permissive — but that is a boundary judgement
|
||||
wanting real diligence, not a settled fact. The clean-room-from-PRM route avoids the question.
|
||||
|
||||
## Prior art outside Linux
|
||||
|
||||
This is a **real contrast with NVIDIA**, where no one has built a from-scratch native driver outside
|
||||
Linux. For Intel there are **multiple independent, non-Linux, clean-room native modeset
|
||||
implementations** to learn from:
|
||||
|
||||
- **coreboot libgfxinit** — SPARK/Ada, G45/GM45 and Arrandale → Coffee Lake + Apollo Lake (TGL
|
||||
in-tree), the strongest structural reference.
|
||||
- **Haiku `intel_extreme`** — modeset-only (no 2D/3D accel), **MIT-licensed**, i845 through Sandy
|
||||
Bridge solid, newer Gemini/Ice/Tiger Lake in progress but "hit or miss, as the driver lags behind
|
||||
the specs" ([Haiku generations](https://www.haiku-os.org/docs/develop/drivers/intel_extreme/generations.html),
|
||||
[Phoronix Sept 2024](https://www.phoronix.com/news/Haiku-OS-September-2024)).
|
||||
- **SerenityOS** — added basic native Intel graphics ([PR #6277](https://github.com/SerenityOS/serenity/pull/6277)),
|
||||
though only for very old ICH7-class hardware.
|
||||
- **managarm** — native Intel G45 support.
|
||||
|
||||
The catch: **every clean-room non-Linux implementation targets old hardware.** A modern Gen12 "Xe"
|
||||
desktop iGPU is beyond all of them; for the very newest parts only GPL i915 covers the registers. So
|
||||
the wealth of prior art is real but concentrated below Tiger Lake.
|
||||
|
||||
## The practical desktop caveat
|
||||
|
||||
Before any effort estimate is trusted, three hardware realities — the honest reason "Intel is easier"
|
||||
does **not** automatically mean "it'll light up the reader's monitor":
|
||||
|
||||
1. **Muxing / cabling.** On a desktop with a discrete RTX 3060, the monitor is almost certainly
|
||||
plugged into the *card's* outputs, not the motherboard's. An iGPU driver would light a
|
||||
**different, currently-dark** output. To see danos on Intel the reader would have to physically
|
||||
move the cable to a motherboard video port **and** likely enable the iGPU / "IGD Multi-Monitor" in
|
||||
BIOS. Intel-first probably does **not** light the current display without re-cabling.
|
||||
2. **No iGPU at all.** Intel **F-SKU** desktop chips (i5-9400F, i5-12400F, i5-13400F, i7-13700KF, …)
|
||||
ship the graphics **fused off** and cannot be re-enabled. These are extremely common in
|
||||
budget/mid gaming builds paired with an RTX 3060. On an F-SKU (or an X-series HEDT part) the
|
||||
Intel-iGPU path is a **non-starter** regardless of cabling.
|
||||
3. **Generation coverage.** If the CPU *is* a recent non-F part, its iGPU may be Gen12 Xe (Alder/
|
||||
Raptor Lake), beyond libgfxinit's verified set and beyond most non-Linux prior art — leaving GPL
|
||||
i915 (or the TGL-class PRM, which covers Alder/Raptor display IP) as the only reference.
|
||||
|
||||
A cleaner path for *learning* without the hardware lottery: an older bare-metal Intel box (Haswell/
|
||||
Skylake NUC or laptop) whose panel is natively on the iGPU. Note QEMU does **not** emulate an Intel
|
||||
iGPU display engine, so a VM cannot exercise a real Intel modeset path — virtio-gpu (already working)
|
||||
is the VM answer.
|
||||
|
||||
## Alternatives, and the honest Intel-vs-NVIDIA verdict
|
||||
|
||||
| Option | What you get | The tradeoff |
|
||||
|---|---|---|
|
||||
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; no runtime mode change, no hardware vsync, no multihead |
|
||||
| **Intel iGPU, reuse-GOP** | EDID read + plane page-flips on the GOP-set mode | Still bounded to GOP's resolution; but real driver-owned scanout |
|
||||
| **Intel iGPU, full modeset** (this doc) | Runtime modeset, vsync, multihead, from public docs | Tier 2–3 effort; DP link training; per-gen churn; **needs a cable-attached, documented iGPU** |
|
||||
| **Native NVIDIA GA106 direct** ([nvidia-gpus.md](nvidia-gpus.md)) | Same, on the RTX 3060 the monitor is actually plugged into | **Tier 4**; GPL-only reference; DMA channel modeset; de-emphasised legacy path |
|
||||
| **GA106 via GSP/OGKM** | Also unlocks 3D later | Tier 5; unstable version-pinned firmware ABI |
|
||||
|
||||
**The verdict for *this reader* (RTX 3060 box):** For pure "see danos on my screen," **NVIDIA-direct
|
||||
is paradoxically the more relevant path**, because the monitor is already cabled to the 3060 and GOP
|
||||
already drives it — a native NVIDIA driver reprograms *that* live display. An Intel driver, however
|
||||
much easier to *write*, likely lights a dark motherboard port the reader isn't looking at, or hits an
|
||||
F-SKU with no iGPU.
|
||||
|
||||
**The verdict for *learning display bring-up*:** **Intel wins decisively.** Public register PRMs, four
|
||||
independent open reference drivers, an MIT precedent (Haiku), a compact formally-analysed blueprint
|
||||
(libgfxinit), no signed-firmware wall, no VRAM/BAR memory manager, and a legitimate permissive
|
||||
clean-room path. It reaches "first pixel" far faster than the NVIDIA native path — *on hardware that
|
||||
actually has a cable-attached, documented Intel iGPU.* Those two goals — "run on my machine" and
|
||||
"learn the craft" — point at different silicon, and that is the honest bottom line.
|
||||
|
||||
## "First light" milestones — a danos `.scanout` service
|
||||
|
||||
Framed as a danos `.scanout` service (like the virtio-gpu and proposed NVIDIA ones), inheriting the
|
||||
GOP-initialized display — no firmware, no cold POST:
|
||||
|
||||
1. **PCI/BAR bring-up** — enumerate the iGPU, map its MMIO BAR (`GTTMMADR` + register block) and the
|
||||
aperture BAR via danos MMIO grants; confirm the display engine is GOP-live.
|
||||
2. **EDID** — implement GMBUS DDC (`0x50`) and DP AUX; read + parse the panel EDID and DPCD caps.
|
||||
*(Smallest self-contained, gen-invariant milestone — a good first commit.)*
|
||||
3. **First pixel = reprogram, don't re-modeset** — with GOP's mode and GGTT mapping inherited,
|
||||
reprogram the running plane (`PRI_CTL`/`PRI_STRIDE`/`PRI_SURF`, linear, 32bpp XRGB) to point at a
|
||||
danos-owned system-RAM buffer; prove a page-flip via the `PRI_SURF` arming write on the *current*
|
||||
mode before changing timings. This defers the entire DPLL/DDI/transcoder/link-training surface —
|
||||
the hardest, most gen-specific ~70% of the work.
|
||||
4. **GGTT ownership** — write your own GGTT PTEs (via an MMIO grant to `GTTMMADR`) pointing at
|
||||
compositor-owned pages, for double-buffered damage-driven present.
|
||||
5. **Wire into the compositor `.scanout` backend** (`attach_scanout`); add vsync via the display
|
||||
vblank interrupt (IRQ-as-IPC).
|
||||
6. **Full mode-set** (the hard, gen-specific step) — for one chosen generation (Haswell/Broadwell or
|
||||
Skylake): WRPLL/DPLL programming, the ~20-step DDI/transcoder/pipe enable sequence, panel power
|
||||
sequencing for eDP (`PP_CONTROL`/`PP_ON_DELAYS`/`PP_OFF_DELAYS` — a common black-screen pitfall).
|
||||
7. **DisplayPort link training** — only if the panel is DP and GOP's link can't be reused; the
|
||||
software CR/EQ state machine over AUX. TMDS/HDMI avoids it; a live eDP panel avoids most of it.
|
||||
8. **Multihead**, then optionally a second generation once one is solid.
|
||||
|
||||
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with a
|
||||
working display, exactly the resilience v2 already provides via re-attach.
|
||||
|
||||
## Reading list
|
||||
|
||||
**Native reference — coreboot libgfxinit (GPL-2.0-or-later, SPARK/Ada):**
|
||||
- `common/hw-gfx-gma.adb` — `Enable_Output`, the end-to-end modeset state machine.
|
||||
- `common/hw-gfx-dp_training.adb` — the complete generic DP link-training CR/EQ loops.
|
||||
- `common/hw-gfx-gma-pipe_setup.adb` — plane/pipe/scaler + `DSPSURF`/`DSPSTRIDE`/`DSPCNTR` scanout.
|
||||
- `common/hw-gfx-gma-transcoder.adb` — timing generator; `common/hw-gfx-edid.adb`,
|
||||
`hw-gfx-gma-i2c.adb`, `hw-gfx-dp_aux_ch.adb` — EDID/DDC/AUX; `hw-gfx-gma-registers.ads` — offsets.
|
||||
- `common/haswell*/`, `skylake/`, `tigerlake/` — the per-gen PLL/PHY/buffer-translation backends.
|
||||
|
||||
**Vendor register specs — Intel OSRC PRMs:**
|
||||
- [Broadwell Vol 11: Display](https://cdrdv2-public.intel.com/690828/intel-gfx-prm-osrc-bdw-vol-11-display.pdf)
|
||||
(CC-BY-ND) — the recommended Haswell/Broadwell-class enable sequences, plane, panel fitter.
|
||||
- [Tiger Lake Vol 12: Display Engine](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)
|
||||
(code samples 0BSD) — the most complete modern reference incl. PLL/voltage-swing value tables.
|
||||
- [DG2/Arc Vol 12: Display Engine](https://www.x.org/docs/intel/ACM/intel-gfx-prm-osrc-acm-vol12-displayengine.pdf)
|
||||
— the newest public display PRM (Gen12.5, 2022).
|
||||
- [Igalia CC-BY-ND archive](https://github.com/Igalia/intel-osrc-gfx-prm) (Gen4–Gen9.5) and the
|
||||
[kiwitree mirror](https://kiwitree.net/~lina/intel-gfx-docs/prm/) — stable mirrors.
|
||||
|
||||
**GPL reference-of-last-resort — Linux i915 display:**
|
||||
- `intel_gmbus.c`, `intel_dp_aux.c` — the concrete EDID/DDC and DP-AUX register sequences.
|
||||
- `intel_ddi.c` / `intel_ddi_buf_trans.c`, `intel_cdclk.c`, `intel_dpll_mgr.c` — DDI/CDCLK/PLL;
|
||||
`i9xx_plane.c`, `intel_crtc.c` — plane/pipe; `intel_dp.c` — link training. Huge and modular; a
|
||||
reference to confirm undocumented quirks, not a template.
|
||||
|
||||
**Permissive prior art — Haiku `intel_extreme` (MIT):**
|
||||
- [`src/add-ons/kernel/drivers/graphics/intel_extreme/`](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/intel_extreme/)
|
||||
— a second independent modeset-only driver; MIT, so structurally readable for a permissive danos.
|
||||
- [generations.html](https://www.haiku-os.org/docs/develop/drivers/intel_extreme/generations.html)
|
||||
— the best plain-English per-generation fault-line map.
|
||||
|
||||
## Open questions (unresolved by the survey)
|
||||
|
||||
- **Does the target machine have a usable, cable-attached iGPU at all?** F-SKU check, CPU generation,
|
||||
and monitor cabling must be resolved before any effort estimate is trusted (see
|
||||
[practical caveat](#the-practical-desktop-caveat)).
|
||||
- **Does danos even need native mode-*setting*, or only plane/scanout control on the GOP-set mode?**
|
||||
If runtime mode changes aren't required, the driver collapses to EDID + plane page-flips, dropping
|
||||
the DPLL/DDI/link-training ~70% of the work.
|
||||
- **GGTT vs raw physical:** confirm from the exact target-gen PRM that `PLANE_SURF` is interpreted as
|
||||
a GGTT graphics address (well-established, but per-gen confirmation advisable), and the PTE size /
|
||||
`GTTMMADR` / aperture layout for writing GGTT entries.
|
||||
- **Reuse the firmware/GOP GGTT + framebuffer, or install your own GGTT entries?** The latter (needed
|
||||
for double-buffering) means writing GGTT PTEs from the userspace driver via an MMIO grant.
|
||||
- **eDP panel power sequencing** (`PP_*`, T1–T12 delays) — not covered in this pass and a common
|
||||
black-screen source.
|
||||
- **IOMMU interaction** — whether the display's GGTT-mediated DMA needs IOMMU passthrough for the
|
||||
framebuffer pages under danos's M16 IOMMU, or sits before the IOMMU on the target platform.
|
||||
- **DP link-training / AUX robustness and per-generation register drift** are the dominant *risks* —
|
||||
not documentation scarcity.
|
||||
- **Exact Haswell/BDW MMIO offsets** (commonly cited: GMBUS ~`0xC5100`, `DDI_AUX_CTL_A` ~`0x64010`,
|
||||
`DDI_BUF_CTL_A` ~`0x64000`, `DP_TP_CTL_A` ~`0x64040`) were not extracted verbatim from the PRM —
|
||||
confirm against `i915_reg.h` before coding.
|
||||
|
||||
---
|
||||
|
||||
*Research snapshot; verify against current libgfxinit / i915 source and the specific target
|
||||
generation's PRM before building. Intel's public-PRM coverage and the muxing/F-SKU realities of a
|
||||
given machine both change what is actually achievable.*
|
||||
+78
-86
@@ -1,104 +1,96 @@
|
||||
# Logging: the diagnostic log vs. the display
|
||||
# Logging
|
||||
|
||||
danos separates two things that are easy to conflate: the **diagnostic log** — the
|
||||
machine-readable stream of *what the kernel is doing* — and the **display**, the
|
||||
framebuffer surface the OS draws on. They are different concerns with different
|
||||
lifetimes, so they're different code paths.
|
||||
Output is a *diagnostic convenience, never a correctness dependency*: the kernel
|
||||
and every service must run correctly with zero output channels. On top of that
|
||||
rule, danos has **per-process logging** — every process's output is attributed
|
||||
by the kernel and lands in its own file on the flash volume, which is what makes
|
||||
a headless real machine (no serial port) debuggable. The display (the
|
||||
framebuffer surface) is a separate concern and deliberately not a log sink;
|
||||
`main.zig` mirrors a few user-facing status lines and panics to it explicitly.
|
||||
|
||||
The guiding rule: **output is a diagnostic convenience, never a correctness
|
||||
dependency.** The kernel must boot and run correctly with *zero* output channels —
|
||||
no serial, no screen. Logging that can take the kernel down isn't robust; it's a
|
||||
liability. This is the same [resilience](resilience.md) posture the rest of the
|
||||
kernel follows.
|
||||
|
||||
## The log is multi-sink
|
||||
|
||||
`system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||
registered **sinks**, each best-effort and self-guarding:
|
||||
|
||||
```zig
|
||||
log.addSink(arch.serialWrite); // the serial UART
|
||||
if (arch.debugconPresent()) log.addSink(arch.debugconWrite); // 0xE9 debug console
|
||||
// later: log.addSink(fs.logWrite); // a file on a ramdisk / USB / SSD
|
||||
log.write("…"); log.print("x={d}\n", .{x});
|
||||
```
|
||||
|
||||
Properties that matter:
|
||||
|
||||
- **No allocation.** The sink table is a fixed array, so the log works before the
|
||||
heap is up and inside a panic.
|
||||
- **Best-effort.** A sink whose device is absent is a no-op (e.g. writing to a
|
||||
missing UART just goes nowhere — the TX-wait is bounded so it can't hang). A
|
||||
message reaches whatever channels exist; if none do, the kernel runs on, silent.
|
||||
- **Order-independent.** Every registered sink gets every message. Adding the file
|
||||
logger later is one `addSink` call and **zero** changes to call sites.
|
||||
|
||||
## The framebuffer is *not* a log sink
|
||||
|
||||
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
||||
Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
||||
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
||||
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
||||
|
||||
Instead the two paths are explicit:
|
||||
## The pipeline
|
||||
|
||||
```
|
||||
verbose diagnostics ──► log ──► serial, debugcon, (file later)
|
||||
user status / panics ──► status() ──► log (above) + framebuffer (if present)
|
||||
process std.log ──▶ debug_write(level) ──▶ tagged kernel ring ──▶ logger service ──▶ /var/log/<boot-stamp>/<binary-path>.log
|
||||
kernel log.print ─┘ │
|
||||
└▶ serial / 0xE9 sinks (QEMU, -Dserial)
|
||||
```
|
||||
|
||||
A handful of user-facing lines (`kernel initialised`, a panic) go through
|
||||
`main.zig`'s `status()` / `statusPrint()`, which write to the log **and** paint the
|
||||
framebuffer when one is present. Everything else uses `log.*` and never touches the
|
||||
screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
||||
1. **Emit.** A program calls `std.log.info("mounted {s}", .{path})` — the
|
||||
runtime's `logFn` (installed for every binary by the root shim,
|
||||
`library/runtime/log.zig`) formats one line and issues one `debug_write`
|
||||
carrying the level. The payload does NOT contain the process's name.
|
||||
`runtime.system.write` remains as the raw/bring-up path (panics, test
|
||||
fixtures); raw bytes ride the same ring, attributed all the same.
|
||||
|
||||
## Optional framebuffer (headless machines)
|
||||
2. **Stamp.** The kernel wraps every payload LINE in a record stamped with the
|
||||
sender's pid, task name (its binary path, e.g. `/system/services/fat`),
|
||||
level, a per-boot sequence number, and a monotonic timestamp
|
||||
(`system/kernel/log.zig` + `log-ring.zig`). Attribution is structural — a
|
||||
payload cannot forge another sender's tag, and an embedded newline just ends
|
||||
the record, so the forged "prefix" lands inside the forger's own next line.
|
||||
|
||||
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||
serial-less machine boots and runs correctly — it just goes quiet.
|
||||
3. **Retain.** The 512 KiB ring overwrites oldest-first; sequence gaps make any
|
||||
loss countable. `klog_read` (#32) copies stream bytes from a free-running
|
||||
offset; `klog_status` (#45) returns the cursors plus the wall-clock time of
|
||||
boot. The framing (`abi.KlogRecordHeader`) is 32 bytes + name + payload,
|
||||
8-byte aligned.
|
||||
|
||||
## Last-resort channels (no text output at all)
|
||||
4. **Render.** Registered sinks (serial under `-Dserial`, the 0xE9 debug
|
||||
console) get a live transcript: kernel/raw output verbatim, leveled records
|
||||
as `<binary path>: message` — one composed write per line, under the log's
|
||||
own spinlock (never the big kernel lock; panic paths try-acquire with a
|
||||
bound and fall back to sinks-only). Sinks are best-effort and self-guarding;
|
||||
a serial-less machine just goes quiet.
|
||||
|
||||
Two signals bypass the sink list, because they must survive even a total
|
||||
output-channel failure:
|
||||
5. **Persist.** The **logger service** (`system/services/logger`) drains the
|
||||
ring every 250 ms and demultiplexes records into one file per source under
|
||||
`/var/log/<boot-stamp>/`, e.g.
|
||||
|
||||
- **`log.checkpoint(code)`** — a one-byte **POST code** to I/O port `0x80` (a POST
|
||||
card or BMC shows it). `main.zig` emits one at each boot milestone (`cp_paging`,
|
||||
`cp_heap`, …) and on a fault/panic, so "where did it hang?" is answerable with no
|
||||
text output whatsoever. Writing `0x80` is universally safe.
|
||||
- **`log.recordPanic(msg)`** — stamps the panic message into a fixed record
|
||||
(`log.panic_record`, with a `magic` written last). A post-mortem — an attached
|
||||
debugger, a RAM dump, or a future file/pstore reader — recovers *what killed it*
|
||||
even though nothing was on screen.
|
||||
```
|
||||
/var/log/2026-07-21T150434Z/kernel.log
|
||||
/var/log/2026-07-21T150434Z/system/services/fat.log
|
||||
/var/log/2026-07-21T150434Z/system/drivers/usb-storage.log
|
||||
```
|
||||
|
||||
The panic and CPU-exception handlers fan out to every sink, emit a POST code, and
|
||||
drop the breadcrumb — they never assume a console.
|
||||
The boot stamp is the RTC anchor from `klog_status` (FAT-safe: no colons; a
|
||||
dead RTC yields the 1970 directory rather than no logs). Each line carries
|
||||
the record's monotonic timestamp and level. Storage is best-effort and late:
|
||||
the ring buffers a whole boot many times over, and the first successful
|
||||
`makePath` of the per-boot directory (also the readiness probe) triggers a
|
||||
full backlog write. Files close — which is the fat server's SCSI cache
|
||||
flush — after a ~2 s quiet period, bounding data-at-risk without per-record
|
||||
flush thrash. At shutdown init stops the logger FIRST (it is last in the
|
||||
boot order), so its final drain runs over a live storage chain.
|
||||
|
||||
## The 0xE9 debug console
|
||||
## Why a ring in the kernel, not a logging server
|
||||
|
||||
Port `0xE9` is the Bochs/QEMU debug console. It's detected safely: the port returns
|
||||
`0xE9` when read if present, and `0xFF` on real hardware, so `debugconPresent()`
|
||||
only enables the sink when it's really there. Under QEMU it's captured with
|
||||
`-debugcon file:…`, giving CI a log channel independent of `-serial`.
|
||||
The storage stack must be able to log. If the fat server wrote its own log file
|
||||
through the VFS it would rendezvous-deadlock on itself; if processes sent
|
||||
records to a logging server over IPC, early boot would need a buffer that is —
|
||||
a ring, one hop later. The kernel ring is that buffer, placed where every
|
||||
process (and the kernel itself) can reach it with one syscall, before any
|
||||
service exists. The logger service is a *reader*, not a hop.
|
||||
|
||||
## The robustness spectrum
|
||||
Two disciplines keep it honest:
|
||||
|
||||
The result handles every combination — framebuffer only, serial only, both, or
|
||||
**neither**. With no channels at all the kernel still boots and runs; port-`0x80`
|
||||
checkpoints track progress and the panic breadcrumb captures failures. *Runs blind
|
||||
but correct* is the goal, not *always has output*.
|
||||
- the logger announces itself **once** — a periodic status line would feed the
|
||||
very stream it drains;
|
||||
- lost records surface as an explicit `-- N records lost --` line, computed
|
||||
from sequence gaps, never silently.
|
||||
|
||||
## Related
|
||||
## Last-resort channels
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the display surface itself (pitch, format), the
|
||||
thing that becomes a graphics device driver.
|
||||
- [efi.md](efi.md) — where the loader captures (or, headless, doesn't capture) the
|
||||
framebuffer before `ExitBootServices`.
|
||||
- [device-interrupts.md](device-interrupts.md) — the serial UART bring-up the log's
|
||||
primary sink rides on.
|
||||
- [resilience.md](resilience.md) — why "never let a missing peripheral take the
|
||||
kernel down" is a core design stance.
|
||||
Unchanged, and independent of the sink list so they survive a total output
|
||||
failure: `checkpoint` (a one-byte POST code on port 0x80) and `recordPanic`
|
||||
(a fixed breadcrumb record, `log.panic_record`, findable in a RAM dump; magic
|
||||
written last so a reader only trusts a complete record).
|
||||
|
||||
## Accepted gaps
|
||||
|
||||
- A write-spamming process can evict other processes' unread records from the
|
||||
ring (a per-process quota is future work); the loss is at least visible via
|
||||
sequence gaps in every affected file.
|
||||
- `/var/log` files have no privacy until the VFS grows permissions.
|
||||
- Records emitted after the logger's final shutdown drain reach serial and the
|
||||
ring but not the files.
|
||||
|
||||
@@ -0,0 +1,246 @@
|
||||
# Native NVIDIA GPU support — feasibility and roadmap
|
||||
|
||||
**Status: research snapshot, not implemented.** This records what a *native* display driver for a
|
||||
real discrete NVIDIA GPU — specifically an **RTX 3060 (Ampere GA106)** — would take, and how it
|
||||
would slot into danos's pluggable scanout architecture. It is a survey of primary sources
|
||||
(NVIDIA's [open-gpu-kernel-modules](https://github.com/NVIDIA/open-gpu-kernel-modules), the Linux
|
||||
[nouveau/nvkm](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/nouveau) driver,
|
||||
NVIDIA's [open-gpu-doc](https://nvidia.github.io/open-gpu-doc/), and
|
||||
[linux-firmware](https://github.com/NVIDIA/linux-firmware)), not an implementation. The NVIDIA
|
||||
driver landscape moves quickly (GSP defaults, firmware ABIs); treat specifics as a mid-decade
|
||||
snapshot and re-verify against current source before building.
|
||||
|
||||
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the
|
||||
v2 model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- A **minimal display-only driver** (EDID + mode-set + framebuffer scanout, **no** 3D/compute)
|
||||
for the RTX 3060 **can and should avoid the GSP entirely**. nouveau has a register-level,
|
||||
CPU-driven display path for Ampere (`nvkm/engine/disp/ga102.c`) that lights up GA106 with no
|
||||
external firmware; the signed-firmware wall gates the **compute/graphics** engines (PGRAPH),
|
||||
**not** the display controller. "GSP is mandatory on Ampere" is true only for NVIDIA's own
|
||||
RM-object route.
|
||||
- **danos's UEFI GOP boot is the single biggest thing in its favour.** The VBIOS/GOP has already
|
||||
run devinit and brought up the display PLLs, so a driver attaches to a **live, initialized**
|
||||
GA106 — no firmware load, no cold-boot POST, no devinit interpreter. You reprogram a running
|
||||
display rather than bring one up from cold.
|
||||
- It is still a **hard, multi-week-to-months expert effort** (effort tier ≈ 4/5) dominated by
|
||||
NVDisplay channel-DMA programming, SOR/head routing, DisplayPort AUX + link training, and the
|
||||
display supervisor handshake. The GSP/RM route is tier 5 (near-infeasible solo).
|
||||
- The **licensing tension is counterintuitive**: the permissively-licensed reference (NVIDIA
|
||||
open-gpu-kernel-modules, MIT/GPLv2) is the **hard GSP path**; the register-level display code
|
||||
you actually want lives in **GPL nouveau**. See [Licensing](#licensing).
|
||||
- The **window is closing**: GA10x (Ampere) is the *last* NVIDIA family with a register-level
|
||||
display path — Ada (RTX 40) deleted its non-GSP display HAL. Targeting Ampere specifically
|
||||
matters.
|
||||
- **Recommendation:** for *this card*, GOP already gives native-resolution scanout with zero GPU
|
||||
code and zero maintenance. A native driver buys only runtime mode changes, hardware
|
||||
vsync/vblank, and multihead. It is justified if that runtime control is a danos goal, or to
|
||||
*learn the craft* — for which an Intel iGPU or a pre-Turing NVIDIA card reaches "first pixel"
|
||||
far faster.
|
||||
|
||||
## The GSP wall, and why display sits on the near side of it
|
||||
|
||||
On Turing and later, NVIDIA split its driver's Resource Manager into a host **CPU-RM** and a
|
||||
**GSP-RM** running on an on-die RISC-V core ("Peregrine"), talking over RPC
|
||||
([LWN 953144](https://lwn.net/Articles/953144/)). The GSP is a *full resource manager*, not a
|
||||
display coprocessor — there is no "display-only" GSP image and no small display RPC subset. Its
|
||||
boot chain is entirely signed and mandatory: a VBIOS-resident **FWSEC-FRTS** app carves a
|
||||
write-protected region (WPR2), a signed **Booter** on the SEC2 falcon loads the GSP bootloader,
|
||||
and that loads **GSP-RM** inside WPR. The firmware ships pre-computed signatures and the driver
|
||||
picks one by an on-chip fuse-version register — **you cannot self-sign**, and there is **no stable
|
||||
firmware ABI** (it is revised every driver release; nouveau and the Rust nova-core driver each pin
|
||||
exactly one version). A GSP driver is a permanent maintenance liability, not a one-time build
|
||||
([LWN 1037379](https://lwn.net/Articles/1037379/),
|
||||
[nova-core cover letter](https://lore.freedesktop.org/nouveau/20250826-nova_firmware-v2-7-93566252fe3a@nvidia.com/T/)).
|
||||
|
||||
**But display doesn't need any of that on Ampere.** `nvkm/engine/disp/ga102.c` dual-dispatches:
|
||||
|
||||
```
|
||||
if (nvkm_gsp_rm(device->gsp)) return r535_disp_new(&ga102_disp, ...); // GSP RPC path
|
||||
return nvkm_disp_new_(&ga102_disp, ...); // direct register path
|
||||
```
|
||||
|
||||
Both branches use the same `ga102_disp` HAL and the same `GA102_DISP_*` class IDs; GSP merely
|
||||
swaps register programming for RPC. GA106 (chipset `0x176`) is wired to `ga102_disp_new` in the
|
||||
device table, identical to GA102/103/104/107. Ampere lit up displays via the **direct** path in
|
||||
Linux 5.11/5.17 — two years before GSP-RM landed (6.7, 2023)
|
||||
([ga102.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/nvkm/engine/disp/ga102.c),
|
||||
[Phoronix GA106](https://www.phoronix.com/news/Nouveau-NVIDIA-GA106)).
|
||||
|
||||
**Caveat — this is now the legacy path.** As of Linux 6.18, nouveau defaults to GSP on
|
||||
Turing/Ampere; the direct path is a retained, forceable fallback (`nouveau.config=NvGspRm=0`, and
|
||||
automatic when GSP firmware is absent). It is stable and proven, but NVIDIA and nova-core are
|
||||
moving to GSP-only, and **Ada already deleted its non-GSP display HAL**. GA10x is the last family
|
||||
that keeps a register-level display path.
|
||||
|
||||
## What "direct" actually entails
|
||||
|
||||
"Direct" is not "plain register pokes." Only SOR / PLL / DP-link / clock setup is bare MMIO. The
|
||||
**mode-set and scanout themselves flow through the NVDisplay channels — a DMA pushbuffer**:
|
||||
|
||||
- Display classes for Ampere (the C670 family): core `GA102_DISP_CORE_CHANNEL_DMA` (`0xc67d`),
|
||||
window `0xc67e`, window-immediate `0xc67b`, cursor `0xc67a` (headers `clc67d.h` / `clc67e.h` /
|
||||
`clc67a.h` in [open-gpu-doc `classes/display/`](https://github.com/NVIDIA/open-gpu-doc/tree/master/classes/display)).
|
||||
- The core channel needs **instance memory, a RAMHT, DMA objects, and a channel user-MMIO
|
||||
region** ([disp/chan.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/nvkm/engine/disp/chan.c)).
|
||||
The register-level "plumbing" to allocate/kick a channel is in NVIDIA's GA102 display register
|
||||
manual: `NV_PDISP_FE_CHNCTL_CORE/WIN/CURS`, `NV_PDISP_FE_PBBASE/PBBASEHI`
|
||||
([dev_display_withoffset.ref.txt](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/manuals/ampere/ga102/dev_display_withoffset.ref.txt)).
|
||||
- **Mode-set is a method stream** on the core channel: `HEAD_SET_RASTER_*`,
|
||||
`HEAD_SET_PIXEL_CLOCK_FREQUENCY`, `HEAD_SET_CONTROL_OUTPUT_RESOURCE`, `SOR_SET_CONTROL`
|
||||
(protocol select), viewport/scaler, then `UPDATE`. The window channel points at the scanout
|
||||
surface (`SET_CONTEXT_DMA_ISO`, `SET_STORAGE`, `SET_OFFSET`).
|
||||
- After `UPDATE` you must complete the display **supervisor** interrupt handshake (SV1/SV2/SV3).
|
||||
|
||||
**EDID and DisplayPort are a separate subdev you must port.** open-gpu-doc documents *none* of
|
||||
EDID/DDC/AUX. On the direct path you read EDID in-driver via nouveau's `nvkm/subdev/i2c`: bit-bang
|
||||
**DDC/I²C at address `0x50`** (E-DDC `0x30`) for TMDS/HDMI, or native **DP AUX** in `i2c/aux.c`
|
||||
for DisplayPort. DisplayPort **link training** (the `dp.c` `train_cr` / `train_eq` state machine
|
||||
over AUX — clock recovery, lane/rate, voltage-swing/pre-emphasis) is the single hardest and most
|
||||
fragile piece; a DVI/HDMI (TMDS) panel avoids it entirely.
|
||||
|
||||
## The memory floor (smaller than you'd fear)
|
||||
|
||||
Neither route hands you a framebuffer allocator — even GSP-RM does not manage the scanout
|
||||
framebuffer; the driver owns VRAM and merely tells GSP where its page directory is. But
|
||||
display-only is a small fraction of a full GEM/TTM stack:
|
||||
|
||||
- **Pitch-linear (untiled) scanout is allowed** on nv50→Ampere — the window's storage method has a
|
||||
`PITCH` layout mode, so you skip block-linear tiling math
|
||||
([wndwc37e.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/dispnv50/wndwc37e.c)).
|
||||
- The window references its surface through a simple **display context-DMA**
|
||||
(`SET_CONTEXT_DMA_ISO` + a 256-byte-granular `SET_OFFSET = addr>>8`) — a base/limit descriptor,
|
||||
**not** the GPU's 5-level compute page tables. **No full GPU VMM is needed** for scanout.
|
||||
- The surface must live in **VRAM** in practice (nouveau always pins scanout to VRAM). *Open
|
||||
question:* whether GA10x can scan out from a system-memory (GART) surface via a sysmem-target
|
||||
ctxdma — which would let danos skip a VRAM allocator. No source forbids it; nouveau never does
|
||||
it (confidence: medium).
|
||||
- **CPU access** to the framebuffer for compositing goes through **BAR1** (a VRAM aperture); BAR0
|
||||
is the 16 MB register window. BAR1 can be smaller than 12 GB of VRAM unless Resizable BAR maps
|
||||
it all.
|
||||
|
||||
**Net:** you need (1) a contiguous aligned VRAM allocator (256-byte base, pitch a multiple of
|
||||
64 bytes — confirm against the Ampere display refs), (2) a little instmem for the channel
|
||||
pushbuffers + iso ctxdma, (3) a BAR1 CPU mapping. You do **not** need the 5-level VMM, GEM/TTM
|
||||
eviction, or tiling.
|
||||
|
||||
## Licensing
|
||||
|
||||
The tension is the opposite of convenient:
|
||||
|
||||
- **NVIDIA open-gpu-kernel-modules is dual MIT/GPLv2** — usable under MIT, no copyleft on your
|
||||
other code — **but its display logic is the GSP/RM-object route.** Its class headers
|
||||
(`cl0073.h`, `cl2080.h`, `ctrl0073*.h`) are useful, permissive references.
|
||||
- **nouveau is GPLv2**, and the **register-level display sequences you actually want live in
|
||||
nouveau**, not in the MIT code. So the *easy technical path is the GPL-licensed one.* Reading
|
||||
GPL nouveau and reimplementing it in Zig is a derivative-work risk proportional to how closely
|
||||
your code tracks its structure/constants.
|
||||
|
||||
Options: **(a)** accept that the danos NVIDIA display driver is a **GPL component**. danos's
|
||||
userspace-driver-over-IPC model (a driver is a separate process behind a defined protocol, not
|
||||
linked into the kernel) is about the cleanest possible GPL boundary, so the GPL would be contained
|
||||
to that one binary and the rest of danos could keep its own license — but this is a
|
||||
licensing-boundary judgement that wants real diligence, not a settled fact. **(b)** clean-room
|
||||
from *specification* rather than *code*: [envytools](https://envytools.readthedocs.io) + NVIDIA's
|
||||
open-gpu-doc register manuals + the MIT OGKM class headers, treating nouveau as
|
||||
documentation-of-last-resort.
|
||||
|
||||
**Firmware licensing is moot for the direct path** (no firmware is loaded). For completeness: the
|
||||
GSP blobs are marked redistributable under `LICENCE.nvidia`, which permits use by **any
|
||||
OSI-approved open-source OS** (not just Linux), on NVIDIA GPUs, **unmodified**, with **no
|
||||
reverse-engineering of the firmware binary**. The one gate — is danos released under an OSI
|
||||
license? — is only reached on the GSP route, which this doc recommends against for this card.
|
||||
|
||||
## Prior art
|
||||
|
||||
**No one has built a from-scratch native NVIDIA driver outside Linux.** FreeBSD ships
|
||||
`nvidia-drm-kmod`, a *port of NVIDIA's own closed `nvidia-drm.ko`* loading the GSP blob (its old
|
||||
nouveau port was removed). Haiku's NVIDIA support is likewise a *port of OGKM* (GSP, Turing+, very
|
||||
alpha). OpenBSD / DragonFly have neither. Every non-Linux OS that supports modern NVIDIA chose to
|
||||
**wrap NVIDIA's GSP stack** rather than write a native driver. A danos direct-register driver
|
||||
would have exactly one reference implementation — GPL nouveau — and no non-Linux precedent.
|
||||
|
||||
## Alternatives
|
||||
|
||||
| Option | What you get | The tradeoff |
|
||||
|---|---|---|
|
||||
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; **no runtime mode change, no hardware vsync, no multihead** |
|
||||
| **Pre-Turing NVIDIA** (Kepler / early Maxwell) | Direct EVO/disp-core + CRTC/PLL modeset, **no signed firmware, no coprocessor**; mature nouveau reference | Older display class; not this card; only reclocking is firmware-gated |
|
||||
| **Intel iGPU** | **Publicly documented** register interfaces (Intel PRMs); no coprocessor mediating modeset | i915 is huge + generation-specific; write one generation from the PRM |
|
||||
| **Native GA106 direct** (this doc) | Runtime modeset, vsync, multihead on the actual card | Tier-4 effort; GPL reference; DP link training; legacy/de-emphasized path |
|
||||
| **GA106 via GSP/OGKM** | Also unlocks 3D / reclocking later | Tier-5; ~14k-line ante; unstable version-pinned ABI; unprecedented outside Linux |
|
||||
|
||||
## "First light" milestones (direct path, inheriting GOP state)
|
||||
|
||||
Framed as a danos `.scanout` service (like the virtio-gpu driver), taking the direct register path
|
||||
and inheriting the GOP-initialized display — no signed firmware, no devinit, no GSP:
|
||||
|
||||
1. **PCI/BAR bring-up** — enumerate GA106 (`0x176`), map **BAR0** (registers) and **BAR1** (VRAM
|
||||
aperture) via danos MMIO grants; confirm the display engine is GOP-live.
|
||||
2. **VRAM + instmem allocator** — contiguous aligned VRAM for the scanout surface (256-byte base)
|
||||
+ small instmem for pushbuffers / RAMHT / iso ctxdma. No VMM, no TTM.
|
||||
3. **EDID** — port `nvkm/subdev/i2c` DDC (`0x50`) + DP-AUX (`aux.c`); read + parse the panel EDID.
|
||||
4. **Core channel up** — allocate the `0xc67d` core channel as a DMA pushbuffer; stand up the
|
||||
SV1/SV2/SV3 supervisor-interrupt handshake.
|
||||
5. **First pixel = reprogram, don't re-POST** — bind a window (`0xc67e`) at the existing WC
|
||||
framebuffer via `SET_CONTEXT_DMA_ISO` + `SET_OFFSET`, pitch-linear, `UPDATE`; prove you can
|
||||
drive the *current* GOP mode from your own channel before changing anything.
|
||||
6. **Modeset** — push raster timings on a head, route head→SOR→connector, program the pixel-clock
|
||||
PLL, switch to an EDID mode (needs the `clc67d/e` method opcodes from the OGKM headers + the
|
||||
supervisor timing from nouveau `head.c`).
|
||||
7. **DisplayPort link training** — only if the panel is DP and GOP's link can't be reused; the
|
||||
`dp.c` `train_cr`/`train_eq` state machine. TMDS/HDMI is far simpler.
|
||||
8. **Wire into the compositor `.scanout` backend** (`attach_scanout`), add vsync via the display
|
||||
interrupt, then multihead.
|
||||
|
||||
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with a
|
||||
working display (exactly the resilience v2 already provides via re-attach).
|
||||
|
||||
## Reading list
|
||||
|
||||
**Direct path — nouveau (GPLv2):**
|
||||
- `nvkm/engine/disp/ga102.c` — the GA10x display HAL + the GSP/non-GSP dispatch.
|
||||
- `nvkm/engine/disp/{head.c, ior.c, dp.c, hdmi.c, chan.c}` — head/SOR routing, DP AUX + link
|
||||
training, channel-DMA plumbing.
|
||||
- `dispnv50/{corec37d.c, corec57d.c, wndwc37e.c, wndwc57e.c, wndwc67e.c, headc37d.c, cursc37a.c}`.
|
||||
- `nvkm/subdev/i2c` (DDC + `aux.c`) for EDID; `nvkm/subdev/bios/init.c` + `devinit/` **only** if
|
||||
you ever have to re-POST (danos's GOP handoff means you shouldn't).
|
||||
|
||||
**Object model / GSP path — NVIDIA OGKM (MIT/GPLv2):** class headers `cl0073.h`, `cl2080.h`,
|
||||
`ctrl0073system.h`, `ctrl0073specific.h`; `src/nvidia/` for RM control sequences.
|
||||
`nvidia-modeset.ko` (NVKMS) is a *policy* layer over RM and can be bypassed entirely.
|
||||
[nova-core](https://lore.freedesktop.org/nouveau/) (Rust) is the forward-looking reference for GSP
|
||||
boot mechanics (falcon signing, queue rings, RPC).
|
||||
|
||||
**Register / method specs — NVIDIA open-gpu-doc:**
|
||||
- [`classes/display/README.txt`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/classes/display/README.txt)
|
||||
— the channel model + class-to-GPU map (read first).
|
||||
- [`classes/display/clc67d.h`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/classes/display/clc67d.h)
|
||||
+ `clc67e.h` / `clc67a.h` — the Ampere core/window/cursor mode-set method vocabulary.
|
||||
- [`manuals/ampere/ga102/dev_display_withoffset.ref.txt`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/manuals/ampere/ga102/dev_display_withoffset.ref.txt)
|
||||
— `NV_PDISP_FE_*` channel/pushbuffer registers + SOR.
|
||||
- [`DCB`](https://github.com/NVIDIA/open-gpu-doc/tree/master/DCB) — connector→output-resource
|
||||
routing; [`Devinit`](https://github.com/NVIDIA/open-gpu-doc/tree/master/Devinit) +
|
||||
[`BIOS-Information-Table`](https://github.com/NVIDIA/open-gpu-doc/tree/master/BIOS-Information-Table)
|
||||
— VBIOS parsing (bring-up reference; not needed if inheriting GOP).
|
||||
- The 632 KB Volta [`dev_display.ref`](https://download.nvidia.com/open-gpu-doc/Display-Ref-Manuals/1/gv100/dev_display.ref)
|
||||
is the best shot at SOR-DP/AUX register detail the smaller Ampere file omits.
|
||||
|
||||
## Open questions (unresolved by the survey)
|
||||
|
||||
Each needs a direct read of the named nouveau file or experimentation on the actual card:
|
||||
|
||||
- Exact GA106 register/method offsets and PADLINK→SOR→connector wiring (can vary by board vendor).
|
||||
- Whether *any* PLL/devinit re-run is unavoidable vs. fully inherited from GOP.
|
||||
- Whether DisplayPort needs full retraining on takeover, or the GOP-established link can be reused.
|
||||
- The precise SV1/SV2/SV3 supervisor sequence.
|
||||
- Whether a system-memory-target scanout ctxdma could eliminate the VRAM allocator.
|
||||
- The exact `clc67d.h`/`clc67e.h` method opcode numbers (not captured verbatim in the survey).
|
||||
|
||||
---
|
||||
|
||||
*Research snapshot; verify against current nouveau / open-gpu-kernel-modules source before
|
||||
building — NVIDIA's GSP defaults and firmware ABIs change per release.*
|
||||
@@ -0,0 +1,9 @@
|
||||
# OS Developer Guide
|
||||
|
||||
This document is for those who need to understand the architectural decisions behind the OS.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The os was initially written in zig because it has excellent support for EFI. With zig, we could forgo using a third party bootloader, reducing the time to boot up the kernel. Following the "Zen of Zig", helped to produce the most readable codebase for an operating system ever created. So those, new to OS development could quickly get up to speed.
|
||||
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
# The release ISO — flashable boot media
|
||||
|
||||
`zig build release-x86-64` produces **`zig-out/danos-x86-64.iso`**, the file you
|
||||
hand to someone who wants to try danos on a real machine: point
|
||||
[balenaEtcher](https://etcher.balena.io) (or Raspberry Pi Imager, or plain `dd`)
|
||||
at it, flash a USB stick, and boot the stick. The same file also burns to
|
||||
optical media. `zig build check-iso-image` validates it without booting.
|
||||
|
||||
```
|
||||
zig build release-x86-64
|
||||
# Etcher: select danos-x86-64.iso → select the stick → Flash
|
||||
# or: sudo dd if=zig-out/danos-x86-64.iso of=/dev/rdiskN bs=4m (macOS; triple-check N)
|
||||
```
|
||||
|
||||
## Why an ISO when danos-usb.img already boots
|
||||
|
||||
`danos-usb.img` is a raw FAT32 **superfloppy** — a filesystem starting at
|
||||
sector 0, no partition table. UEFI firmware accepts that from a USB stick (it
|
||||
probes whole-disk FAT before giving up), which is why `dd`-ing the .img works
|
||||
and why QEMU and the test harness boot it directly. But it is a
|
||||
developer-shaped artifact: flashing apps expect an ISO, and a superfloppy
|
||||
can't be burned to a CD/DVD or carry a partition table for pickier firmware.
|
||||
|
||||
The ISO wraps that same FAT image — bit-identical, built by the same
|
||||
`tools/make-fat-image.py` — in a container that boots everywhere release media
|
||||
gets consumed. One payload, two images: the .img stays the raw volume the QEMU
|
||||
harness mounts and boots, the .iso is what leaves the building.
|
||||
|
||||
## How a hybrid ISO boots twice
|
||||
|
||||
The trick (the same one Linux distribution ISOs use, usually via `xorriso
|
||||
-isohybrid…`) is that ISO9660 reserves its first 32 KiB as a **system area** it
|
||||
never touches — exactly where an MBR lives on a disk. So one file can carry two
|
||||
tables of contents, both pointing at the same embedded FAT image:
|
||||
|
||||
* **Flashed to USB (Etcher, dd):** firmware sees a disk whose sector 0 is an
|
||||
MBR with one partition of type `0xEF` (EFI System Partition) covering the
|
||||
embedded FAT image. It mounts that ESP and runs `\EFI\BOOT\BOOTX64.efi` —
|
||||
the standard removable-media path ([efi.md](efi.md)).
|
||||
* **Burned to optical media:** firmware reads the ISO9660 volume descriptors
|
||||
at sector 16 and finds an **El Torito** boot record. Its catalog has one
|
||||
entry, platform ID `0xEF` (EFI), whose start LBA is — again — the embedded
|
||||
FAT image. The firmware exposes that image as a virtual disk and runs the
|
||||
same `BOOTX64.efi` off it.
|
||||
|
||||
Neither path involves the legacy BIOS boot-sector machinery: danos is
|
||||
UEFI-only ([system-requirements.md](system-requirements.md)), so the MBR holds
|
||||
no boot code, just the partition entry, and the El Torito entry is EFI-class,
|
||||
not floppy emulation.
|
||||
|
||||
One El Torito wrinkle: the catalog's sector-count field is 16-bit (units of
|
||||
512 bytes), so it can name at most 32 MiB — less than the 64 MiB FAT image.
|
||||
That is fine in practice: firmware sizes the FAT filesystem from its own BPB,
|
||||
and the boot files sit in the first few MiB of the image (clusters are
|
||||
allocated from the front) either way. The USB path has no such cap.
|
||||
|
||||
## The builder
|
||||
|
||||
`tools/make-iso-image.py` follows the house rule of
|
||||
[make-fat-image.py](../tools/make-fat-image.py): pure Python 3 standard
|
||||
library, no external tools (no xorriso, mkisofs, or isohybrid), with a
|
||||
`--verify` mode the `check-iso-image` step runs — it checks that the MBR
|
||||
partition and the El Torito catalog agree on where the FAT image lives and
|
||||
that a FAT32 boot sector is actually there. Every timestamp field in the ISO
|
||||
is zeroed, so the build is reproducible byte-for-byte.
|
||||
|
||||
The ISO9660 filesystem around the boot machinery is minimal but real: a root
|
||||
directory listing `BOOT.CAT` (the catalog) and `EFI.IMG` (the FAT image), so
|
||||
`file`, mount tools, and archive browsers can open the ISO and see what's in
|
||||
it.
|
||||
+10
-7
@@ -1,15 +1,18 @@
|
||||
# System Calls
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
|
||||
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||
> **Status:** danos has real user processes. User programs enter the kernel
|
||||
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
|
||||
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||
> dispatcher); the `int 0x80` gate is kept alongside as a minimal test path.
|
||||
> The live table is `system/abi.zig` (private, renumberable — see
|
||||
> [vdso.md](vdso.md) for the public boundary): process lifecycle + threads,
|
||||
> memory (mmap/dma/shared-memory), synchronous + async IPC with capability passing,
|
||||
> device access, time, the tagged-log diagnostics (`debug_write` with a level,
|
||||
> `klog_read`/`klog_status`), and filesystem NAMING (`fs_resolve`/`fs_node`/
|
||||
> `fs_mount`/`fs_unmount` — the kernel VFS root routes paths and serves the
|
||||
> read-only /system initrd mount; file DATA stays with userspace filesystem
|
||||
> servers over the vfs-protocol, docs/vfs-protocol.md).
|
||||
|
||||
## The Mechanism of a Syscall
|
||||
|
||||
|
||||
@@ -27,6 +27,15 @@ transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
||||
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
||||
a new architecture's UART is what makes the same tests run there.
|
||||
|
||||
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||
A real machine often has no live legacy COM1 — writing to a dead one is slow —
|
||||
and the boot log is kept in a RAM buffer (`klog`) and flushed to disk instead,
|
||||
so serial is now purely a QEMU/dev aid. The harness (`test/qemu_test.py`) builds
|
||||
every case with `-Dserial=true`, and `zig build run-x86-64` boots a serial-enabled
|
||||
image variant, so both get the transcript; a flashable `zig build` image leaves
|
||||
serial out. (Even with `-Dserial`, a loopback probe disables a dead port at boot,
|
||||
so a serial-enabled image is still safe on real hardware.)
|
||||
|
||||
## In-kernel test cases
|
||||
|
||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||
|
||||
@@ -0,0 +1,469 @@
|
||||
# Threading — build plan (`runtime.Thread` over a private thread ABI)
|
||||
|
||||
The ordered, checkpointable build-out for [threading.md](threading.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-v2-plan.md](display-v2-plan.md). Read threading.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **`runtime.Thread` mirrors `std.Thread`'s API; the implementation is danos-native.**
|
||||
Not literal `std.Thread` — that would break the [private ABI](syscall.md).
|
||||
- **Threads are a narrow, per-binary opt-in.** Default concurrency stays process + IPC
|
||||
([resilience.md](resilience.md)); only a service that asks is built
|
||||
`single_threaded = false`.
|
||||
- **Blocking is futex-backed, never spin-backed** — waiters park in the kernel so an
|
||||
idle core still halts ([halting.md](halting.md)).
|
||||
- **New syscalls are private**: extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shared_memory_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
`futex_wait = 40`, `futex_wake = 41`) + a `library/runtime` wrapper; user code never names a number.
|
||||
- **Restart granularity stays the process** — a faulting thread kills its process; the
|
||||
supervisor restarts the process, which respawns its threads.
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||
exercise and register a `ServiceId` if they must be looked up.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
**Every gate is serial-checkable — no screenshots** (this plan runs unattended). A
|
||||
thread proves it ran by writing to **shared memory** the parent reads back, and proves
|
||||
parallelism by stamping the **core index** it ran on (like the `smp`/`affinity` cases).
|
||||
|
||||
- `zig build test` — host unit tests (closure packing, mutex state machine, futex
|
||||
wrapper encodings).
|
||||
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial
|
||||
markers. Thread cases set `smp: true` (real parallelism) and bump `mem` (they boot
|
||||
the process/scheduler stack); each milestone **adds its case to `CASES`** so its gate
|
||||
is runnable.
|
||||
- **Guardrail every milestone:** the concurrency-sensitive existing cases stay green —
|
||||
`smoke`, `sched`, `priority`, `smp`, `affinity`, `process`, `process-kill`,
|
||||
`supervision`, `fault-recovery`, `vfs-client-death`, `ipc`/`ipc-cap`,
|
||||
`display-service`. A threading change that regresses those is rejected.
|
||||
|
||||
## Unattended execution (the loop contract)
|
||||
|
||||
This plan runs to completion **without human input**. Every design choice is already
|
||||
fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration must:
|
||||
|
||||
1. **Resume** at the first milestone that still has an unchecked `- [ ]`. (All earlier
|
||||
milestones are done — do not revisit them.)
|
||||
2. **Work on a branch.** On the first iteration, branch off the current `main` into a new
|
||||
branch (e.g. `threading-phase2` — Phase 1's `threading` is already merged); never
|
||||
commit to `main` directly. Push that **branch** to `origin` after each milestone (step
|
||||
5) so progress is backed up remotely; **do not push `main`** — merging Phase 2 into
|
||||
`main` stays a human step.
|
||||
3. **Implement** every unchecked item in that milestone, including adding its
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../test/qemu_test.py) (with
|
||||
`smp: true` / a `mem` bump where noted) so the gate is runnable.
|
||||
4. **Run the gate**: `python3 test/qemu_test.py <case>`, then the full **guardrail
|
||||
set**, then `zig build` (clean) and `zig build test` (green).
|
||||
5. **Decide, do not ask:**
|
||||
- **Green** = the milestone's case prints its stated marker(s) and reports `PASS`,
|
||||
the whole guardrail set passes, `zig build` is clean, and host tests are green.
|
||||
→ tick this milestone's boxes **and** its `**Gate:**`-referenced case, `git commit`
|
||||
(`threads(M<n>): <summary>`, no `Co-Authored-By` trailer per
|
||||
[coding-standards.md](coding-standards.md)), then **`git push` the working branch to
|
||||
`origin`** (use `-u` on the first push to set upstream). Continue to the next
|
||||
milestone in the same iteration if budget remains; otherwise let the loop re-fire.
|
||||
- **Red** = anything above fails. Diagnose from the captured serial log
|
||||
(`zig-out/qemu-test/<case>-failed-serial.log`) and fix in place, then re-run — up to
|
||||
**3 fix attempts** for that gate. A concurrency case that fails then passes on a
|
||||
bare re-run is **flaky, not green**: re-run it **twice more** and treat green only
|
||||
if it passes all; otherwise fix the race (a real threading bug), don't paper over
|
||||
it.
|
||||
6. **A genuinely ambiguous fork is not a stop.** Pick the option most consistent with
|
||||
[threading.md](threading.md)'s *Locked decisions*, note the choice in the commit
|
||||
message, and continue. Do not pause for confirmation on in-scope, reversible work —
|
||||
this plan is that authorization.
|
||||
|
||||
**The only stop conditions:**
|
||||
|
||||
- **Done** — every milestone box **in this plan** is checked (M1 through M11), `zig build`
|
||||
clean, the whole `thread-*` suite + guardrail green. Phase 1 (M1–M6) is *already*
|
||||
checked, so do **not** read that as Done: the loop's real work is the first plan section
|
||||
that still has unchecked boxes — Phase 2 (M7–M11). Only stop when M7–M11 are all checked
|
||||
too. Update threading.md's status line, push the final branch state to `origin`, and
|
||||
stop. The branch is on `origin` for review; **merging Phase 2 into `main` is the user's
|
||||
step**, not the loop's.
|
||||
- **Blocked** — a gate is still red after 3 fix attempts, or a step needs something
|
||||
outside the repo (a toolchain change, new hardware, a decision no locked decision
|
||||
covers). Append `> **BLOCKED (M<n>):** <what failed, what was tried, the serial
|
||||
marker missing>` under that milestone, commit **and push** the WIP on the branch, and
|
||||
stop. Do not thrash further and do not silently skip the milestone.
|
||||
|
||||
Nothing else warrants stopping — not "should I proceed?", not "is this right?". The
|
||||
checkboxes + git history are the resumable record; the next iteration picks up from the
|
||||
first unchecked box.
|
||||
|
||||
---
|
||||
|
||||
## M1 — Address-space refcount (kernel foundation, no API, no behaviour change) ✅
|
||||
|
||||
The one invariant change threads require, landed and proven **before** anything shares
|
||||
an address space. Today address space is 1:1 with a task and teardown destroys it on any user
|
||||
task's exit; make destruction happen on the **last** exit.
|
||||
|
||||
- [x] A refcount keyed by the address-space root, held in `scheduler.zig`
|
||||
(`address_space_refs`): `retainAddressSpace` takes a reference in `spawnUserLocked` (on the
|
||||
success path, after the slot + stack are secured), all under the big kernel lock.
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig):
|
||||
`exitUserLocked` and `destroyTaskLocked`) call `releaseAspace`, which decrements
|
||||
and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test
|
||||
spaces) is destroyed directly, preserving prior behaviour.
|
||||
- [x] `-Dtest-case=address-space-refcount`: spawn and reap several ring-3 processes in sequence
|
||||
and assert (via test-observable `liveAddressSpaceCount`/`addressSpaceDestroyCount`) that the
|
||||
live-space count returns to **baseline** and destructions advance by exactly that
|
||||
many — each space destroyed exactly once, no leak, no double-free. (Refcount
|
||||
observables, not raw frame counts, since kernel stacks are still leaked on exit.)
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py address-space-refcount` passes
|
||||
(`address-space-refcount: spaces released to baseline ok` → `DANOS-TEST-RESULT: PASS`), and the
|
||||
full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `smp`,
|
||||
`affinity`, `process`, `process-kill`, `supervision`, `fault-recovery`,
|
||||
`vfs-client-death`, `ipc`, `ipc-cap`, `display-service`); default `zig build` clean,
|
||||
`zig build test` green. The reframing is invisible until an address space is actually shared.
|
||||
|
||||
## M2 — `thread_spawn` + `thread_exit`: a thread runs in the shared address space ✅
|
||||
|
||||
Spawn only — no join yet. Prove a second task executes in the **caller's** address
|
||||
space and exits cleanly.
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
process.zig; `thread_spawn` calls `scheduler.spawnThread` (shares the caller's
|
||||
address space, `retainAddressSpace`); `thread_exit` ends the task like a process `exit(0)`
|
||||
(`terminateCurrent` → `releaseAspace`). The closure pointer is delivered in the new
|
||||
thread's **rdi** via a new `jump_to_user_arg` asm path (`t.user_arg`, 0 for a
|
||||
process) — no naked runtime asm.
|
||||
- [x] `library/runtime/thread.zig` (barrel-exported as `runtime.Thread`): `spawn` maps a
|
||||
stack (`mmap`), heap-allocates the `{args}` closure, and calls
|
||||
`thread_spawn(&Closure.entry, stack_top, closure)`; `Closure.entry` (a plain C-ABI
|
||||
Zig fn, closure in rdi) runs the function and calls `thread_exit`. Stack top is
|
||||
16-aligned-minus-8 for the C entry.
|
||||
- [x] A `threaded` flag on the user-binary recipe (`addThreadedUserBinary` →
|
||||
`single_threaded = false`); `thread-test` is the first opt-in binary.
|
||||
- [x] `-Dtest-case=thread-spawn`: `thread-test` spawns a worker that writes a sentinel to
|
||||
a **shared** global and release-stores `done`; the main thread acquire-polls `done`
|
||||
and asserts the shared global holds the sentinel — proof the worker ran in the same
|
||||
address space.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-spawn` passes
|
||||
(`thread-test: child ran in shared address space ok` → `DANOS-TEST-RESULT: PASS`); guardrail set
|
||||
16/16 green (incl. `args`/`init`/`process`, which exercise the new `jump_to_user_arg`
|
||||
process path with arg 0) plus `address-space-refcount`; `zig build` clean, `zig build test`
|
||||
green.
|
||||
|
||||
> **Note (deferred to M3+):** the mmap arena is per-*task* (`heap_next`), so two threads
|
||||
> in one address space that both `mmap` would collide. Fine for M2 (only the parent maps, for the
|
||||
> child's stack); make the arena per-address-space and the runtime heap thread-safe alongside the
|
||||
> `Mutex` work (M5).
|
||||
|
||||
## M3 — `join` + `detach` + real parallelism ✅
|
||||
|
||||
- [x] `join` over the existing exit-notification path
|
||||
([process-lifecycle.md](process-lifecycle.md)): `thread_spawn` gained a 4th arg, an
|
||||
`exit_endpoint` handle (resolved + refcounted like `spawnProcessSupervised`, via
|
||||
`spawnThreadSupervised`); `join` blocks in `ipc_reply_wait` on that endpoint until
|
||||
the child-exit notice for its `tid`, then `munmap`s the stack. `detach` relinquishes
|
||||
the join right (its stack is reclaimed at process exit — kernel-reaper reclaim for
|
||||
detached threads is deferred; see note).
|
||||
- [x] `runtime.Thread.join` / `detach`, plus `Thread.currentCore()` (a new `current_core`
|
||||
= 39 syscall) for the parallelism proof. `getCurrentId` deferred to M6 (TLS), where
|
||||
a lighter self-id fits. The closure now rides the **thread's own stack** (not the
|
||||
heap) — private per thread, so spawn/join touch no shared heap.
|
||||
- [x] `-Dtest-case=thread-join` (`smp: 4`): `thread-test` join mode spawns N=4 workers
|
||||
that each do K=100k `@atomicRmw`-increments on a shared counter and stamp the core
|
||||
they ran on; the main thread joins all N and asserts `counter == N*K` **and**
|
||||
`@popCount(cores_seen) > 1` (genuine cross-core parallelism), then a detached worker
|
||||
proves `detach` runs without a join.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-join` passes (`thread-test: join ok` →
|
||||
`DANOS-TEST-RESULT: PASS`), robust across 4 runs; guardrail 17/17 green (incl. `smp`,
|
||||
`affinity`, `process-kill`, and `args`/`init`/`process` on the exit-endpoint spawn path)
|
||||
plus `address-space-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green.
|
||||
|
||||
> **Note (deferred):** a detached thread's stack is freed only at process exit (not by the
|
||||
> reaper on thread exit) — kernel user-stack tracking + reclaim is a later refinement. And
|
||||
> the runtime heap is still not thread-safe: threads that both allocate concurrently would
|
||||
> race (the thread *machinery* avoids the heap, but worker code sharing an allocator does
|
||||
> not). Both fold into the M5 `Mutex`/allocator work.
|
||||
|
||||
## M4 — Futex: the one blocking primitive ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
`.blocked` task tagged with `Task.futex_addr` (no queue linkage);
|
||||
`futex_wait(addr, expected, timeout_ns)` reads the user word under the big lock,
|
||||
parks iff `*addr == expected`, and returns on wake or timeout; `futex_wake(addr,
|
||||
count)` scans the task table and readies up to `count` matching waiters (same
|
||||
address space). No spinning — a parked waiter leaves its core free to `hlt`. A
|
||||
timed wait also sets `wake_at`, so the timer's `wakeExpired` wakes it; `futex_addr`
|
||||
staying non-zero (only `futex_wake` clears it) is how the waiter tells timeout from
|
||||
a real wake.
|
||||
- [x] `runtime.Thread.Futex` (`wait` / `timedWait` / `wake`) over the syscall wrappers.
|
||||
- [x] `-Dtest-case=thread-futex` (`smp: 4`): a waiter thread prints `waiting` and
|
||||
`futex_wait`s on a word; the main thread publishes it, prints `waking`, and
|
||||
`futex_wake`s; the waiter prints `woke`. Then a `timedWait` on an unwoken word
|
||||
reports `error.Timeout`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-futex` passes, robust across 3 runs —
|
||||
the case's **ordered** regex asserts `waiting → waking → woke → PASS` on the serial
|
||||
stream (the handoff proof), and `thread-futex: timeout ok` confirms the timeout.
|
||||
Guardrail 18/18 green (incl. `sleep`/`event`/`ipc` blocking paths) + `address-space-refcount`,
|
||||
`thread-spawn`, `thread-join`; `zig build` clean, `zig build test` green.
|
||||
|
||||
> **Note:** the kernel test checks only the freshest verdict marker via `bufferHas` (the
|
||||
> in-memory log ring buffer evicts older lines); ordering is asserted against the full
|
||||
> serial stream by the qemu regex instead.
|
||||
|
||||
## M5 — `Mutex` + `Condition` + `Semaphore` ✅
|
||||
|
||||
- [x] `runtime.Thread.Mutex` (three-state futex mutex: CAS fast path, `futex_wait`/`wake`
|
||||
slow path), `Condition` (`wait`/`timedWait`/`signal`/`broadcast`, a futex sequence
|
||||
counter), `Semaphore` (permits over `Mutex`+`Condition`) — the same state machines
|
||||
`std.Thread` uses, ported onto our `Futex`.
|
||||
- [x] `-Dtest-case=thread-mutex` (`smp: 4`): a bounded producer/consumer — 2 producers +
|
||||
2 consumers over one `Mutex` and two `Condition`s move N=2000 unique items through
|
||||
an 8-slot ring; the consumed checksum and tally match exactly (no lost/duplicated
|
||||
item, no overrun) under real cross-core contention. The small ring forces producers
|
||||
to block on full and consumers on empty, exercising `Condition.wait`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-mutex` passes (`thread-mutex: ok` →
|
||||
`DANOS-TEST-RESULT: PASS`), robust across 3 runs; guardrail 17/17 green (incl.
|
||||
`sleep`/`event`/`ipc`) + all M1–M4 thread cases; `zig build` clean, `zig build test`
|
||||
green.
|
||||
|
||||
> **Deferred (with rationale):**
|
||||
> - **`join` → futex completion word** — the exit-endpoint join (M3) is correct and
|
||||
> tested. A futex-completion join needs the *kernel* to clear+wake a word after the
|
||||
> thread is fully off its stack (a CLONE_CHILD_CLEARTID-style mechanism); doing it in
|
||||
> the thread's own trampoline would let `join` `munmap` the stack while the thread still
|
||||
> runs on it (use-after-free). Left on the exit-endpoint path; the kernel clear-on-exit
|
||||
> is a later, separate refinement.
|
||||
> - **Host unit tests for the state machines** — `Mutex`/`Condition` bottom out in the
|
||||
> `futex_*` syscalls, unavailable on the host without a mockable `Futex` seam. The QEMU
|
||||
> `thread-mutex` gate exercises them under real concurrency instead; a host-side mock is
|
||||
> future work.
|
||||
|
||||
## M6 — `getCurrentId`, docs, and CI wiring ✅
|
||||
|
||||
- [x] `getCurrentId` via a small `thread_self = 42` syscall (`runtime.Thread.getCurrentId`
|
||||
returns the kernel task id). **Per-thread `threadlocal` TLS is deferred** — no
|
||||
consumer needs it, and it would require context-switching the thread pointer per task
|
||||
(real kernel + per-switch cost) for an unused feature; threaded binaries have run fine
|
||||
without it through M2–M5. threading.md's TLS reasoning already scoped it as
|
||||
deferred-unless-needed. When a consumer appears, the shape is: `thread_spawn`
|
||||
allocates a per-thread TLS block, sets the thread pointer, and the context switch saves/
|
||||
restores it.
|
||||
- [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same
|
||||
`Futex`/`Mutex`/`Condition` when wanted.
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../test/qemu_test.py)
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`); threading.md + docs/README.md
|
||||
status updated to **built**; the worked example is threading.md's win-condition.
|
||||
- [x] `-Dtest-case=thread-id` (`smp: 4`): two workers read `getCurrentId`; the main
|
||||
thread confirms all three ids are non-zero and distinct — each thread has its own
|
||||
kernel identity. (Renamed from `thread-tls`, which implied `threadlocal`.)
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-id` passes; the whole `thread-*` suite
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`) plus the full guardrail set pass; default
|
||||
`zig build` clean, `zig build test` green.
|
||||
|
||||
---
|
||||
|
||||
## Status
|
||||
|
||||
**Phase 1 (M1–M6): built.** danos has `runtime.Thread` — `spawn`/`join`/`detach`,
|
||||
cross-core parallelism, futex, and `Mutex`/`Condition`/`Semaphore`, all over a private
|
||||
thread ABI behind the runtime.
|
||||
|
||||
**Phase 2 (M7–M11): built.** Thread-safe allocation (M7), a task reaper that reclaims dead
|
||||
tasks' kernel stacks (M8), endpoint-free `thread_join` (M9), the per-thread thread pointer (M10),
|
||||
and `RwLock`/`WaitGroup` + host-testable sync (M11). Two things stay deferred by design
|
||||
(no consumer): the Zig `threadlocal` *compiler* layer (M10) and detached-thread user-stack
|
||||
reclaim (M9) — both noted in place.
|
||||
|
||||
---
|
||||
|
||||
## Phase 2 — hardening (M7–M11)
|
||||
|
||||
The organising principle, so Phase 2 reinforces danos's goals rather than eroding them:
|
||||
|
||||
- **Everything a thread owns is reclaimed on process death.** Thread stacks, TLS blocks,
|
||||
and futex words live in the process's **address space**, and the kernel's per-process
|
||||
state is keyed by the address-space root — so the M1 refcount + `destroyAddressSpace` already
|
||||
free all of it when the last thread exits. A crashed or killed threaded process leaves
|
||||
**nothing** behind. Phase 2 closes the one thing that is *not* address-space-owned — the
|
||||
per-task **kernel** stack (kernel heap) — with a reaper (M8). This is the
|
||||
[resilience](resilience.md) restart guarantee, extended to threads.
|
||||
- **Kernel owns mechanism; the runtime owns policy.** The kernel maps pages, saves/
|
||||
restores the thread pointer, and reaps dead tasks; the runtime decides allocation, TLS layout,
|
||||
and lock algorithms. Every new kernel entry stays a private syscall behind the runtime
|
||||
([syscall.md](syscall.md)) — the ABI stays renumberable.
|
||||
- **The process is still the isolation and restart boundary.** Threads share fate within
|
||||
one process; Phase 2 never adds a way for one process to reach into another (the
|
||||
cross-process futex stays explicitly out of scope, below).
|
||||
|
||||
### M7 — Thread-safe allocation (the correctness gap) ✅
|
||||
|
||||
Today the mmap arena cursor is per-*task* and the runtime heap is unlocked, so two
|
||||
threads in one process that both allocate corrupt each other. The thread *machinery*
|
||||
avoids this (closure on the stack, stacks mmap'd only by the spawner), but real
|
||||
multi-threaded code would hit it. Closed it:
|
||||
|
||||
- [x] **Kernel — per-address-space mmap arena.** Grew M1's `address_space_refs` entry into the
|
||||
per-address-space object holding the `mmap`/`mmio` arena cursors (moved off `Task`);
|
||||
`scheduler.addressSpaceMmapNextPtr`/`addressSpaceDeviceMapNextPtr` expose them. `systemMmap`
|
||||
reserves a disjoint range under a *brief* lock, then maps **per page** under a
|
||||
short-held lock — not the whole grant — because the big lock is held with interrupts
|
||||
disabled, so pinning it across a multi-MiB memset+map froze other cores (it timed
|
||||
the `affinity` scenario out mid-bring-up). Freed at refcount zero, so the cursors
|
||||
vanish with the process.
|
||||
- [x] **Runtime — thread-safe heap.** The allocator's two free-list mutators
|
||||
(`rawAlloc`/`rawFree`) take a `Thread.Mutex`, gated on
|
||||
`!@import("builtin").single_threaded` so single-threaded binaries compile it out and
|
||||
pay nothing. Uncontended acquisition is a single CAS (no syscall).
|
||||
- [x] `-Dtest-case=thread-alloc` (`smp: 4`): 4 threads each do 500 `alloc`/fill/verify/
|
||||
`free` cycles of varied sizes; each block is filled with a per-thread pattern and
|
||||
verified before free, so any overlap between concurrent allocations is caught.
|
||||
|
||||
**Gate (met):** `thread-alloc` passes (3× non-flaky); full guardrail 23/23 green,
|
||||
`zig build`/`zig build test` clean.
|
||||
|
||||
> **Also fixed here:** the `affinity` guardrail's fixed-count busy-loop (`while (spins <
|
||||
> 3e9)`) had codegen-dependent wall-time — adding a function to `tests.zig` flipped how
|
||||
> the optimiser compiled it, swinging affinity from ~4 s to ~63 s and timing it out.
|
||||
> Reworked it (and the settle loop) to wait on the wall clock instead, so its duration is
|
||||
> independent of unrelated code changes.
|
||||
|
||||
### M8 — The task reaper (cleanup + resilience) ✅
|
||||
|
||||
A dead task's **kernel** stack was leaked ("no reaper yet") — every process *and* thread
|
||||
death lost one, so a crash loop bled kernel memory. The reaper fixes it and serves the
|
||||
[resilience](resilience.md) restart goal directly:
|
||||
|
||||
- [x] A dying task cannot free the kernel stack it runs on, so `exit()`/`exitUserLocked`
|
||||
record it in a **per-core `reap_after_switch` slot** and switch away; the task that
|
||||
resumes on that core frees the stack in `switchTo`'s tail (it's on its own stack, the
|
||||
big lock is still held so the slot can't have been reused). A **tick-time drain**
|
||||
(`reapKillPendingLocked`) is the safety net for the case where the next task is
|
||||
*fresh* (enters via the trampoline, bypassing `switchTo`'s tail). A task killed while
|
||||
*not* running is freed immediately in `destroyTaskLocked`. A `live_stack_bytes`
|
||||
counter is the observable. *(Detached-thread user-stack reclaim moves to M9, which
|
||||
adds the joinable/detached flag.)*
|
||||
- [x] `-Dtest-case=task-reap` (`smp: 4`): spawn and kill 12 processes; poll the
|
||||
test-observable `scheduler.liveStackBytes()` until it returns to **baseline** (a
|
||||
correct reaper gets there in a few ms; a genuine leak times out) — every kernel
|
||||
stack reclaimed, no leak. Threads exit through the same `exitUserLocked`, so covered.
|
||||
|
||||
**Gate (met):** `task-reap` passes (5× isolated + 2× in the full batch); `fault-recovery`,
|
||||
`supervision`, `process-kill`, `address-space-refcount`, `smp`, `affinity` all still green (24/24
|
||||
full guardrail); `zig build`/`zig build test` clean.
|
||||
|
||||
> **Bug found + fixed here (touches every context switch):** the post-`switchContext` reap
|
||||
> first read the `pc` **parameter**, but a task that migrated cores carries a *stale* `pc`
|
||||
> in its saved `switchTo` frame — so it read the wrong core's slot and freed a live stack
|
||||
> (a #GP under SMP). Fixed to re-fetch `thisCpu()` after the switch (the switch only swaps
|
||||
> stacks on the current core).
|
||||
|
||||
### M9 — Futex-completion join (retire the per-thread endpoint)
|
||||
|
||||
With the reaper (M8) able to act *after* a thread is fully off its stack, migrate `join`
|
||||
to the std shape and drop M3's per-thread exit endpoint:
|
||||
|
||||
- [x] A **`thread_join(tid)` syscall** (not a user futex word): it blocks the caller until
|
||||
the task with id `tid` exits, and the exit paths call `wakeJoinersLocked`. `join`
|
||||
only reclaims the joined thread's **user** stack, which the thread vacates the moment
|
||||
it enters the kernel to exit — so waking at *exit* time (not reap time) is safe, and
|
||||
no reaper/address-space juggling or user-memory write is needed. This is equally
|
||||
std-shaped (like `pthread_join`) and much simpler/safer than the planned
|
||||
reaper-written completion word. `thread_spawn` no longer takes an exit endpoint (the
|
||||
runtime passes `no_cap`); the per-thread IPC endpoint is gone.
|
||||
- [x] `thread-join` passes on the new path, and its join mode now runs **40 spawn+join
|
||||
cycles** — under the old per-thread-endpoint scheme those leaked handles would
|
||||
exhaust the 16-slot handle table; here they all succeed, proving join is endpoint-free.
|
||||
|
||||
**Gate (met):** `thread-join` passes (3× isolated) on the `thread_join` path; full
|
||||
guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-reap`);
|
||||
`zig build`/`zig build test` clean.
|
||||
|
||||
> **Reaper hardened here (fixes an M8 flake).** M8's single per-core reap slot could be
|
||||
> *overwritten* by a second death on that core before the first drained (a fresh-task/SMP
|
||||
> timing window) — an intermittent one-stack leak (`task-reap` flaked ~20%). Replaced it
|
||||
> with a per-core reap **list** plus a `.reaping` task state so a pending slot can't be
|
||||
> reused before its stack is freed. `task-reap` now 11/11 isolated + 2× in the batch.
|
||||
|
||||
> **Deferred:** detached-thread **user-stack** reclaim (still freed at process exit, as in
|
||||
> M3). Doing it in the reaper needs the saved address space + stack range and a
|
||||
> translate/unmap in a not-currently-loaded address space — real complexity for a bounded leak.
|
||||
> A follow-up when a consumer needs it.
|
||||
|
||||
### M10 — Per-thread TLS: the thread-pointer mechanism ✅
|
||||
|
||||
Give each thread its own thread pointer and private TLS storage — the foundation
|
||||
self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on.
|
||||
|
||||
- [x] **Kernel** stores `thread_pointer` on `Task` and restores it on every context switch
|
||||
**only when it changes** (the same conditional-load discipline as CR3;
|
||||
`architecture.setThreadPointer` → `wrmsr IA32_FS_BASE` on x86_64). A
|
||||
`set_thread_pointer(addr)` = 44 syscall sets the caller's `thread_pointer` and loads it
|
||||
now. The kernel never touches FS, so there is no swapgs complication.
|
||||
- [x] **Runtime** lays a small per-thread TLS block at the top of each thread's stack
|
||||
(self-pointer at `%fs:0` + scratch slots) and the thread trampoline calls
|
||||
`set_thread_pointer` before any user code — so every spawned thread has a private,
|
||||
switch-stable thread pointer. Reclaimed with the stack.
|
||||
- [x] `-Dtest-case=thread-tls` (`smp: 4`): two threads each write a unique marker to their
|
||||
own `%fs:8` slot and — after both have written — read it back; a shared (non-per-thread)
|
||||
FS base would clobber one and cause cross-talk. Both read their own marker → pass.
|
||||
|
||||
**Gate (met):** `thread-tls` passes (3×); full guardrail 25/25 (the switch-time thread-pointer
|
||||
restore touches every context switch); `zig build`/`zig build test` clean.
|
||||
|
||||
> **Deferred: the Zig `threadlocal` *compiler* layer.** Real `threadlocal` variables need
|
||||
> the ELF **variant-II TLS** surface — `.tdata`/`.tbss` sections + a `PT_TLS` program header
|
||||
> in `user.ld`, a runtime that copies the template with exact negative-offset layout, and
|
||||
> the `.large`-code-model TLS section names — a high-uncertainty lift for a feature with
|
||||
> **no consumer today** (threading.md scopes it "only if a consumer needs it"). What lands
|
||||
> here is the load-bearing piece — the per-thread thread pointer, context-switched — so adding the
|
||||
> compiler layer later is purely runtime+linker work on top, no kernel change. `getCurrentId`
|
||||
> stays the `thread_self` syscall (M6) rather than an fs self-slot (which would need the
|
||||
> main thread's TLS set up in `_start` too).
|
||||
|
||||
**Gate:** `thread-tls` passes; full `thread-*` suite + guardrail green.
|
||||
|
||||
### M11 — `RwLock`, `WaitGroup`, and host-testable sync ✅
|
||||
|
||||
- [x] `runtime.Thread.RwLock` (reader-preferring: `>0` readers / `-1` writer / `0` free,
|
||||
with `lock`/`tryLock`/`unlock` + `lockShared`/`tryLockShared`/`unlockShared`) and
|
||||
`WaitGroup` (`start`/`finish`/`wait`), both on the existing `Mutex`/`Condition`.
|
||||
- [x] A compile-time `Futex` seam gated on `builtin.os.tag == .freestanding`: the futex
|
||||
syscalls on danos, a spin+yield mock off-target (Zig 0.16 has no `std.Thread.Futex`;
|
||||
`wake` is a no-op since the state machines re-check). `thread.zig` is wired into
|
||||
`zig build test`, so `Mutex`/`RwLock`/`WaitGroup` run as **host unit tests** with real
|
||||
`std.Thread` threads (`test` blocks only compile under test).
|
||||
- [x] `-Dtest-case=thread-rwlock` (`smp: 4`): 2 writers set both halves of a value under
|
||||
the exclusive lock while 3 readers check the halves match under the shared lock —
|
||||
zero half-write observations across ~150k reads. Host tests cover the Mutex,
|
||||
RwLock, and WaitGroup state machines.
|
||||
|
||||
**Gate (met):** `zig build test` covers the sync primitives (host threads); `thread-rwlock`
|
||||
passes (3×); full Done gate **26/26** (whole `thread-*` suite + guardrail); `zig build`
|
||||
clean.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a
|
||||
physical-address key so two processes share a futex through a [shared-memory](display-v2.md)
|
||||
region. Not needed for intra-process threads.
|
||||
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||
- **Per-thread signal delivery** — signals stay process-scoped
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **A `pthread`/POSIX surface** — the API is `std.Thread`-shaped Zig, nothing more.
|
||||
- **A real `std.Thread` backend** — arrives with self-hosting
|
||||
([zig-self-hosting.md](zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
swaps the impl under `runtime.Thread`, not the call sites.
|
||||
@@ -0,0 +1,332 @@
|
||||
# Threading: `runtime.Thread`, a std-shaped API over a private thread ABI
|
||||
|
||||
A note on danos **threads** — several tasks sharing one address space — provided by a
|
||||
`runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M11, see
|
||||
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core parallelism,
|
||||
a futex, `Mutex`/`Condition`/`Semaphore`/`RwLock`/`WaitGroup`, `getCurrentId`/`currentCore`,
|
||||
per-thread thread-pointer TLS, thread-safe allocation, and a task reaper that reclaims dead
|
||||
tasks' kernel stacks. Deferred by design (no consumer yet): the Zig `threadlocal`
|
||||
*compiler* layer (the per-thread thread pointer is in place, so it's runtime+linker work on top) and
|
||||
detached-thread user-stack reclaim — see the plan's M9/M10 notes. The analysis is against
|
||||
**Zig 0.16** (the pinned toolchain); `std.Thread`'s internals move between releases, so
|
||||
treat upstream shapes as "0.16.x."
|
||||
|
||||
## The win condition
|
||||
|
||||
A danos service can write
|
||||
|
||||
```zig
|
||||
const t = try runtime.Thread.spawn(.{}, worker, .{ctx});
|
||||
// ... do other work concurrently ...
|
||||
t.join();
|
||||
```
|
||||
|
||||
and get real parallelism across cores — with `runtime.Thread.Mutex`,
|
||||
`runtime.Thread.Condition`, and `runtime.Thread.Semaphore` available for
|
||||
coordination — **without any code path reaching the kernel except through the
|
||||
runtime**. The call sites read exactly like `std.Thread`, so the day danos becomes a
|
||||
real Zig target (see [self-hosting](#the-self-hosting-endgame)) we swap the
|
||||
implementation underneath, not the API above.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **We build `runtime.Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
features*; the implementation underneath is danos-native. See
|
||||
[Why not literal std.Thread](#why-not-literal-stdthread).
|
||||
- **Threads are a narrow, opt-in capability — not the default concurrency tool.** The
|
||||
default for resilience stays **process + IPC** ([resilience.md](resilience.md),
|
||||
[ipc.md](ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
- **Blocking synchronization is futex-backed, never spin-backed.** Waiters sleep in
|
||||
the kernel so an idle core still halts ([halting.md](halting.md)).
|
||||
- **Per-binary opt-in to multi-threaded codegen.** Only a service that asks for
|
||||
threads is built `single_threaded = false`; the rest stay lean and single-threaded.
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/runtime` wrappers, exactly like
|
||||
every other danos syscall ([syscall.md](syscall.md)) — numbers stay renumberable.
|
||||
|
||||
## Why not literal `std.Thread`
|
||||
|
||||
danos's ABI invariant is that the **runtime is the sole holder of the syscall ABI**,
|
||||
and that ABI is private and renumberable ([syscall.md](syscall.md) — "unstable
|
||||
private ABI"). That is a security and evolvability asset: no compiled binary can
|
||||
hardcode a syscall number, and the kernel can renumber freely because only the
|
||||
runtime — rebuilt in lockstep — knows the mapping.
|
||||
|
||||
`std.Thread` is incompatible with that invariant on two counts:
|
||||
|
||||
1. **It selects its backend from `builtin.os.tag`, and issues syscalls directly.**
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../build.zig)), for which
|
||||
`std.Thread` resolves to an unsupported stub that `@compileError`s. Adding a real
|
||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||
cost that buys nothing the native type doesn't.
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../build.zig)
|
||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||
single-threaded. Threads need this flipped per binary regardless.
|
||||
|
||||
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
||||
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
||||
invariant.
|
||||
|
||||
## Where threads fit: the resilience tension
|
||||
|
||||
Threads are in genuine tension with a resilience-first microkernel, and it is worth
|
||||
being explicit so we do not reach for them by reflex.
|
||||
|
||||
The reason danos pays for a microkernel is **fault isolation**
|
||||
([resilience.md](resilience.md)): a component corrupts its own address space, faults,
|
||||
and is **restarted** without touching anyone else — because the boundary *is* the
|
||||
address space. Threads deliberately remove that boundary *within* a process:
|
||||
|
||||
- Threads share one address space, so one thread's stray write corrupts them all —
|
||||
there is no isolation **between** threads.
|
||||
- Threads share fate: a fault in any thread, or a "kill the process" decision, takes
|
||||
down **all** of them. Restartability lives at the process level, not the thread
|
||||
level.
|
||||
- Shared mutable state reintroduces data races — the failure class the
|
||||
isolate-and-message model was chosen to avoid.
|
||||
|
||||
**Therefore:** the default answer to "make X concurrent" stays *another process over
|
||||
IPC* (isolated, independently restartable) or a single event loop with several
|
||||
message sources. Reach for a thread only inside **one** service that needs genuine
|
||||
**shared-memory, low-latency parallelism** and can accept intra-service fate-sharing —
|
||||
e.g. a compositor splitting tile compositing across cores, where per-tile IPC would be
|
||||
too chatty. "Input on one thread, display on another" is *not* that case; it wants two
|
||||
processes. The isolation boundary stays at process granularity.
|
||||
|
||||
## The API surface (mirrors `std.Thread`)
|
||||
|
||||
Lives in `library/runtime/thread.zig`, re-exported as `runtime.Thread`.
|
||||
|
||||
```zig
|
||||
pub const Thread = struct {
|
||||
pub const Id = u32; // the kernel task id
|
||||
pub const SpawnConfig = struct {
|
||||
stack_size: usize = default_stack_size,
|
||||
allocator: ?std.mem.Allocator = null, // for the closure + stack bookkeeping
|
||||
};
|
||||
pub const SpawnError = error{ OutOfMemory, ThreadQuotaExceeded, SystemResources };
|
||||
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread;
|
||||
pub fn join(self: Thread) void; // block until the thread ends, reclaim its stack
|
||||
pub fn detach(self: Thread) void; // give up the right to join; kernel reclaims on exit
|
||||
pub fn getCurrentId() Id;
|
||||
pub fn yield() void; // -> existing `yield` syscall
|
||||
|
||||
pub const Mutex = struct { pub fn lock(*Mutex) void; pub fn tryLock(*Mutex) bool; pub fn unlock(*Mutex) void; };
|
||||
pub const Condition = struct { pub fn wait(*Condition, *Mutex) void; pub fn timedWait(*Condition, *Mutex, u64) error{Timeout}!void; pub fn signal(*Condition) void; pub fn broadcast(*Condition) void; };
|
||||
pub const Semaphore = struct { pub fn wait(*Semaphore) void; pub fn post(*Semaphore) void; };
|
||||
pub const Futex = struct { pub fn wait(*const atomic.Value(u32), u32) void; pub fn timedWait(...) error{Timeout}!void; pub fn wake(*const atomic.Value(u32), u32) void; };
|
||||
// RwLock / ResetEvent / WaitGroup follow the same pattern, added as needed.
|
||||
};
|
||||
```
|
||||
|
||||
Deviations from `std.Thread`, called out honestly:
|
||||
|
||||
- **The thread function's return value is discarded** (as `std.Thread.join` returns
|
||||
`void`). Return data through shared state or a `Semaphore`/`Condition`, not the
|
||||
return.
|
||||
- `getCpuCount()` maps to the existing SMP core count ([smp.md](smp.md)); a service
|
||||
rarely needs it.
|
||||
|
||||
## Kernel primitives (new private syscalls)
|
||||
|
||||
Four new entries extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shared_memory_physical = 36`, each with a `library/runtime` wrapper:
|
||||
|
||||
| Syscall | Signature | Purpose |
|
||||
|---|---|---|
|
||||
| `thread_spawn` | `(entry, stack_top, arg) -> tid` | create a task sharing the **caller's** address space |
|
||||
| `thread_exit` | `(stack_base, stack_len)` | end the calling thread; hand back its stack range for reclaim |
|
||||
| `futex_wait` | `(addr, expected, timeout_ns) -> status` | block if `*addr == expected`, until woken or timeout |
|
||||
| `futex_wake` | `(addr, count) -> woken` | wake up to `count` waiters on `addr` |
|
||||
|
||||
Plus one invariant change with no new syscall: **address-space reference counting**.
|
||||
|
||||
## Mechanics
|
||||
|
||||
### Address-space reference counting
|
||||
|
||||
Today an address space is 1:1 with a task: `spawnUserLocked` records `address_space` on the
|
||||
Task, and teardown does `destroyAddressSpace(t.address_space)` when **any** user task exits
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
one `address_space`, so the first to exit would rip the address space out from under its
|
||||
siblings.
|
||||
|
||||
Fix: a small refcount keyed by the address-space root (`createAddressSpace` in
|
||||
[process.zig](../system/kernel/process.zig) sets it to 1). `thread_spawn` increments
|
||||
it; task teardown decrements and only calls `destroyAddressSpace` at **zero**. All of
|
||||
this is already under the big kernel lock, so no new locking. This is the one piece
|
||||
that must land and be proven before anything shares an address space.
|
||||
|
||||
### `thread_spawn` and the trampoline
|
||||
|
||||
The scheduler already accepts an arbitrary `address_space` and does **not** smuggle values
|
||||
through registers — `startUserTask` reads the entry/stack from the Task and
|
||||
`jumpToUser`s ([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread
|
||||
path clean:
|
||||
|
||||
1. The runtime's `spawn` `mmap`s a stack (syscall `4`), heap-allocates a closure —
|
||||
`{ fn_ptr, args_tuple, completion }`, the std "Instance" pattern — and writes the
|
||||
closure pointer to the **top word of the new stack**.
|
||||
2. It calls `thread_spawn(entry = &threadTrampoline, stack_top, arg = closure_ptr)`.
|
||||
The kernel calls the same `spawnUserLocked` path with the **caller's address space**
|
||||
(refcount++), `entry`, and `user_sp = stack_top`.
|
||||
3. `threadTrampoline` (a small runtime shim) reads the closure off its stack, calls
|
||||
the user function, then calls `thread_exit`. No new register ABI — the closure
|
||||
pointer rides the stack the runtime set up, mirroring how `startUserTask` avoids
|
||||
register smuggling.
|
||||
|
||||
Unlike a process start, there is **no** System V argc/argv/auxv block
|
||||
([sysv.md](sysv.md)) — a thread stack carries only the closure pointer.
|
||||
|
||||
### Lifetime: exit, join, detach, stack reclaim
|
||||
|
||||
- **`thread_exit`** marks the task dead and hands the kernel the thread's user-stack
|
||||
range. The kernel reaps the task on the scheduler (already running on a *kernel*
|
||||
stack, so it can safely unmap the user stack), decrements the address-space refcount, and
|
||||
frees the task slot.
|
||||
- **`join` — Stage 1** reuses the existing exit-notification machinery
|
||||
([process-lifecycle.md](process-lifecycle.md)): `spawn` passes a per-thread
|
||||
`exit_endpoint`, and `join` blocks in `ipc_reply_wait` until the child-exit
|
||||
notification for that `tid` arrives, then `munmap`s the stack. No futex needed to
|
||||
land spawn/join.
|
||||
- **`join` — Stage 2 refinement** migrates to the std shape: a `completion` word in
|
||||
the closure that `thread_exit`'s trampoline `futex_wake`s and `join` `futex_wait`s
|
||||
on — dropping the per-thread endpoint. Kept as a refinement so Stage 1 ships first.
|
||||
- **`detach`** relinquishes the join right; the kernel reclaims the stack and slot on
|
||||
`thread_exit` (a detached thread's stack range is unmapped by the reaper, since no
|
||||
joiner will).
|
||||
|
||||
### Futex, and the sync primitives on top
|
||||
|
||||
`futex_wait`/`futex_wake` are the one blocking primitive; `Mutex`, `Condition`, and
|
||||
`Semaphore` are ordinary user-space state machines over an `atomic.Value(u32)` that
|
||||
call the futex wrappers on the slow path — the same construction `std.Thread` uses,
|
||||
so the algorithms port directly.
|
||||
|
||||
Keying: threads share an address space, so a **virtual address within that address space**
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`.
|
||||
Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a
|
||||
deliberate forward door: it lets two *processes* share a futex through an
|
||||
[shared-memory](display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-address-space key and note the physical-key upgrade.
|
||||
|
||||
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||
other work or `hlt` ([halting.md](halting.md)). This is why futex is a locked
|
||||
decision, not a "maybe later."
|
||||
|
||||
### TLS and `getCurrentId`
|
||||
|
||||
danos sets up no thread-pointer TLS today (fine under `single_threaded`). Two scoped needs:
|
||||
|
||||
- **`getCurrentId`** returns the kernel task id — either a trivial syscall or, better,
|
||||
a value the runtime stashes in a per-thread control block.
|
||||
- **`threadlocal` variables** need a real per-thread TLS block and the thread pointer set per
|
||||
thread. `thread_spawn` sets the thread pointer to a runtime-allocated per-thread block; full
|
||||
`threadlocal` support is Stage 3, only if a consumer needs it. Nothing in the core
|
||||
spawn/join/mutex path requires `threadlocal`.
|
||||
|
||||
### Build: multi-threaded codegen, opt-in
|
||||
|
||||
`addUserBinary` gains a `threaded: bool = false` parameter; when set it builds that
|
||||
binary `single_threaded = false` so atomics and (later) TLS are real. Threads and
|
||||
atomics are unsound in a `single_threaded` image, so a binary must opt in **before**
|
||||
it may call `runtime.Thread.spawn`. Everyone else stays single-threaded and lean.
|
||||
|
||||
## Interaction with the rest of the kernel
|
||||
|
||||
- **Scheduler / SMP** ([scheduling.md](scheduling.md), [smp.md](smp.md)): a thread is
|
||||
just another `Task` with an `address_space` shared with its siblings; the existing
|
||||
per-core ready queues, priorities, and affinity apply unchanged. Threads of one
|
||||
process can run on different cores simultaneously — that is the point.
|
||||
- **Halting** ([halting.md](halting.md)): futex-parked waiters keep the "idle core
|
||||
halts" property intact under lock contention — no busy-wait.
|
||||
- **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): killing a process
|
||||
must kill *all* its threads and only then drop the last address-space ref. The kill path
|
||||
already targets a process; it fans out to every task on that address space.
|
||||
- **Resilience** ([resilience.md](resilience.md)): a faulting thread kills its whole
|
||||
process (shared fate). The supervisor restarts the **process**, which respawns its
|
||||
threads from a known-good state — restart granularity stays the process.
|
||||
- **IPC — two consequences threads forced ([ipc.md](ipc.md)):**
|
||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||
A thread that needs to reach an endpoint another thread owns looks it up
|
||||
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
||||
This is how the display's mouse-listener thread reaches the compositor loop's endpoint
|
||||
to poke it awake (docs/display.md).
|
||||
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
||||
`create_ipc_endpoint`/`ipc_register`/`ipc_lookup` allocate from the kernel heap and
|
||||
mutate the global service registry, endpoint refcounts, and handle tables. Those paths
|
||||
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
||||
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
||||
`send` already did — the kernel heap has no lock of its own yet (heap.zig: "a lock comes
|
||||
with threads/SMP"), so the big lock is what keeps its callers serialized.
|
||||
|
||||
## Build-out plan (staged, each gate serial-checkable)
|
||||
|
||||
The ordered, `/loop`-runnable milestones live in
|
||||
**[threading-plan.md](threading-plan.md)** (shaped like
|
||||
[display-v2-plan.md](display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||
|
||||
- **Stage 0 — address-space refcount.** Refcount on the address-space root; teardown destroys
|
||||
at zero. No API yet; nothing shares an address space, so refcount is 1 everywhere.
|
||||
*Gate:* the full QEMU suite stays green (no regression) — proves the reframing is
|
||||
invisible until used.
|
||||
- **Stage 1 — spawn / join / detach.** `thread_spawn` + `thread_exit`, the trampoline,
|
||||
stacks via `mmap`, join over the exit-endpoint, the `threaded` build flag.
|
||||
*Gate:* `-Dtest-case=thread-spawn` — a threaded test service spawns N threads that
|
||||
each `@atomicRmw`-increment a shared counter, the parent joins all N, and asserts
|
||||
the total is exactly N × iterations. Runs `smp` (multi-core) to prove real
|
||||
parallelism.
|
||||
- **Stage 2 — blocking synchronization.** `futex_wait`/`futex_wake` + `Futex`,
|
||||
`Mutex`, `Condition`, `Semaphore`; optionally migrate join to a futex completion
|
||||
word. *Gate:* `-Dtest-case=thread-mutex` — a bounded producer/consumer over a
|
||||
`Mutex` + `Condition` moves K items with no lost wakeups and no busy-wait (assert
|
||||
the consumer blocked, e.g. via a low idle tick count).
|
||||
- **Stage 3 — polish.** Per-thread TLS / thread pointer and `threadlocal` (only if a
|
||||
consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired
|
||||
into [test/qemu_test.py](../test/qemu_test.py).
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym
|
||||
abbreviations, kebab-case file names, no `Co-Authored-By` trailers. New syscalls
|
||||
extend [abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper
|
||||
([syscall.md](syscall.md)). `runtime.Thread` is a first-class runtime module, the same
|
||||
way `runtime.process` ([process-lifecycle.md](process-lifecycle.md)) and `runtime.ipc`
|
||||
are — user code never names a syscall.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- **No preemptive user-space signals delivered to a specific thread.** Signals stay
|
||||
process-scoped ([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **No thread priorities distinct from the process.** Threads inherit the process
|
||||
priority; per-thread priority is a later question if it ever earns its keep.
|
||||
- **No cross-process shared-memory futex yet** — the physical-address key leaves the
|
||||
door open, but the first cut is private-per-address-space.
|
||||
- **No `pthread`/POSIX surface.** The API is `std.Thread`-shaped Zig, nothing more.
|
||||
|
||||
## The self-hosting endgame
|
||||
|
||||
When danos becomes a real Zig target and we (eventually) add a danos backend to std
|
||||
([zig-self-hosting.md](zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
same kernel primitives — the danos `std.Thread.Impl` would call the very
|
||||
`thread_spawn`/`futex_*` wrappers `runtime.Thread` already uses. Because
|
||||
`runtime.Thread` was built API-compatible from day one, that transition swaps the
|
||||
implementation, not a single call site. Designing to the std shape now is what makes
|
||||
the later self-hosting lift cheap.
|
||||
|
||||
## Further reading
|
||||
|
||||
- [scheduling.md](scheduling.md), [smp.md](smp.md) — the task model these threads join.
|
||||
- [resilience.md](resilience.md), [vision.md](vision.md) — why isolation is the default
|
||||
and threads are the exception.
|
||||
- [syscall.md](syscall.md), [ipc.md](ipc.md) — the private ABI and the messaging model
|
||||
threads sit beside.
|
||||
- [halting.md](halting.md) — the idle/halt property futex-backed blocking preserves.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the target this bends toward.
|
||||
+111
@@ -0,0 +1,111 @@
|
||||
# USB hubs (M22)
|
||||
|
||||
A hub is USB **bus infrastructure**, not an application peripheral, so hub
|
||||
topology is handled **inside the `usb-xhci-bus` driver** — the process that owns
|
||||
the controller's device slots and contexts. A device behind a hub is not reached
|
||||
by any hub-specific software path: it is reached by the **controller**,
|
||||
programmed with a *route string* in its slot context. Route strings and slot
|
||||
contexts are xHCI hardware concepts that only exist inside the controller driver,
|
||||
so that is where hub handling belongs. Class drivers (HID, storage) stay separate
|
||||
and unaware — the hub is transparent to them; a keyboard behind a hub reaches the
|
||||
same `usb-hid-keyboard` driver as one on a root port.
|
||||
|
||||
This is a deliberate scoping choice, not a microkernel compromise: the USB *bus*
|
||||
driver handles USB *bus* topology. The alternative — a separate `usb-hub`
|
||||
class-driver process plus a cross-process enumeration protocol — would only
|
||||
shuttle the bus's own topology state (slot ids, route strings, TT linkage) out to
|
||||
another process and back, since the hub driver cannot build a slot context
|
||||
itself.
|
||||
|
||||
## The compound-hub reality
|
||||
|
||||
A USB 3.0 hub is physically **two hubs** sharing each connector: a SuperSpeed hub
|
||||
and a USB 2.0 companion hub, enumerated as **separate devices on separate root
|
||||
ports**. A full- or low-speed device plugged into a USB 3.0 hub attaches to the
|
||||
**USB 2.0 companion**, not the SuperSpeed hub. So supporting full-speed devices
|
||||
(keyboards, mice) behind a hub means driving the USB 2.0 companion and handling
|
||||
**transaction translators** — there is no SuperSpeed-only shortcut that reaches a
|
||||
full-speed keyboard.
|
||||
|
||||
## Slot-context fields for a downstream device
|
||||
|
||||
`buildAddressInputContext` fills the Slot Context. Today it hardcodes route 0 and
|
||||
the root-hub port. A downstream device additionally needs:
|
||||
|
||||
- **Route String** (Slot Context dword 0, bits 19:0) — 5 tiers × 4 bits, each
|
||||
tier the downstream hub-port number. Composed as
|
||||
`route = (parent_route << 4) | hub_port`, capped at the xHCI 5-tier max.
|
||||
- **Root Hub Port Number** (dword 1, bits 23:16) — the *root* port the whole hub
|
||||
chain hangs off, inherited from the parent hub (not the hub's own port number).
|
||||
- **Speed** (dword 0, bits 23:20) — read from the hub's downstream port status
|
||||
after reset, not assumed.
|
||||
- **Parent Hub Slot ID** (dword 2, bits 7:0) + **Parent Port Number** (dword 2,
|
||||
bits 13:8) — the **transaction translator**: set when a full/low-speed device
|
||||
sits behind a high-speed hub, so the controller routes split transactions
|
||||
through that hub's TT. For a multi-TT hub, **MTT** (Slot Context dword 0 bit
|
||||
25) is set and the TT port is the device's own hub port.
|
||||
|
||||
## Detection: the status-change interrupt endpoint
|
||||
|
||||
A hub has one interrupt IN endpoint that returns a **port-status-change bitmap**
|
||||
(bit N set = port N changed). The bus arms an interrupt transfer on it (reusing
|
||||
the controller's existing interrupt-endpoint machinery, but serviced
|
||||
**in-process** — no class-driver subscription IPC), and on each report:
|
||||
|
||||
1. For each changed port, `GET_STATUS` (hub class request) reads the port's
|
||||
connect/enable/reset state and speed, and `CLEAR_FEATURE(C_PORT_*)`
|
||||
acknowledges the change.
|
||||
2. On a **connect**: `SET_FEATURE(PORT_RESET)`, wait for reset-complete via a
|
||||
later status-change report, read the enabled speed, then `setupDevice` with
|
||||
the composed route string / root port / TT fields, `enumerate`, and register
|
||||
the interfaces — exactly the existing path, recursing if the new device is
|
||||
itself a hub.
|
||||
3. On a **disconnect**: tear down the downstream device (report each interface
|
||||
`ChildRemoved`, Disable Slot) — the B3 teardown path, keyed by the device's
|
||||
route rather than a root port.
|
||||
|
||||
## Hub setup (once, when the hub enumerates)
|
||||
|
||||
When the bus scan (or a hot-plug bring-up) finds a device of class 9:
|
||||
|
||||
1. Read the **hub descriptor** (class GET_DESCRIPTOR, type 0x2A for a USB 3.0
|
||||
hub / 0x29 for USB 2.0) → downstream port count, characteristics.
|
||||
2. For a USB 3.0 hub, `SET_FEATURE(BH_PORT_RESET)` semantics and the depth
|
||||
(`SET_HUB_DEPTH`) so the hub knows its tier for route-string forwarding.
|
||||
3. `SET_FEATURE(PORT_POWER)` each downstream port.
|
||||
4. Configure the hub's slot as a hub: **Hub** bit (Slot Context dword 0 bit 26),
|
||||
**Number of Ports** (dword 1, bits 31:24), **TT Think Time** and **MTT** for a
|
||||
USB 2.0 multi-TT hub — via an Evaluate/Configure Endpoint on the hub's slot.
|
||||
5. Arm the status-change interrupt endpoint.
|
||||
|
||||
## Testing
|
||||
|
||||
QEMU's `usb-hub` is a USB 2.0 single-TT hub. A **static boot topology**
|
||||
(`-device usb-hub,bus=xhci.0,id=h -device usb-kbd,bus=h.0`) presents the
|
||||
downstream device connected from the start, so the bus reads it on the first
|
||||
status-change report — exercising the full path (hub setup, TT slot context,
|
||||
downstream enumerate, class-driver bind) without needing a hot-plug event. A new
|
||||
`usb-hub` QEMU case asserts the hub enumerates, the downstream keyboard
|
||||
enumerates behind it, and `usb-hid-keyboard` binds.
|
||||
|
||||
Real-hardware validation (the user's SuperSpeed Genesys hub + full-speed
|
||||
keyboard/mouse on its USB 2.0 companion) is flagged separately — the compound
|
||||
USB 3.0 hub path is not modelled by QEMU's USB 2.0 hub.
|
||||
|
||||
## Milestones (all complete)
|
||||
|
||||
- **B4a** ✓ — hub recognition + setup: detect class 9 in the scan, read the hub
|
||||
descriptor, configure the slot as a hub, power downstream ports, log the
|
||||
topology.
|
||||
- **B4b** ✓ — downstream enumeration: the in-process status-change subscription,
|
||||
port reset, Address Device with route string + root port + TT fields,
|
||||
enumerate + register. A full-speed keyboard behind a USB2 hub binds
|
||||
`usb-hid-keyboard` in QEMU.
|
||||
- **B4c** ✓ — disconnect teardown (recursive: a hub takes its subtree with it)
|
||||
and hub-behind-hub recursion (route strings compose across tiers). QEMU's hub
|
||||
*does* raise downstream status changes, so both connect and disconnect are
|
||||
harness-tested (`usb-hub`, `usb-hub-nested`, `usb-hub-unplug`).
|
||||
|
||||
Real-hardware validation of the user's SuperSpeed Genesys hub with full-speed
|
||||
devices on its USB 2.0 companion remains pending — QEMU's USB 2.0 hub does not
|
||||
model the compound USB 3.0 hub.
|
||||
+209
@@ -0,0 +1,209 @@
|
||||
# The vDSO — the public system-call boundary
|
||||
|
||||
> **Status:** design note, not built. The runtime today issues raw `syscall`
|
||||
> instructions from `library/runtime/system-call.zig` using the numbers in
|
||||
> `system/abi.zig`. This note designs the layer that replaces that arrangement:
|
||||
> a **kernel-supplied, C-ABI entry library** mapped into every process — the
|
||||
> only supported way into the kernel — so the raw numbers can stay private,
|
||||
> be renumbered at will, and eventually be randomised per boot.
|
||||
|
||||
## Why: the ABI danos promises, and the one it doesn't
|
||||
|
||||
`system/abi.zig` is the **private** kernel ↔ runtime contract. Its header says
|
||||
so: the numbers are an implementation detail the runtime hides and may
|
||||
renumber, the same split as libSystem over the XNU syscalls on macOS or win32
|
||||
over the NT syscalls on Windows. Linux — with its world-visible, frozen
|
||||
syscall table — is the outlier, not the norm.
|
||||
|
||||
That stance has consequences the moment binaries exist that we don't rebuild
|
||||
ourselves:
|
||||
|
||||
1. **Third-party binaries** (docs/zig-self-hosting.md) must keep working across
|
||||
kernel updates. If they contain raw `syscall` instructions with today's
|
||||
numbers baked in, every renumbering breaks the world — the ABI would be
|
||||
*de facto* public no matter what the header says. Go on macOS made exactly
|
||||
this mistake: it issued XNU syscalls directly instead of going through
|
||||
libSystem, and macOS updates repeatedly broke every Go binary until Go
|
||||
switched to the library like everyone else.
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the `runtime`
|
||||
module. The public boundary has to be expressible in the one calling
|
||||
convention every language speaks: the C ABI.
|
||||
3. **Randomised syscall numbers** — a hardening option we want open — only
|
||||
work if no user binary anywhere knows a number at build time. The binding
|
||||
must happen at *load time*, from something the kernel controls.
|
||||
|
||||
All three point at the same well-known shape: a **vDSO** (virtual dynamic
|
||||
shared object). The kernel carries a small blob of user-mode code, maps it
|
||||
into every process at spawn, and that blob — not the application — contains
|
||||
the `syscall` instructions. Fuchsia works exactly this way: its vDSO is the
|
||||
*only* kernel entry, version-matched by construction because the kernel itself
|
||||
injects it. Because the kernel and the blob ship as one artifact, there is
|
||||
**no version skew, no loader, no search path, and no shared file on disk** —
|
||||
which is what makes this the resilient way to have a private ABI
|
||||
(docs/resilience.md), where a conventional `ld.so` + `/lib/libdanos.so`
|
||||
arrangement would add a loader to every spawn and a single shared point of
|
||||
failure.
|
||||
|
||||
The public danos ABI then has exactly two layers, neither of which is
|
||||
`abi.zig`:
|
||||
|
||||
| Layer | Contract | Spoken by |
|
||||
|-------|----------|-----------|
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (`runtime.system` for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
|
||||
Everything above those — the heap, `runtime.fs`, the service harness — is
|
||||
per-language convenience, compiled into each binary from source, exactly as
|
||||
today. Nothing about the Zig runtime's shape changes; it just stops being the
|
||||
*only* door.
|
||||
|
||||
## The blob
|
||||
|
||||
A single copy of the vDSO code lives in the kernel image (built by
|
||||
`build.zig` as a tiny freestanding object, embedded like the AP trampoline).
|
||||
At boot the kernel finalises it once — this is where randomised numbers would
|
||||
be patched in — and thereafter maps the **same physical pages** read-execute
|
||||
into every process's address space. The blob is:
|
||||
|
||||
- **Position-independent.** It is mapped at a per-process randomised base, so
|
||||
it must be PIC (rip-relative addressing only — no relocations to process).
|
||||
- **Stateless and re-entrant.** No writable data. Anything stateful belongs to
|
||||
the process, not the vDSO.
|
||||
- **Architecture-specific.** The x86-64 blob wraps `syscall`; an aarch64 blob
|
||||
wraps `svc #0`. It lives beside the other per-architecture kernel sources
|
||||
(`system/kernel/architecture/<arch>/`), selected the same way the
|
||||
`architecture` module is (docs/arch.md).
|
||||
|
||||
### Shape: a function table, not an ELF
|
||||
|
||||
A real `.so` with a dynamic symbol table is the conventional vDSO shape, but
|
||||
linking against one at load time needs a dynamic linker in every binary —
|
||||
machinery danos deliberately doesn't have. Instead the v1 shape is the
|
||||
simplest thing that is still a stable contract — a **function-pointer table**
|
||||
at the vDSO base:
|
||||
|
||||
```
|
||||
offset 0 u64 magic 'danosVDS' — a mapped-the-wrong-thing guard
|
||||
offset 8 u64 api_level incremented when the table grows
|
||||
offset 16 u64 count number of table entries that follow
|
||||
offset 24 u64 table[count] function pointers into the vDSO's own code
|
||||
```
|
||||
|
||||
Table *indices* are the public constants (published in a C header,
|
||||
`danos.h`), assigned once and append-only — the same discipline the IPC
|
||||
protocols use for operation values. The pointers point at stubs inside the
|
||||
blob; what those stubs put in `rax` is nobody's business but the kernel's.
|
||||
A language shim binds in one step: read the base from the init block, check
|
||||
the magic, keep the table pointer. Feature detection for a binary built
|
||||
against older headers is `count`/`api_level` — a kernel never removes or
|
||||
reorders entries.
|
||||
|
||||
(If danos ever grows a real dynamic linker, the same blob can additionally
|
||||
present an ELF `dynsym` without breaking the table — Fuchsia's vDSO is
|
||||
likewise both a mappable blob and a linkable `.so`. That is a later
|
||||
convenience, not a requirement.)
|
||||
|
||||
### Delivery: the auxiliary vector
|
||||
|
||||
The kernel already builds a System V entry block — argc, argv, envp
|
||||
terminator, **auxiliary vector** — on every new process's stack
|
||||
(`buildEntryStack`, read by `runtime.start`). The vDSO base rides in a new
|
||||
auxv entry, exactly Linux's `AT_SYSINFO_EHDR` move. No new syscall, no magic
|
||||
address, and a language shim finds it the same portable way on every
|
||||
architecture.
|
||||
|
||||
## The function surface
|
||||
|
||||
One table entry per kernel call, C ABI (System V AMD64), names prefixed
|
||||
`danos_`. The current `SystemCall` set maps directly; integer arguments and
|
||||
returns are `u64`, errors return as negative values exactly as today.
|
||||
|
||||
The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||
(virtual_address + physical_address), `msi_bind` (address + data), `shared_memory_create` (virtual_address + handle) —
|
||||
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||
C-ABI spelling of the existing convention, at zero cost.
|
||||
|
||||
Grouped as `abi.zig` groups them:
|
||||
|
||||
| Group | Functions |
|
||||
|-------|-----------|
|
||||
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shared_memory_create`, `danos_shared_memory_map`, `danos_shared_memory_physical` |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
||||
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||
| diagnostics | `danos_debug_write` (leveled, kernel-stamped records), `danos_klog_read`, `danos_klog_status` |
|
||||
| filesystem naming | `danos_fs_resolve`, `danos_fs_node`, `danos_fs_mount`, `danos_fs_unmount` (naming only — file DATA still crosses the vfs-protocol IPC, see below) |
|
||||
|
||||
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
||||
ids, `page_size`, the IPC message maximum — move to the public header too:
|
||||
they are wire values a Rust program needs verbatim. What stays private in
|
||||
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
||||
numbers and the trap convention.
|
||||
|
||||
## Enforcement, and an honest threat model
|
||||
|
||||
Renumbering only has teeth if the kernel **refuses syscalls that don't come
|
||||
from the vDSO**. The check is cheap: on kernel entry, the saved user `rip`
|
||||
must lie inside the calling process's vDSO mapping; otherwise the process is
|
||||
killed with a fault-class exit reason (its supervisor restarts or gives up,
|
||||
docs/process-lifecycle.md — a foreign-syscall attempt is a bug or an attack,
|
||||
never something to limp past). Fuchsia enforces exactly this.
|
||||
|
||||
What this buys, precisely:
|
||||
|
||||
- **ABI freedom** — the real prize. The numbers can change per release or per
|
||||
boot and nothing outside the kernel image cares. The private ABI stays
|
||||
actually private, permanently.
|
||||
- **A single audited chokepoint** for kernel entry, per process, at a
|
||||
randomised address.
|
||||
- **Raised bar for exploits**: shellcode can't issue a hard-coded `syscall`;
|
||||
it must first discover the per-process vDSO base (ASLR) and call through
|
||||
it.
|
||||
|
||||
What it does *not* buy: an attacker with arbitrary code execution in a
|
||||
process can still *call* the vDSO functions — they are mapped executable in
|
||||
that process, and return-oriented chains reach them. Syscall randomisation is
|
||||
hardening, not a security boundary; the security boundary remains the
|
||||
capability model (what the process's endpoints and device claims let it do).
|
||||
It is worth building anyway — for the ABI freedom first and the hardening
|
||||
second — but the design should never be sold as more than that.
|
||||
|
||||
## Migration
|
||||
|
||||
Phased so every step ships alone (the M-milestone discipline):
|
||||
|
||||
1. **The blob + the table.** Build the vDSO, map it at spawn, deliver the
|
||||
base via auxv. `runtime.system-call.zig` binds through the table when the
|
||||
auxv entry is present, falls back to raw `syscall` when absent — the whole
|
||||
tree keeps booting during the transition.
|
||||
2. **Cut the runtime over.** Delete the raw stubs; `runtime` no longer
|
||||
imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
kernel-internal). The QEMU suite passing proves the table carries the
|
||||
whole system.
|
||||
3. **Enforce + randomise.** Add the `rip`-range check, then per-boot number
|
||||
randomisation patched into the blob at kernel init. A test boots with
|
||||
randomisation on and runs the full suite.
|
||||
4. **The other languages.** Publish `danos.h`; a Rust `danos-sys` crate wraps
|
||||
the table. This is also the seam `std.os.danos` calls through when the Zig
|
||||
self-hosting fork lands (docs/zig-self-hosting.md) — the vDSO is what
|
||||
makes that seam stable across kernel versions.
|
||||
|
||||
## What deliberately stays out
|
||||
|
||||
- **No dynamic linker, no `/lib/*.so`.** The vDSO is kernel-injected precisely
|
||||
so danos binaries can stay fully static above it. Sharing *library code*
|
||||
across processes stays what it is today: a service behind IPC, or source
|
||||
compiled into each binary.
|
||||
- **No file/device I/O in the vDSO.** The microkernel line doesn't move: the
|
||||
vDSO wraps the same deliberately tiny table (docs/syscall.md). The kernel
|
||||
resolves file NAMES (`fs_resolve` — the mount table moved in-kernel), but
|
||||
file data is still the filesystem server's business over the vfs-protocol
|
||||
IPC; the kernel never blocks on a userspace filesystem.
|
||||
- **No fast-path user-mode implementations yet.** Linux's vDSO exists mostly
|
||||
to answer `gettimeofday` without a kernel entry. `danos_clock` could one
|
||||
day read the calibrated TSC in user mode the same way — the blob is where
|
||||
such an optimisation would live — but that is an optimisation, not part of
|
||||
this design's contract.
|
||||
@@ -0,0 +1,178 @@
|
||||
# The VFS wire protocol
|
||||
|
||||
> **Status:** built and spoken today between `runtime.fs` (the client) and the
|
||||
> filesystem BACKENDS (the FAT server). The mount router lives in the
|
||||
> **kernel** (`system/kernel/vfs.zig`): `fs_resolve` routes a path and either
|
||||
> serves it directly (the read-only /system initrd mount, via `fs_node`) or
|
||||
> redirects the caller to the owning backend's endpoint plus the rewritten
|
||||
> mount-relative path — after which the client speaks THIS protocol to the
|
||||
> backend, unchanged. The Zig source of truth is `system/vfs-protocol.zig`
|
||||
> (the `vfs-protocol` module), whose unit tests pin the sizes and values
|
||||
> below. This page is the **language-neutral wire specification** of that
|
||||
> contract — what a Rust or C client implements ([vdso.md](vdso.md) explains
|
||||
> why the IPC protocols, not the syscall numbers, are danos's public ABI).
|
||||
|
||||
## Transport
|
||||
|
||||
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
|
||||
docs/ipc.md): the client sends one message and blocks; the server replies
|
||||
with one message. The endpoint comes from the kernel's `fs_resolve` — which
|
||||
also hands back the path rewritten relative to the mount — not from a
|
||||
registry lookup. (Service id 1, the old userspace router, is retired.)
|
||||
|
||||
- A message is at most **256 bytes** (`message_maximum`).
|
||||
- A request is a fixed 32-byte **Request** header followed by an inline
|
||||
payload of at most **224 bytes** (`maximum_payload`) — a path, or write
|
||||
bytes. There is no multi-message request: paths and single reads/writes
|
||||
must fit, and larger transfers loop (see *read* / *write*).
|
||||
- A reply is a fixed 24-byte **Reply** header followed by an inline payload —
|
||||
read bytes, a `FileStatus`, or a `DirectoryEntry`.
|
||||
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||
(`extern struct`), offsets given below so nothing need be inferred.
|
||||
|
||||
The kernel resolves NAMES (the mount table) but never parses these
|
||||
messages — it moves the bytes; file state is entirely the backend's affair.
|
||||
With clients holding backend node ids directly, a backend records each open
|
||||
handle's owner and sweeps a dead client's handles via the published process
|
||||
exit events.
|
||||
|
||||
## Request header — 32 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `operation` | an **Operation** value (below) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the server-side open-node id from a prior `open`; 0 for path-based operations |
|
||||
| 16 | 8 | `offset` | byte position for read/write; entry index (cursor) for readdir; else 0 |
|
||||
| 24 | 4 | `len` | payload length for path/write operations; requested byte count for read |
|
||||
| 28 | 4 | `flags` | open flags (below); else 0 |
|
||||
|
||||
## Reply header — 24 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `status` | **0 = success**, negative = failure (signed) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the new open-node id (for `open`); else 0 |
|
||||
| 16 | 4 | `len` | reply payload length in bytes |
|
||||
| 20 | 4 | — | padding |
|
||||
|
||||
On failure the router replies `status = -1`; a mounted backend's negative
|
||||
status is forwarded to the client verbatim. A richer errno vocabulary is
|
||||
future work — clients must treat *any* negative status as failure, not match
|
||||
on -1.
|
||||
|
||||
## Operations
|
||||
|
||||
Values are append-only and never renumbered (the same evolution rule every
|
||||
danos protocol follows); an unrecognised operation gets a `status = -1`
|
||||
reply.
|
||||
|
||||
| value | operation | request payload | reply |
|
||||
|------:|-----------|-----------------|-------|
|
||||
| 0 | `open` | the path (`len` = its length), `flags` as below | `node` = open-node id |
|
||||
| 1 | `close` | — (`node` set) | status only |
|
||||
| 2 | `read` | — (`node`, `offset`, `len` = wanted count) | `len` bytes read, payload = the bytes; `len` 0 at end of file |
|
||||
| 3 | `write` | the bytes (`node`, `offset`, `len` = count) | `len` = bytes accepted (may be short — loop) |
|
||||
| 4 | `status` | — (`node` set) | payload = **FileStatus** (24 bytes) |
|
||||
| 5 | `readdir` | — (`node` = a directory, `offset` = cursor) | payload = one **DirectoryEntry** + name; `len` 0 at end |
|
||||
| 6 | `mount` | the mount-point path; the backend endpoint rides as the call's **capability** | status only |
|
||||
| 7 | `unmount` | the mount-point path | status only |
|
||||
| 8 | `mkdir` | the path | status only |
|
||||
| 9 | `unlink` | the path | status only |
|
||||
| 10 | `rename` | old path, one `0x00`, new path (`len` = total) | status only |
|
||||
|
||||
Notes per operation:
|
||||
|
||||
- **open** — paths are absolute (`/mnt/usb/notes.txt`) or bare names
|
||||
(`greeting`); bare names resolve in the VFS's flat ramfs, absolute paths
|
||||
route through the mount table (below). The returned `node` is an id in the
|
||||
*router's* open table; clients never see a backend's own ids.
|
||||
- **read / write** — a single exchange moves at most 224 bytes
|
||||
(`maximum_payload`); the client loops, advancing `offset` by the returned
|
||||
`len`, until done (read) or the slice is written (write). A `write` reply
|
||||
shorter than requested is progress, not an error; a `len` of 0 means no
|
||||
forward progress — stop rather than spin.
|
||||
- **readdir** — `offset` is a **cursor: the entry index**, not a byte
|
||||
position. Each call returns exactly one entry; the client increments the
|
||||
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
|
||||
have been opened with the `directory` flag.
|
||||
- **mount / unmount** — RETIRED from the wire: mounting is the `fs_mount`
|
||||
syscall now (a filesystem server passes its endpoint handle; possession is
|
||||
the capability, exactly the trust of the old cap-passing op). The op
|
||||
numbers stay reserved. Mount-prefix semantics are unchanged: prefixes
|
||||
match at path boundaries only (`/mnt/usb` never captures `/mnt/usbextra`),
|
||||
the longest matching prefix wins, and an optional backend-side rewrite
|
||||
prefix maps a mount into the backend's namespace (fat serves `/mnt/usb`
|
||||
from its volume root and `/var` from its `/var` subtree).
|
||||
- **rename** — same-directory rename only (the router requires old and new to
|
||||
resolve under one mount).
|
||||
|
||||
## Open flags
|
||||
|
||||
Bitwise OR in `Request.flags`, meaningful for `open` only:
|
||||
|
||||
| bit | name | meaning |
|
||||
|----:|------|---------|
|
||||
| 1 | `create` | create the file if it does not exist |
|
||||
| 2 | `directory` | open a directory node for `readdir` rather than a file |
|
||||
| 4 | `truncate` | truncate an existing file to zero length on open (replace, don't overwrite in place) |
|
||||
|
||||
## FileStatus — 24 bytes (the `status` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 8 | `size` | file size in bytes |
|
||||
| 8 | 4 | `kind` | a **NodeKind** value |
|
||||
| 12 | 4 | — | padding |
|
||||
| 16 | 8 | `mtime` | modification time, Unix epoch seconds UTC; 0 if the backend keeps none |
|
||||
|
||||
## DirectoryEntry — 16 bytes + name (the `readdir` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `kind` | a **NodeKind** value |
|
||||
| 4 | 4 | `name_len` | length of the name that follows |
|
||||
| 8 | 8 | `size` | the entry's size in bytes |
|
||||
| 16 | `name_len` | name | the entry's name, not NUL-terminated |
|
||||
|
||||
## NodeKind
|
||||
|
||||
Aligned to the FSH file-type table
|
||||
(docs/danos-file-system-hierarchy-FSH.md):
|
||||
|
||||
| value | kind |
|
||||
|------:|------|
|
||||
| 0 | regular file |
|
||||
| 1 | directory |
|
||||
| 2 | character device |
|
||||
| 3 | block device |
|
||||
| 4 | symbolic link |
|
||||
| 5 | fifo |
|
||||
| 6 | socket |
|
||||
|
||||
Clients should map unknown values to *regular* rather than reject — the
|
||||
table can grow.
|
||||
|
||||
## Lifetimes and trust
|
||||
|
||||
Open-node ids live in the server. A client that dies without closing leaks
|
||||
nothing permanently: the VFS subscribes to the kernel's published process-exit
|
||||
events (docs/process-lifecycle.md) and releases a dead client's handles,
|
||||
closing forwarded backend nodes best-effort. Ids are plain integers, not
|
||||
capabilities — the VFS trusts its callers with each other's ids today, which
|
||||
is acceptable while every client is part of the system image and worth
|
||||
revisiting (per-client id namespaces) before third-party binaries arrive.
|
||||
|
||||
## Evolution rules
|
||||
|
||||
What a non-Zig implementation may rely on, and what it must not:
|
||||
|
||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||
**append-only and frozen once shipped** — the unit tests in `protocol.zig`
|
||||
pin them exactly so a refactor can't silently move them.
|
||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||
- Negative statuses beyond -1 will appear (an errno vocabulary); success is
|
||||
exactly 0.
|
||||
@@ -38,6 +38,13 @@ pub const Device = struct {
|
||||
return self.transfer(.write, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||
pub fn flush(self: Device) bool {
|
||||
return self.transfer(.flush, 0, 0, 0);
|
||||
}
|
||||
|
||||
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
@@ -47,6 +54,13 @@ pub const Device = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// One lookup attempt, no waiting — for a server that retries on its own
|
||||
/// timer (the fat service) instead of blocking its harness in here.
|
||||
pub fn tryOpen() ?Device {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Look up the block device, retrying generously while the USB storage chain
|
||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||
pub fn open() ?Device {
|
||||
@@ -54,7 +68,10 @@ pub fn open() ?Device {
|
||||
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||
// tens of seconds under emulation.
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 1200) : (attempts += 1) {
|
||||
// 30 s covers the slowest observed healthy chain (a flaky QEMU enumeration
|
||||
// completed at ~24 s); a machine whose stick genuinely failed setup should
|
||||
// not sit a further minute pretending otherwise.
|
||||
while (attempts < 600) : (attempts += 1) {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
system.sleep(50);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||
//! shape: a cached `.display` lookup with a boot-race retry, then extern-struct request/
|
||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("display-protocol");
|
||||
|
||||
/// The display's current mode, as `info()` reports it.
|
||||
pub const Info = struct {
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32, // bytes per row (may exceed width*4; see docs/framebuffer.md)
|
||||
format: u32, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||
};
|
||||
|
||||
/// The service endpoint, looked up once and cached.
|
||||
var handle: ?ipc.Handle = null;
|
||||
|
||||
/// Look up the display service, retrying while it comes up (a client races its
|
||||
/// registration at boot). Returns the endpoint, or null if it never appears.
|
||||
fn service() ?ipc.Handle {
|
||||
if (handle) |h| return h;
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.display)) |h| {
|
||||
handle = h;
|
||||
return h;
|
||||
}
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Send one request, receive its reply; true on a zero status. `out` receives the reply
|
||||
/// so callers can read `info`/`layer` fields on success.
|
||||
fn transact(request: protocol.Request, out: *protocol.Reply) bool {
|
||||
const h = service() orelse return false;
|
||||
var req = request;
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&req), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
out.* = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
return out.status == 0;
|
||||
}
|
||||
|
||||
/// The display's current mode, or null if the service never came up.
|
||||
pub fn info() ?Info {
|
||||
var reply: protocol.Reply = undefined;
|
||||
if (!transact(.{ .operation = @intFromEnum(protocol.Operation.info) }, &reply)) return null;
|
||||
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||
}
|
||||
|
||||
/// Composite the dirty layers and flush the frame to the screen.
|
||||
pub fn present() bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.present) }, &reply);
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const h = service() orelse return 0;
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.get_modes) };
|
||||
var reply: [protocol.modes_reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (len < protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(protocol.ModesReply, reply[0..protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
|
||||
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||
var mode: ?Info = null;
|
||||
|
||||
fn cachedInfo() ?Info {
|
||||
if (mode) |m| return m;
|
||||
const i = info() orelse return null;
|
||||
mode = i;
|
||||
return i;
|
||||
}
|
||||
|
||||
/// The native pixel value for an 8-bit-per-channel colour, in the display's format. A
|
||||
/// client packs colours through this so it never has to know the byte order itself.
|
||||
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||
const format = if (cachedInfo()) |i| i.format else 0;
|
||||
return protocol.pack(format, r, g, b);
|
||||
}
|
||||
|
||||
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||
/// into by command. Create with `createLayer`; drawing and moves take effect on the next
|
||||
/// `present`. Coordinates are signed (a layer may sit partly off-screen).
|
||||
pub const Layer = struct {
|
||||
id: u32,
|
||||
|
||||
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.fill_rect),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
.colour = colour,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||
/// layer at (`x`, `y`). The tile rides inline in the request, so `w*h*4` must fit
|
||||
/// `protocol.maximum_payload`.
|
||||
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
var request = protocol.Request{
|
||||
.operation = @intFromEnum(protocol.Operation.blit_tile),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
};
|
||||
const header = std.mem.asBytes(&request);
|
||||
if (header.len + pixels.len > protocol.message_maximum) return false;
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(buffer[0..header.len], header);
|
||||
@memcpy(buffer[header.len..][0..pixels.len], pixels);
|
||||
const h_svc = service() orelse return false;
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h_svc, buffer[0 .. header.len + pixels.len], &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Move / restack / show or hide the layer.
|
||||
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.configure_layer),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.z = z,
|
||||
.visible = if (visible) 1 else 0,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.damage),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Release the layer and its surface.
|
||||
pub fn destroy(self: Layer) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||
}
|
||||
};
|
||||
|
||||
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||
var reply: protocol.Reply = undefined;
|
||||
if (!transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.create_layer),
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
.z = z,
|
||||
.visible = 1,
|
||||
}, &reply)) return null;
|
||||
return .{ .id = reply.layer };
|
||||
}
|
||||
+129
-49
@@ -1,5 +1,5 @@
|
||||
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
|
||||
//! lists files served by the user-space VFS (system/services/vfs), each call
|
||||
//! lists files through the kernel VFS root (resolve + redirect), each call
|
||||
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
|
||||
//! danos programs use directly; it is also where the file operations that later
|
||||
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
|
||||
@@ -12,6 +12,7 @@
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("vfs-protocol");
|
||||
|
||||
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||
@@ -60,23 +61,37 @@ pub const OpenOptions = struct {
|
||||
}
|
||||
};
|
||||
|
||||
// The VFS server endpoint, looked up once by well-known id and cached.
|
||||
var vfs_handle: ipc.Handle = 0;
|
||||
var vfs_resolved = false;
|
||||
fn vfs() ?ipc.Handle {
|
||||
if (!vfs_resolved) {
|
||||
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||
vfs_resolved = true;
|
||||
// The route to a path: the kernel resolves (fs_resolve) and either serves the
|
||||
// node itself (the initrd at /system — a permanent token) or redirects us to
|
||||
// the owning filesystem backend's endpoint, to which we speak the vfs-protocol
|
||||
// rendezvous directly with the rewritten mount-relative path.
|
||||
const Route = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: ipc.Handle, path: [224]u8, path_len: usize },
|
||||
|
||||
fn backendPath(self: *const Route) []const u8 {
|
||||
return self.backend.path[0..self.backend.path_len];
|
||||
}
|
||||
};
|
||||
|
||||
fn resolve(path: []const u8, flags: usize) ?Route {
|
||||
var out: [224]u8 = undefined;
|
||||
const route = system.fsResolve(path, flags, &out) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .kernel = token },
|
||||
.backend => |b| {
|
||||
var r: Route = .{ .backend = .{ .handle = b.handle, .path = undefined, .path_len = b.path_len } };
|
||||
@memcpy(r.backend.path[0..b.path_len], out[0..b.path_len]);
|
||||
return r;
|
||||
},
|
||||
}
|
||||
return vfs_handle;
|
||||
}
|
||||
|
||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||
|
||||
// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||
// One request/reply round trip: [Request header][send payload] -> backend ->
|
||||
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
const h = vfs() orelse return null;
|
||||
fn transact(h: ipc.Handle, request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, protocol.maximum_payload);
|
||||
@@ -95,13 +110,21 @@ fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
pub const File = struct {
|
||||
node: u64,
|
||||
offset: u64 = 0,
|
||||
/// The owning backend's endpoint, or null for a kernel-served node (the
|
||||
/// read-only /system tree), whose `node` is a permanent fs_node token.
|
||||
backend: ?ipc.Handle = null,
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
|
||||
/// null on error.
|
||||
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||
const h = self.backend orelse {
|
||||
const n = system.fsNodeRead(self.node, self.offset, buffer) orelse return null;
|
||||
self.offset += n;
|
||||
return n;
|
||||
};
|
||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, &.{}, buffer) orelse return null;
|
||||
const r = transact(h, request, &.{}, buffer) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
@@ -109,11 +132,13 @@ pub const File = struct {
|
||||
|
||||
/// Write `data` at the current offset; returns the count written. A single
|
||||
/// call is capped at the VFS payload size, so the return may be short — use
|
||||
/// `writeAll` to write the whole slice. Null on error.
|
||||
/// `writeAll` to write the whole slice. Null on error (kernel-served nodes
|
||||
/// are read-only).
|
||||
pub fn write(self: *File, data: []const u8) ?usize {
|
||||
const h = self.backend orelse return null;
|
||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, data[0..want], &.{}) orelse return null;
|
||||
const r = transact(h, request, data[0..want], &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
@@ -138,27 +163,40 @@ pub const File = struct {
|
||||
|
||||
/// This file's metadata.
|
||||
pub fn attributes(self: *File) ?Attributes {
|
||||
const h = self.backend orelse {
|
||||
const a = system.fsNodeStatus(self.node) orelse return null;
|
||||
return .{ .size = a.size, .kind = if (a.kind == system.file_kind_directory) .directory else .regular, .mtime = a.mtime };
|
||||
};
|
||||
const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &buffer) orelse return null;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return null;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]);
|
||||
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||
}
|
||||
|
||||
/// Release the VFS's open handle for this file.
|
||||
/// Release the backend's open handle for this file. Kernel-served node
|
||||
/// tokens are permanent — nothing to release.
|
||||
pub fn close(self: *File) void {
|
||||
const h = self.backend orelse return;
|
||||
const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(request, &.{}, &.{});
|
||||
_ = transact(h, request, &.{}, &.{});
|
||||
}
|
||||
};
|
||||
|
||||
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
|
||||
pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = options.wireFlags() };
|
||||
const r = transact(request, path, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node };
|
||||
const route = resolve(path, options.wireFlags()) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .node = token, .backend = null },
|
||||
.backend => |b| {
|
||||
const relative = route.backendPath();
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() };
|
||||
const r = transact(b.handle, request, relative, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node, .backend = b.handle };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// A path's metadata without keeping it open (open -> status -> close).
|
||||
@@ -190,13 +228,27 @@ pub const Entry = struct {
|
||||
pub const Directory = struct {
|
||||
node: u64,
|
||||
cursor: u64 = 0,
|
||||
backend: ?ipc.Handle = null,
|
||||
|
||||
/// Fill `entry` with the next directory entry; false at end of directory or
|
||||
/// on error.
|
||||
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||
const h = self.backend orelse {
|
||||
var buffer: [@sizeOf(system.DirectoryEntryHeader) + 64]u8 = undefined;
|
||||
const n = system.fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false;
|
||||
if (n < @sizeOf(system.DirectoryEntryHeader)) return false; // end
|
||||
const header = std.mem.bytesToValue(system.DirectoryEntryHeader, buffer[0..@sizeOf(system.DirectoryEntryHeader)]);
|
||||
entry.kind = if (header.kind == system.file_kind_directory) .directory else .regular;
|
||||
entry.size = header.size;
|
||||
const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(system.DirectoryEntryHeader)..][0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
};
|
||||
const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
const r = transact(request, &.{}, &buffer) orelse return false;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return false;
|
||||
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||
if (r.payload.len < protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]);
|
||||
@@ -210,9 +262,9 @@ pub const Directory = struct {
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release the VFS's open handle for this directory.
|
||||
/// Release the backend's open handle for this directory.
|
||||
pub fn close(self: *Directory) void {
|
||||
var f = File{ .node = self.node };
|
||||
var f = File{ .node = self.node, .backend = self.backend };
|
||||
f.close();
|
||||
}
|
||||
};
|
||||
@@ -220,13 +272,18 @@ pub const Directory = struct {
|
||||
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
|
||||
pub fn openDirectory(path: []const u8) ?Directory {
|
||||
const file = open(path, .{ .directory = true }) orelse return null;
|
||||
return .{ .node = file.node };
|
||||
return .{ .node = file.node, .backend = file.backend };
|
||||
}
|
||||
|
||||
// A path-based request that returns only a status (mkdir, unlink).
|
||||
// A path-based request that returns only a status (mkdir, unlink). Kernel-served
|
||||
// paths (the read-only /system) refuse mutation by construction: the resolve
|
||||
// must land on a backend.
|
||||
fn pathOperation(operation: protocol.Operation, path: []const u8) bool {
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = 0 };
|
||||
const r = transact(request, path, &.{}) orelse return false;
|
||||
const route = resolve(path, 0) orelse return false;
|
||||
if (route != .backend) return false;
|
||||
const relative = route.backendPath();
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
const r = transact(route.backend.handle, request, relative, &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
@@ -236,39 +293,62 @@ pub fn makeDirectory(path: []const u8) bool {
|
||||
return pathOperation(.mkdir, path);
|
||||
}
|
||||
|
||||
/// Create every missing directory along `path` (mkdir -p). Probes each prefix
|
||||
/// with `exists` first — a FAT mkdir of an existing name is refused, and the
|
||||
/// probe keeps the common "already there" case cheap. Returns true when the
|
||||
/// whole path exists afterwards.
|
||||
pub fn makePath(path: []const u8) bool {
|
||||
var end: usize = 0;
|
||||
while (end < path.len) {
|
||||
end += 1;
|
||||
while (end < path.len and path[end] != '/') end += 1;
|
||||
const prefix = path[0..end];
|
||||
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
||||
// Best-effort per prefix: components at or above a mount point ("/mnt")
|
||||
// are router names, not filesystem nodes — they neither exist as nodes
|
||||
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
||||
if (!exists(prefix)) _ = makeDirectory(prefix);
|
||||
}
|
||||
return exists(path);
|
||||
}
|
||||
|
||||
/// Remove the file at `path`. Returns true on success. Directories are refused
|
||||
/// (a separate directory-removal would have to check emptiness).
|
||||
pub fn remove(path: []const u8) bool {
|
||||
return pathOperation(.unlink, path);
|
||||
}
|
||||
|
||||
/// Rename `old_path` to `new_path`. Both must be in the same directory (same-
|
||||
/// directory, 8.3-name rename only for now). Returns true on success.
|
||||
/// Rename `old_path` to `new_path`. Both must resolve to the SAME filesystem
|
||||
/// backend (same-directory, 8.3-name rename only for now). Returns true on
|
||||
/// success.
|
||||
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
const total = old_path.len + 1 + new_path.len;
|
||||
const old_route = resolve(old_path, 0) orelse return false;
|
||||
const new_route = resolve(new_path, 0) orelse return false;
|
||||
if (old_route != .backend or new_route != .backend) return false;
|
||||
if (old_route.backend.handle != new_route.backend.handle) return false; // cross-filesystem
|
||||
const old_relative = old_route.backendPath();
|
||||
const new_relative = new_route.backendPath();
|
||||
const total = old_relative.len + 1 + new_relative.len;
|
||||
if (total > protocol.maximum_payload) return false;
|
||||
var payload: [protocol.maximum_payload]u8 = undefined;
|
||||
@memcpy(payload[0..old_path.len], old_path);
|
||||
payload[old_path.len] = 0;
|
||||
@memcpy(payload[old_path.len + 1 ..][0..new_path.len], new_path);
|
||||
@memcpy(payload[0..old_relative.len], old_relative);
|
||||
payload[old_relative.len] = 0;
|
||||
@memcpy(payload[old_relative.len + 1 ..][0..new_relative.len], new_relative);
|
||||
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const r = transact(request, payload[0..total], &.{}) orelse return false;
|
||||
const r = transact(old_route.backend.handle, request, payload[0..total], &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||
/// the VFS then routes everything under `target` to that backend. This is the one
|
||||
/// call that hands the VFS a capability (the backend endpoint). Returns true on
|
||||
/// success.
|
||||
/// the kernel VFS then routes everything under `target` to that backend.
|
||||
/// Possession of the endpoint handle is the capability. Returns true on success.
|
||||
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||
const h = vfs() orelse return false;
|
||||
const request = protocol.Request{ .operation = .mount, .node = 0, .offset = 0, .len = @intCast(target.len), .flags = 0 };
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const tlen = @min(target.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..tlen], target[0..tlen]);
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const result = ipc.callCap(h, message[0 .. protocol.request_size + tlen], &rbuf, backend) catch return false;
|
||||
if (result.len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]).status == 0;
|
||||
return system.fsMount(target, backend, "");
|
||||
}
|
||||
|
||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||
return system.fsMount(target, backend, rewrite);
|
||||
}
|
||||
|
||||
@@ -9,15 +9,32 @@
|
||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||
//! itself, and the kernel picks the base address.
|
||||
//!
|
||||
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
||||
//! lock and larger alignments come when user programs gain threads.
|
||||
//! 16-byte maximum alignment, exactly like the kernel heap. The free list is guarded by
|
||||
//! a `Thread.Mutex` **only in multi-threaded binaries** (`addThreadedUserBinary`): the
|
||||
//! guard is gated on `builtin.single_threaded`, so an ordinary single-threaded binary
|
||||
//! compiles it out and pays nothing, while a threaded one can allocate safely from
|
||||
//! several threads at once (docs/threading-plan.md M7). The lock lives at the two
|
||||
//! free-list mutators — `rawAlloc`/`rawFree` — which every entry point funnels through.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const system_calls = @import("system.zig");
|
||||
const Mutex = @import("thread.zig").Thread.Mutex;
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// Guards `free_list`. A no-op in single-threaded builds (compiled out); a real futex
|
||||
/// mutex in threaded ones. Uncontended acquisition is a single CAS — no syscall.
|
||||
var heap_mutex: Mutex = .{};
|
||||
|
||||
inline fn lockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.lock();
|
||||
}
|
||||
inline fn unlockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.unlock();
|
||||
}
|
||||
|
||||
/// A block header, at the start of every block; while free it also links the
|
||||
/// free list via `next`.
|
||||
const Block = extern struct {
|
||||
@@ -84,8 +101,11 @@ fn insertFree(block: *Block) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory. Holds the heap lock
|
||||
/// across the free-list search and any `grow` (which also touches the free list).
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
@@ -119,6 +139,8 @@ fn rawAlloc(len: usize) ?[*]u8 {
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
//! The per-process logger: std.log wired to the tagged kernel log ring.
|
||||
//!
|
||||
//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped
|
||||
//! logger); this backend formats the line into a fixed buffer and emits ONE
|
||||
//! `debug_write` record carrying the level. The kernel stamps the record with
|
||||
//! the sender's pid and task name (its binary path) — the process does NOT put
|
||||
//! its own name in the payload; attribution is the kernel's, structural and
|
||||
//! unforgeable. Serial shows the kernel-rendered `<path>: message` line, and
|
||||
//! the logger service demultiplexes the ring into one file per process.
|
||||
//!
|
||||
//! Installed for every user binary by the root shim (library/runtime/root.zig)
|
||||
//! via `std_options`; a program can override by declaring its own
|
||||
//! `pub const std_options`.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
fn levelOf(comptime level: std.log.Level) system.KlogLevel {
|
||||
return switch (level) {
|
||||
.err => .err,
|
||||
.warn => .warn,
|
||||
.info => .info,
|
||||
.debug => .debug,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn logFn(
|
||||
comptime level: std.log.Level,
|
||||
comptime scope: @EnumLiteral(),
|
||||
comptime format: []const u8,
|
||||
args: anytype,
|
||||
) void {
|
||||
// One record = one line = at most klog_maximum_message bytes of payload.
|
||||
// On overflow keep what fits and end with "~" so the record is still a
|
||||
// whole line (the kernel would split an embedded rest anyway).
|
||||
var buffer: [256]u8 = undefined;
|
||||
const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") ";
|
||||
const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: {
|
||||
buffer[buffer.len - 1] = '~';
|
||||
break :truncated buffer[0..];
|
||||
};
|
||||
_ = system.writeRecord(levelOf(level), line);
|
||||
}
|
||||
|
||||
/// The std.Options the root shim installs unless the program overrides it.
|
||||
/// Debug level: filtering is the log *reader's* job here — the ring is cheap,
|
||||
/// serial is a dev convenience, and the logger service keeps everything.
|
||||
pub const default_options: std.Options = .{
|
||||
.log_level = .debug,
|
||||
.logFn = logFn,
|
||||
};
|
||||
@@ -0,0 +1,25 @@
|
||||
//! The root module every user binary is compiled through (build.zig,
|
||||
//! `addUserBinary`). The program's own file is imported as `program`, and this
|
||||
//! shim contributes the declarations Zig resolves from the compilation root —
|
||||
//! `main` (dispatched by runtime.start) and the panic handler — and pulls in the
|
||||
//! `_start` entry shim. A program therefore only defines `pub fn main`; nothing
|
||||
//! else is required in its source file.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const program = @import("program");
|
||||
|
||||
/// Resolved as `@import("root").main` by runtime.start's comptime dispatch.
|
||||
pub const main = program.main;
|
||||
|
||||
/// The panic handler for every safety check in the image (runtime.start.panic).
|
||||
pub const panic = runtime.panic;
|
||||
|
||||
/// std.log for every user binary goes to the tagged kernel log ring (the kernel
|
||||
/// stamps the sender; see runtime.log). A program overrides by declaring its
|
||||
/// own `pub const std_options`.
|
||||
pub const std_options: @import("std").Options =
|
||||
if (@hasDecl(program, "std_options")) program.std_options else runtime.log.default_options;
|
||||
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -4,14 +4,14 @@
|
||||
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||
//! freestanding target), so all user programs share one implementation.
|
||||
//!
|
||||
//! A user binary needs three lines:
|
||||
//! const runtime = @import("runtime");
|
||||
//! pub const panic = runtime.panic;
|
||||
//! comptime { _ = &runtime.start._start; } // pull the entry shim in
|
||||
//! and a `pub fn main() void` or `pub fn main(init: runtime.process.Init) void`
|
||||
//! (arguments arrive via `init`).
|
||||
//! A user binary only defines a `pub fn main() void` or
|
||||
//! `pub fn main(init: runtime.process.Init) void` (arguments arrive via `init`).
|
||||
//! The panic handler and the `_start` entry pull live in the shared compilation
|
||||
//! root, library/runtime/root.zig, which build.zig wires around every program —
|
||||
//! nothing to declare per source file.
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
pub const log = @import("log.zig");
|
||||
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||
pub const time = @import("time.zig");
|
||||
@@ -38,6 +38,11 @@ pub const device = @import("device.zig");
|
||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||
pub const dma = @import("dma.zig");
|
||||
|
||||
/// Shared cacheable memory: create a region + capability, pass the capability to another
|
||||
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shared-memory.zig
|
||||
/// and docs/display-v2.md.
|
||||
pub const shared_memory = @import("shared-memory.zig");
|
||||
|
||||
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||
pub const usb = @import("usb.zig");
|
||||
@@ -46,17 +51,30 @@ pub const usb = @import("usb.zig");
|
||||
/// usb-storage). See library/runtime/block.zig.
|
||||
pub const block = @import("block.zig");
|
||||
|
||||
/// Display-service client: query the mode, and (from D3) create layers, draw, and
|
||||
/// present frames. See library/runtime/display.zig and system/services/display/.
|
||||
pub const display = @import("display.zig");
|
||||
/// The display wire protocol (shared with the display service and its clients).
|
||||
pub const display_protocol = @import("display-protocol");
|
||||
/// The scanout wire protocol: the compositor's present channel to a native scanout driver
|
||||
/// (virtio-gpu). See system/services/display/scanout-protocol.zig and docs/display-v2.md.
|
||||
pub const scanout_protocol = @import("scanout-protocol");
|
||||
|
||||
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||
/// layer danos programs use directly, and where the operations that later become
|
||||
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||
pub const fs = @import("fs.zig");
|
||||
|
||||
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||
/// Re-exported so the root shim (root.zig) can install it as the panic handler.
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||
pub const process = @import("process.zig");
|
||||
|
||||
/// Threads: `runtime.Thread`, std.Thread-shaped, over the private thread ABI
|
||||
/// (docs/threading.md). A binary must be built multi-threaded to spawn.
|
||||
pub const Thread = @import("thread.zig").Thread;
|
||||
|
||||
/// The service harness: one replyWait loop folding requests, signals, and
|
||||
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||
pub const service = @import("service.zig");
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
//! User-space shared memory: `shared_memory_create` / `shared_memory_map`. A process creates a shareable,
|
||||
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||
//! `shared_memory_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||
//! generalization of capability passing from endpoints to memory objects.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||
}
|
||||
|
||||
/// A shared region: the `ptr` the CPU touches, and the `handle` (a capability) to hand to
|
||||
/// another process as an `ipc_call` send_cap.
|
||||
pub const Region = struct {
|
||||
ptr: [*]u8,
|
||||
handle: ipc.Handle,
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||
/// Returns the region or null on failure. Two return values — virtual_address in rax, handle in rdx —
|
||||
/// so this is a hand-written stub like `dma.alloc`.
|
||||
pub fn create(len: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: the capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shared_memory_create)),
|
||||
[a0] "{rdi}" (len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .ptr = @ptrFromInt(rax), .handle = rdx, .len = len };
|
||||
}
|
||||
|
||||
/// Map the shared region named by a capability `handle` this process received (via an
|
||||
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||
/// Returns the pointer, or null on failure.
|
||||
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
const r = sc.systemCall1(.shared_memory_map, handle);
|
||||
if (failed(r)) return null;
|
||||
return @ptrFromInt(r);
|
||||
}
|
||||
|
||||
/// The guest-physical base of the shared region named by `handle` (which this process must
|
||||
/// hold a capability for). The region's frames are contiguous, so this single address plus
|
||||
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||
/// `attach_backing`. Returns null on failure.
|
||||
pub fn physical(handle: ipc.Handle) ?usize {
|
||||
const r = sc.systemCall1(.shared_memory_physical, handle);
|
||||
if (failed(r)) return null;
|
||||
return r;
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
||||
//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in.
|
||||
//! `entry = _start` in build.zig); the shared compilation root, root.zig, forces
|
||||
//! this file to be analysed with `comptime { _ = &runtime.start._start; }`, so
|
||||
//! the whole runtime is linked in.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
@@ -36,7 +37,7 @@ export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||
/// zero parameters or one `process.Init`; returns void, noreturn, u8, !void, or !u8.
|
||||
fn callMain(init: process.Init) u8 {
|
||||
const root = @import("root"); // the user binary's root source file
|
||||
const root = @import("root"); // root.zig, re-exporting the program's main
|
||||
const main_information = @typeInfo(@TypeOf(root.main)).@"fn";
|
||||
|
||||
const call_arguments = switch (main_information.params.len) {
|
||||
|
||||
+112
-11
@@ -21,10 +21,34 @@ pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||
/// The tagged-log level of a record — re-exported so runtime.log and the logger
|
||||
/// service don't import `abi` themselves.
|
||||
pub const KlogLevel = abi.KlogLevel;
|
||||
pub const KlogStatus = abi.KlogStatus;
|
||||
pub const KlogRecordHeader = abi.KlogRecordHeader;
|
||||
pub const klog_record_header_size = abi.klog_record_header_size;
|
||||
pub const klog_record_alignment = abi.klog_record_alignment;
|
||||
pub const klog_record_magic = abi.klog_record_magic;
|
||||
pub const klog_flag_truncated = abi.klog_flag_truncated;
|
||||
pub const klog_maximum_message = abi.klog_maximum_message;
|
||||
pub const maximum_process_name = abi.maximum_process_name;
|
||||
pub const FileAttributes = abi.FileAttributes;
|
||||
pub const DirectoryEntryHeader = abi.DirectoryEntryHeader;
|
||||
pub const file_kind_regular = abi.file_kind_regular;
|
||||
pub const file_kind_directory = abi.file_kind_directory;
|
||||
|
||||
/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary
|
||||
/// output goes through std.log -> writeRecord). The kernel stamps the record
|
||||
/// with this process's id and name. Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||
return writeRecord(.raw, message);
|
||||
}
|
||||
|
||||
/// Emit one leveled record into the tagged kernel log ring. The kernel stamps
|
||||
/// pid/name/sequence/timestamp; the payload should be a single line (embedded
|
||||
/// newlines split into further records).
|
||||
pub fn writeRecord(level: KlogLevel, message: []const u8) usize {
|
||||
return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level));
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds.
|
||||
@@ -59,14 +83,91 @@ pub fn wallClock() u64 {
|
||||
return @intCast(sc.systemCall0(.wall_clock));
|
||||
}
|
||||
|
||||
/// Copy bytes out of the kernel's in-memory diagnostic log — the accumulated
|
||||
/// stream of everything `write` (and the kernel itself) has emitted — starting at
|
||||
/// `offset`, into `out`. Returns the number of bytes copied (0 at end of buffer).
|
||||
/// A program reads the whole log by looping from offset 0, advancing by the return
|
||||
/// value, until it gets 0. This is how the boot log is persisted to disk on a
|
||||
/// headless/real machine where serial output is otherwise lost.
|
||||
pub fn klogRead(offset: usize, out: []u8) usize {
|
||||
return sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
/// Copy bytes out of the tagged kernel log ring — framed records of everything
|
||||
/// every process (and the kernel) has emitted — starting at stream offset
|
||||
/// `offset`, into `out`. Returns the byte count (0 = caught up), or null when
|
||||
/// `offset` fell behind the ring's tail (those records were overwritten) or
|
||||
/// lies past its head; re-sync via `klogStatus`. A reader parses
|
||||
/// [KlogRecordHeader][name][message] frames (8-byte aligned) from the bytes.
|
||||
pub fn klogRead(offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// The log ring's live cursors (oldest retained offset, end of stream, next
|
||||
/// sequence number) plus the wall-clock time of boot — how a log reader starts,
|
||||
/// detects loss, and names a per-boot log directory.
|
||||
pub fn klogStatus() ?KlogStatus {
|
||||
var status: KlogStatus = undefined;
|
||||
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null;
|
||||
return status;
|
||||
}
|
||||
|
||||
/// Where fs_resolve routed a path: served by the kernel (a permanent node
|
||||
/// token for fs_node) or by a userspace filesystem backend (an endpoint handle
|
||||
/// plus the rewritten mount-relative path, returned in the caller's buffer).
|
||||
pub const FsRoute = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: usize, path_len: usize },
|
||||
};
|
||||
|
||||
/// Route `path` through the kernel VFS. For a backend route the rewritten
|
||||
/// mount-relative path lands in `out` (behind a kernel-written length prefix,
|
||||
/// already stripped here: out[0..path_len] is the path).
|
||||
pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)),
|
||||
[a0] "{rdi}" (@intFromPtr(path.ptr)),
|
||||
[a1] "{rsi}" (path.len),
|
||||
[a3] "{r10}" (@intFromPtr(out.ptr)),
|
||||
[a4] "{r8}" (out.len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (@as(isize, @bitCast(rax)) < 0) return null;
|
||||
if (rax == abi.fs_route_kernel) return .{ .kernel = rdx };
|
||||
if (rax != abi.fs_route_backend) return null;
|
||||
const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8);
|
||||
if (path_len + 2 > out.len) return null;
|
||||
std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]);
|
||||
return .{ .backend = .{ .handle = rdx, .path_len = path_len } };
|
||||
}
|
||||
|
||||
/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read).
|
||||
pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// A kernel-served node's metadata (fs_node status).
|
||||
pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes {
|
||||
var attributes: abi.FileAttributes = undefined;
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attributes), @sizeOf(abi.FileAttributes));
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return attributes;
|
||||
}
|
||||
|
||||
/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills
|
||||
/// `out` with [DirectoryEntryHeader][name]; returns total bytes (0 = end).
|
||||
pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// Mount a userspace filesystem's endpoint at `prefix`, with an optional
|
||||
/// backend-side `rewrite` prefix ("" = none). Possession of the endpoint
|
||||
/// handle is the capability.
|
||||
pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool {
|
||||
return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0;
|
||||
}
|
||||
|
||||
pub fn fsUnmount(prefix: []const u8) bool {
|
||||
return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0;
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
|
||||
@@ -0,0 +1,484 @@
|
||||
//! `runtime.Thread` — threads for danos, shaped like Zig's `std.Thread` but built on
|
||||
//! danos's private thread ABI (docs/threading.md). Several tasks share one address
|
||||
//! space; `spawn` starts one, the kernel delivers the closure pointer in the new
|
||||
//! thread's rdi, a plain Zig trampoline runs the user function and calls `thread_exit`,
|
||||
//! and `join` blocks on the thread's exit notification. See docs/threading.md for why
|
||||
//! this mirrors `std.Thread`'s API rather than being the literal type.
|
||||
//!
|
||||
//! The closure (the function's captured args) lives at the **top of the thread's own
|
||||
//! stack**, not the heap — each thread's stack is private, so there is no shared-heap
|
||||
//! concurrency in the spawn/join machinery (the runtime heap is not yet thread-safe).
|
||||
//! A binary must be built multi-threaded (`addThreadedUserBinary`) before it may spawn.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const system = @import("system.zig");
|
||||
|
||||
/// True in a real danos binary; false when this module is compiled for host unit tests.
|
||||
/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state
|
||||
/// machines can be exercised on the host against `std.Thread.Futex` (docs/threading-plan.md
|
||||
/// M11), while the danos build uses the futex syscalls.
|
||||
const on_danos = builtin.os.tag == .freestanding;
|
||||
|
||||
/// A thread stack, if the caller does not override it. 64 KiB of mmap'd, zeroed pages.
|
||||
pub const default_stack_size: usize = 64 * 1024;
|
||||
|
||||
/// Bytes reserved at the top of each thread's stack for its per-thread TLS block (the
|
||||
/// self-pointer plus scratch slots reachable via `%fs`). docs/threading-plan.md M10.
|
||||
const tls_block_size: usize = 64;
|
||||
|
||||
pub const Thread = struct {
|
||||
/// The kernel task id of the spawned thread — what `join` waits on.
|
||||
tid: u32,
|
||||
/// The mmap'd stack, reclaimed by `join` (or at process exit after `detach`).
|
||||
stack_base: usize,
|
||||
stack_size: usize,
|
||||
|
||||
pub const Id = u32;
|
||||
|
||||
pub const SpawnConfig = struct {
|
||||
/// Bytes of stack, rounded up to whole pages by the kernel's mmap.
|
||||
stack_size: usize = default_stack_size,
|
||||
};
|
||||
|
||||
pub const SpawnError = error{
|
||||
/// The kernel refused the thread, the stack mmap failed, or no endpoint was free.
|
||||
SystemResources,
|
||||
};
|
||||
|
||||
/// Start `function(args...)` on a new thread sharing this address space. Mirrors
|
||||
/// `std.Thread.spawn`. The thread's return value is discarded (as in `std.Thread`);
|
||||
/// return data through shared state.
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread {
|
||||
const Args = @TypeOf(args);
|
||||
const Closure = struct {
|
||||
tls_base: usize,
|
||||
args: Args,
|
||||
/// Entered directly by the kernel with `self` in rdi (C ABI). Establishes this
|
||||
/// thread's TLS pointer, runs the user function, then ends the thread.
|
||||
fn entry(self_addr: usize) callconv(.c) noreturn {
|
||||
const self: *@This() = @ptrFromInt(self_addr);
|
||||
setThreadPointer(self.tls_base); // per-thread thread pointer before any user code
|
||||
@call(.auto, function, self.args);
|
||||
exitThread();
|
||||
}
|
||||
};
|
||||
|
||||
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(base)) return error.SystemResources;
|
||||
|
||||
// Top of the thread's own stack, downward: the closure, then a small per-thread TLS
|
||||
// block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is
|
||||
// scratch for user TLS), then the stack proper (rsp starts below the TLS block, so
|
||||
// the growing stack never overwrites either).
|
||||
var closure_addr = (base + config.stack_size) - @sizeOf(Closure);
|
||||
closure_addr &= ~@as(usize, @alignOf(Closure) - 1); // align the closure down
|
||||
|
||||
const tls_base = (closure_addr - tls_block_size) & ~@as(usize, 15);
|
||||
const tls: [*]usize = @ptrFromInt(tls_base);
|
||||
tls[0] = tls_base; // self-pointer (fs:0), as the x86_64 TLS ABI expects
|
||||
|
||||
const closure: *Closure = @ptrFromInt(closure_addr);
|
||||
closure.* = .{ .tls_base = tls_base, .args = args };
|
||||
|
||||
var stack_top = tls_base & ~@as(usize, 15); // 16-align below the TLS block
|
||||
stack_top -= 8; // ...then rsp % 16 == 8 at the C entry
|
||||
|
||||
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr);
|
||||
if (threadSpawnFailed(tid)) {
|
||||
_ = system.munmap(base, config.stack_size);
|
||||
return error.SystemResources;
|
||||
}
|
||||
return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size };
|
||||
}
|
||||
|
||||
/// Block until this thread finishes, then reclaim its stack. Mirrors
|
||||
/// `std.Thread.join`. The exit endpoint is private to this thread, so the first
|
||||
/// child-exit notification on it is this thread's.
|
||||
pub fn join(self: Thread) void {
|
||||
_ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited
|
||||
_ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||
}
|
||||
|
||||
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
|
||||
/// reclaimed at process exit (docs/threading-plan.md M3 — kernel-reaper stack reclaim
|
||||
/// for detached threads is a later refinement). Mirrors `std.Thread.detach`.
|
||||
pub fn detach(self: Thread) void {
|
||||
_ = self;
|
||||
}
|
||||
|
||||
/// The calling thread's id (its kernel task id). Mirrors `std.Thread.getCurrentId`.
|
||||
pub fn getCurrentId() Id {
|
||||
return @intCast(sc.systemCall0(.thread_self));
|
||||
}
|
||||
|
||||
/// The dense 0-based index of the core the calling thread is running on. A danos
|
||||
/// extension beyond `std.Thread`, used to observe genuine cross-core parallelism.
|
||||
pub fn currentCore() Id {
|
||||
return @intCast(sc.systemCall0(.current_core));
|
||||
}
|
||||
|
||||
/// `std.Thread.Futex`-shaped block/wake on a `u32` atomic — the primitive the
|
||||
/// blocking `Mutex`/`Condition`/`Semaphore` are built on. Waiters park in the
|
||||
/// kernel (no busy-wait), so an idle core still halts (docs/halting.md).
|
||||
pub const Futex = struct {
|
||||
/// Block while `ptr.* == expect`. Returns when woken by `wake`, or promptly if
|
||||
/// the value already differs (safe against spurious returns, as in std): the
|
||||
/// caller re-checks its condition in a loop.
|
||||
pub fn wait(ptr: *const std.atomic.Value(u32), expect: u32) void {
|
||||
if (comptime on_danos) {
|
||||
_ = futexWait(@intFromPtr(ptr), expect, 0);
|
||||
} else {
|
||||
// Host unit-test mock: spin+yield until the value changes (`wake` is a
|
||||
// no-op — the callers re-check their condition in a loop anyway). Correct,
|
||||
// if busy; fine for the state-machine tests.
|
||||
while (ptr.load(.acquire) == expect) std.Thread.yield() catch {};
|
||||
}
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first.
|
||||
pub fn timedWait(ptr: *const std.atomic.Value(u32), expect: u32, timeout_ns: u64) error{Timeout}!void {
|
||||
if (comptime on_danos) {
|
||||
if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout;
|
||||
} else {
|
||||
var spins: u64 = 0;
|
||||
const limit = timeout_ns / 1000 + 1;
|
||||
while (ptr.load(.acquire) == expect) : (spins += 1) {
|
||||
if (spins >= limit) return error.Timeout;
|
||||
std.Thread.yield() catch {};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Wake up to `max_waiters` threads blocked on `ptr`.
|
||||
pub fn wake(ptr: *const std.atomic.Value(u32), max_waiters: u32) void {
|
||||
if (comptime on_danos) {
|
||||
_ = futexWake(@intFromPtr(ptr), max_waiters);
|
||||
} else {
|
||||
// host mock: spin-waiters re-check their condition, so no wake is needed.
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// A mutual-exclusion lock, `std.Thread.Mutex`-shaped. The classic three-state
|
||||
/// futex mutex (unlocked / locked / contended): the fast path is a single CAS, and
|
||||
/// only a contended lock ever enters the kernel.
|
||||
pub const Mutex = struct {
|
||||
state: std.atomic.Value(u32) = std.atomic.Value(u32).init(unlocked),
|
||||
|
||||
const unlocked: u32 = 0;
|
||||
const locked: u32 = 1;
|
||||
const contended: u32 = 2;
|
||||
|
||||
/// Try to take the lock without blocking; returns whether it was acquired.
|
||||
pub fn tryLock(m: *Mutex) bool {
|
||||
return m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) == null;
|
||||
}
|
||||
|
||||
/// Acquire the lock, blocking in the kernel while it is contended.
|
||||
pub fn lock(m: *Mutex) void {
|
||||
if (m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) != null) m.lockSlow();
|
||||
}
|
||||
|
||||
fn lockSlow(m: *Mutex) void {
|
||||
@branchHint(.cold);
|
||||
// Mark the lock contended and take it as soon as it falls unlocked; park on
|
||||
// the futex while it stays contended. Marking contended may cause a spurious
|
||||
// wake on unlock (harmless), never a missed one.
|
||||
while (m.state.swap(contended, .acquire) != unlocked) {
|
||||
Futex.wait(&m.state, contended);
|
||||
}
|
||||
}
|
||||
|
||||
/// Release the lock; wake one waiter if the lock was contended.
|
||||
pub fn unlock(m: *Mutex) void {
|
||||
if (m.state.swap(unlocked, .release) == contended) Futex.wake(&m.state, 1);
|
||||
}
|
||||
};
|
||||
|
||||
/// A condition variable, `std.Thread.Condition`-shaped. Spurious wakeups are
|
||||
/// allowed — always wait in a predicate loop with the mutex held. Built on a futex
|
||||
/// sequence counter: a waiter samples the seq, drops the mutex, and parks until the
|
||||
/// seq changes (a signal that races the unlock bumps the seq, so it is not missed).
|
||||
pub const Condition = struct {
|
||||
seq: std.atomic.Value(u32) = std.atomic.Value(u32).init(0),
|
||||
|
||||
/// Atomically release `mutex` and block until signalled, then re-acquire it.
|
||||
pub fn wait(c: *Condition, mutex: *Mutex) void {
|
||||
const seq = c.seq.load(.acquire);
|
||||
mutex.unlock();
|
||||
Futex.wait(&c.seq, seq);
|
||||
mutex.lock();
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first. The
|
||||
/// mutex is re-acquired either way.
|
||||
pub fn timedWait(c: *Condition, mutex: *Mutex, timeout_ns: u64) error{Timeout}!void {
|
||||
const seq = c.seq.load(.acquire);
|
||||
mutex.unlock();
|
||||
const timed_out = if (Futex.timedWait(&c.seq, seq, timeout_ns)) |_| false else |_| true;
|
||||
mutex.lock();
|
||||
if (timed_out) return error.Timeout;
|
||||
}
|
||||
|
||||
/// Wake one waiter.
|
||||
pub fn signal(c: *Condition) void {
|
||||
_ = c.seq.fetchAdd(1, .release);
|
||||
Futex.wake(&c.seq, 1);
|
||||
}
|
||||
|
||||
/// Wake all waiters.
|
||||
pub fn broadcast(c: *Condition) void {
|
||||
_ = c.seq.fetchAdd(1, .release);
|
||||
Futex.wake(&c.seq, std.math.maxInt(u32));
|
||||
}
|
||||
};
|
||||
|
||||
/// A counting semaphore, `std.Thread.Semaphore`-shaped: a permit count guarded by a
|
||||
/// `Mutex` + `Condition`.
|
||||
pub const Semaphore = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
permits: usize = 0,
|
||||
|
||||
/// Take a permit, blocking until one is available.
|
||||
pub fn wait(s: *Semaphore) void {
|
||||
s.mutex.lock();
|
||||
defer s.mutex.unlock();
|
||||
while (s.permits == 0) s.cond.wait(&s.mutex);
|
||||
s.permits -= 1;
|
||||
}
|
||||
|
||||
/// Return a permit and wake a waiter.
|
||||
pub fn post(s: *Semaphore) void {
|
||||
s.mutex.lock();
|
||||
defer s.mutex.unlock();
|
||||
s.permits += 1;
|
||||
s.cond.signal();
|
||||
}
|
||||
};
|
||||
|
||||
/// A reader/writer lock, `std.Thread.RwLock`-shaped: many concurrent readers OR one
|
||||
/// exclusive writer. Reader-preferring (a steady stream of readers can delay a writer),
|
||||
/// built on `Mutex` + `Condition` over a signed state: `>0` = that many readers hold
|
||||
/// it, `-1` = a writer holds it, `0` = free.
|
||||
pub const RwLock = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
state: i64 = 0,
|
||||
|
||||
/// Acquire shared (read) access, blocking while a writer holds the lock.
|
||||
pub fn lockShared(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
while (rw.state < 0) rw.cond.wait(&rw.mutex);
|
||||
rw.state += 1;
|
||||
}
|
||||
|
||||
/// Try to acquire shared access without blocking.
|
||||
pub fn tryLockShared(rw: *RwLock) bool {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
if (rw.state < 0) return false;
|
||||
rw.state += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release shared access; wake a waiting writer once the last reader leaves.
|
||||
pub fn unlockShared(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
rw.state -= 1;
|
||||
if (rw.state == 0) rw.cond.broadcast();
|
||||
}
|
||||
|
||||
/// Acquire exclusive (write) access, blocking until no readers or writer remain.
|
||||
pub fn lock(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
while (rw.state != 0) rw.cond.wait(&rw.mutex);
|
||||
rw.state = -1;
|
||||
}
|
||||
|
||||
/// Try to acquire exclusive access without blocking.
|
||||
pub fn tryLock(rw: *RwLock) bool {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
if (rw.state != 0) return false;
|
||||
rw.state = -1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release exclusive access; wake all waiters (they re-check their condition).
|
||||
pub fn unlock(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
rw.state = 0;
|
||||
rw.cond.broadcast();
|
||||
}
|
||||
};
|
||||
|
||||
/// A `std.Thread.WaitGroup`-shaped counter: `start` before spawning work, `finish` as
|
||||
/// each unit completes, `wait` blocks until the count returns to zero.
|
||||
pub const WaitGroup = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
counter: usize = 0,
|
||||
|
||||
/// Register one pending unit of work.
|
||||
pub fn start(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
wg.counter += 1;
|
||||
}
|
||||
|
||||
/// Mark one unit done; wake waiters if that was the last.
|
||||
pub fn finish(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
wg.counter -= 1;
|
||||
if (wg.counter == 0) wg.cond.broadcast();
|
||||
}
|
||||
|
||||
/// Block until every started unit has finished.
|
||||
pub fn wait(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
while (wg.counter != 0) wg.cond.wait(&wg.mutex);
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
/// thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid, or a wrapped error.
|
||||
fn threadSpawn(entry: usize, stack_top: usize, arg: usize) usize {
|
||||
const exit_endpoint: usize = @intCast(abi.no_cap); // join uses thread_join, not an endpoint
|
||||
return sc.systemCall4(.thread_spawn, entry, stack_top, arg, exit_endpoint);
|
||||
}
|
||||
|
||||
/// The kernel returns a real (small) task id on success and a wrapped `-1` on failure;
|
||||
/// no valid task id ever exceeds a u32.
|
||||
inline fn threadSpawnFailed(ret: usize) bool {
|
||||
return ret > std.math.maxInt(u32);
|
||||
}
|
||||
|
||||
/// End the calling thread. Never returns.
|
||||
fn exitThread() noreturn {
|
||||
_ = sc.systemCall0(.thread_exit);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Set the calling thread's FS base (its user TLS thread pointer).
|
||||
fn setThreadPointer(addr: usize) void {
|
||||
_ = sc.systemCall1(.set_thread_pointer, addr);
|
||||
}
|
||||
|
||||
/// futex_wait(addr, expect, timeout_ns) -> status (abi.futex_*).
|
||||
fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize {
|
||||
return sc.systemCall3(.futex_wait, addr, expect, timeout_ns);
|
||||
}
|
||||
|
||||
/// futex_wake(addr, count) -> number woken.
|
||||
fn futexWake(addr: usize, count: u32) usize {
|
||||
return sc.systemCall2(.futex_wake, addr, count);
|
||||
}
|
||||
|
||||
// --- host unit tests (docs/threading-plan.md M11) ---------------------------
|
||||
//
|
||||
// These run under `zig build test` on the host: the `Futex` seam above uses
|
||||
// `std.Thread.Futex` off-danos, so the lock/condvar state machines can be exercised by
|
||||
// real host threads. They are never compiled into a danos binary (test blocks only build
|
||||
// under test), so their `std.Thread` use is fine even though `std.Thread` is unavailable
|
||||
// on the freestanding target.
|
||||
|
||||
test "Mutex serialises concurrent increments across host threads" {
|
||||
var m: Thread.Mutex = .{};
|
||||
var counter: u64 = 0;
|
||||
const workers = 8;
|
||||
const per = 20_000;
|
||||
const Ctx = struct {
|
||||
m: *Thread.Mutex,
|
||||
c: *u64,
|
||||
fn run(ctx: @This()) void {
|
||||
var i: usize = 0;
|
||||
while (i < per) : (i += 1) {
|
||||
ctx.m.lock();
|
||||
ctx.c.* += 1;
|
||||
ctx.m.unlock();
|
||||
}
|
||||
}
|
||||
};
|
||||
var handles: [workers]std.Thread = undefined;
|
||||
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .m = &m, .c = &counter }});
|
||||
for (handles) |h| h.join();
|
||||
try std.testing.expectEqual(@as(u64, workers * per), counter);
|
||||
}
|
||||
|
||||
test "RwLock never lets a reader observe a half-written pair" {
|
||||
var rw: Thread.RwLock = .{};
|
||||
var a: u64 = 0;
|
||||
var b: u64 = 0; // invariant while a lock is held: a == b
|
||||
var stop = std.atomic.Value(bool).init(false);
|
||||
var ok = std.atomic.Value(bool).init(true);
|
||||
|
||||
const Writer = struct {
|
||||
rw: *Thread.RwLock,
|
||||
a: *u64,
|
||||
b: *u64,
|
||||
stop: *std.atomic.Value(bool),
|
||||
fn run(w: @This()) void {
|
||||
var v: u64 = 1;
|
||||
while (!w.stop.load(.acquire)) : (v +%= 1) {
|
||||
w.rw.lock();
|
||||
w.a.* = v; // update both halves under the exclusive lock...
|
||||
w.b.* = v;
|
||||
w.rw.unlock();
|
||||
}
|
||||
}
|
||||
};
|
||||
const Reader = struct {
|
||||
rw: *Thread.RwLock,
|
||||
a: *u64,
|
||||
b: *u64,
|
||||
ok: *std.atomic.Value(bool),
|
||||
fn run(r: @This()) void {
|
||||
var i: usize = 0;
|
||||
while (i < 200_000) : (i += 1) {
|
||||
r.rw.lockShared();
|
||||
if (r.a.* != r.b.*) r.ok.store(false, .release); // ...so a reader must never see them differ
|
||||
r.rw.unlockShared();
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
var writers: [2]std.Thread = undefined;
|
||||
for (&writers) |*w| w.* = try std.Thread.spawn(.{}, Writer.run, .{Writer{ .rw = &rw, .a = &a, .b = &b, .stop = &stop }});
|
||||
var readers: [4]std.Thread = undefined;
|
||||
for (&readers) |*rd| rd.* = try std.Thread.spawn(.{}, Reader.run, .{Reader{ .rw = &rw, .a = &a, .b = &b, .ok = &ok }});
|
||||
for (readers) |rd| rd.join();
|
||||
stop.store(true, .release);
|
||||
for (writers) |w| w.join();
|
||||
try std.testing.expect(ok.load(.acquire));
|
||||
}
|
||||
|
||||
test "WaitGroup blocks until every started unit finishes" {
|
||||
var wg: Thread.WaitGroup = .{};
|
||||
var done = std.atomic.Value(u32).init(0);
|
||||
const n = 6;
|
||||
const Ctx = struct {
|
||||
wg: *Thread.WaitGroup,
|
||||
done: *std.atomic.Value(u32),
|
||||
fn run(c: @This()) void {
|
||||
_ = c.done.fetchAdd(1, .monotonic);
|
||||
c.wg.finish();
|
||||
}
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < n) : (i += 1) wg.start();
|
||||
var handles: [n]std.Thread = undefined;
|
||||
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .wg = &wg, .done = &done }});
|
||||
wg.wait(); // must not return until all n finished
|
||||
try std.testing.expectEqual(@as(u32, n), done.load(.acquire));
|
||||
for (handles) |h| h.join();
|
||||
}
|
||||
+116
-6
@@ -39,13 +39,13 @@ pub const SystemCall = enum(u64) {
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> virtual_address: map a claimed device's MMIO into this address space
|
||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
||||
dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory
|
||||
dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc
|
||||
dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory
|
||||
dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc
|
||||
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||
io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource
|
||||
io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource
|
||||
@@ -58,11 +58,32 @@ pub const SystemCall = enum(u64) {
|
||||
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
||||
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy tagged log-ring stream bytes from `offset` out to a user buffer; fails once `offset` falls behind the ring's tail (re-sync via klog_status)
|
||||
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
||||
shared_memory_create = 34, // shared_memory_create(len) -> virtual_address (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md)
|
||||
shared_memory_map = 35, // shared_memory_map(cap) -> virtual_address: map the shared region named by a received capability into this address space (the same physical pages the creator sees)
|
||||
shared_memory_physical = 36, // shared_memory_physical(cap) -> physical_address: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md)
|
||||
thread_spawn = 37, // thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid: start a task sharing the caller's address space at `entry` on `stack_top`, `arg` in rdi; exit_endpoint (a handle, or no_cap) is notified when it ends — how join waits (docs/threading.md)
|
||||
thread_exit = 38, // thread_exit(): end the calling thread, dropping one reference to its address space (destroyed on the last)
|
||||
current_core = 39, // current_core() -> index: the dense 0-based index of the core the caller is running on (for parallelism/affinity introspection)
|
||||
futex_wait = 40, // futex_wait(addr, expected, timeout_ns) -> status: if *addr == expected, block until woken or the timeout; returns futex_woken/mismatch/timed_out (docs/threading.md)
|
||||
futex_wake = 41, // futex_wake(addr, count) -> woken: wake up to `count` tasks blocked in futex_wait on `addr` in this address space
|
||||
thread_self = 42, // thread_self() -> tid: the calling thread's kernel task id (runtime.Thread.getCurrentId)
|
||||
thread_join = 43, // thread_join(tid) -> 0: block until the thread with id `tid` has exited (runtime.Thread.join; no per-thread IPC endpoint) (docs/threading.md)
|
||||
set_thread_pointer = 44, // set_thread_pointer(addr) -> 0: set the caller's thread pointer (user-space TLS base; x86_64 IA32_FS_BASE, aarch64 TPIDR_EL0); restored per task across context switches (docs/threading-plan.md M10)
|
||||
klog_status = 45, // klog_status(ptr) -> 0: copy a KlogStatus (ring cursors + the boot wall-clock anchor) out to a user buffer
|
||||
fs_resolve = 46, // fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap) -> route tag (rax: fs_route_*) + node token or backend handle (rdx); a backend resolve writes the rewritten mount-relative path into out (length in r8 via third result)
|
||||
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
|
||||
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
|
||||
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
|
||||
_,
|
||||
};
|
||||
|
||||
/// `futex_wait` return codes (in rax).
|
||||
pub const futex_woken: u64 = 0; // woken by a futex_wake
|
||||
pub const futex_mismatch: u64 = 1; // *addr != expected on entry; the caller did not block
|
||||
pub const futex_timed_out: u64 = 2; // the timeout elapsed before a wake
|
||||
|
||||
/// How a process ended — recorded by the kernel at death, queried by the
|
||||
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||
@@ -71,7 +92,7 @@ pub const SystemCall = enum(u64) {
|
||||
/// never delivered to the faulting process (recovery is restart, not a handler).
|
||||
pub const ExitReason = enum(u8) {
|
||||
exited = 0, // returned from main / called exit
|
||||
aborted = 1, // deliberate self-termination (reserved: no abort path yet)
|
||||
aborted = 1, // deliberate FAILURE exit (exit with a nonzero code): supervisors restart these, unlike a clean .exited
|
||||
segmentation_fault = 2, // page fault
|
||||
illegal_instruction = 3, // invalid opcode
|
||||
arithmetic_fault = 4, // divide error, x87 or SIMD fault
|
||||
@@ -171,11 +192,97 @@ pub const ProcessDescriptor = extern struct {
|
||||
name: [maximum_process_name]u8, // argv[0] at spawn; empty for kernel tasks
|
||||
};
|
||||
|
||||
// --- the tagged kernel log ring (klog) ---------------------------------------
|
||||
// Every `debug_write` becomes one RECORD per payload line, stamped by the kernel
|
||||
// with the sender's pid, task name (its binary path), level, a per-boot sequence
|
||||
// number, and a monotonic timestamp. `klog_read` copies raw stream bytes — a
|
||||
// reader parses [KlogRecordHeader][name][message] frames, each padded to
|
||||
// `klog_record_alignment`. Sequence gaps tell a reader exactly how many records
|
||||
// the ring overwrote while it wasn't looking.
|
||||
|
||||
/// Log level of a klog record — std.log's levels plus `raw` (untagged bytes:
|
||||
/// kernel prints and legacy runtime.system.write output).
|
||||
pub const KlogLevel = enum(u8) { err = 0, warn = 1, info = 2, debug = 3, raw = 4 };
|
||||
|
||||
/// "RK" — leads every record; a parser's resync/corruption guard.
|
||||
pub const klog_record_magic: u16 = 0x4B52;
|
||||
|
||||
/// KlogRecordHeader.flags bit: the emitter truncated the payload to fit.
|
||||
pub const klog_flag_truncated: u8 = 1;
|
||||
|
||||
/// Header of one ring record, followed by `name_len` bytes of task name and
|
||||
/// `message_len` bytes of payload; the whole record is padded to 8 bytes.
|
||||
pub const KlogRecordHeader = extern struct {
|
||||
magic: u16, // klog_record_magic
|
||||
level: KlogLevel,
|
||||
name_len: u8, // 0..maximum_process_name
|
||||
pid: u32, // sender process id; 0 = the kernel itself
|
||||
sequence: u64, // per-boot monotonic record number (gaps = lost records)
|
||||
timestamp_ns: u64, // monotonic ns since boot (the `clock` timebase)
|
||||
message_len: u16, // payload bytes (excludes the record's trailing pad)
|
||||
flags: u8, // klog_flag_* bits
|
||||
_reserved: [5]u8,
|
||||
};
|
||||
|
||||
pub const klog_record_header_size: usize = 32; // @sizeOf(KlogRecordHeader), pinned by a test
|
||||
pub const klog_record_alignment: usize = 8;
|
||||
/// Per-record payload cap (one line; longer emitter lines are truncated).
|
||||
pub const klog_maximum_message: usize = 256;
|
||||
|
||||
// --- the kernel VFS root (resolve + redirect) --------------------------------
|
||||
// fs_resolve routes a path through the kernel mount table. Kernel-backed mounts
|
||||
// (the initrd at /system) resolve to a permanent node TOKEN served by fs_node;
|
||||
// userspace mounts resolve to the backend's endpoint handle (installed in the
|
||||
// caller's table, deduplicated) plus the rewritten mount-relative path — the
|
||||
// caller then speaks the vfs-protocol to the backend directly. The kernel never
|
||||
// blocks on a userspace filesystem.
|
||||
|
||||
/// fs_resolve result tags (rax).
|
||||
pub const fs_route_kernel: u64 = 0; // rdx = node token; serve via fs_node
|
||||
pub const fs_route_backend: u64 = 1; // rdx = endpoint handle; speak vfs-protocol
|
||||
|
||||
/// fs_node operations — the same numbers as the vfs-protocol Operation enum, so
|
||||
/// client code shares one vocabulary.
|
||||
pub const fs_node_read: u64 = 2;
|
||||
pub const fs_node_status: u64 = 4;
|
||||
pub const fs_node_readdir: u64 = 5;
|
||||
|
||||
/// fs_resolve flags (same values as the vfs-protocol open flags).
|
||||
pub const fs_flag_create: u64 = 1;
|
||||
|
||||
/// FileStatus-shaped node metadata (matches the vfs-protocol payload layout).
|
||||
pub const file_kind_regular: u32 = 0;
|
||||
pub const file_kind_directory: u32 = 1;
|
||||
pub const FileAttributes = extern struct {
|
||||
size: u64,
|
||||
kind: u32,
|
||||
_pad: u32 = 0,
|
||||
mtime: u64 = 0,
|
||||
};
|
||||
|
||||
/// One fs_node readdir result: the header, followed by `name_len` name bytes in
|
||||
/// the caller's buffer (matches the vfs-protocol DirectoryEntry layout).
|
||||
pub const DirectoryEntryHeader = extern struct {
|
||||
kind: u32,
|
||||
name_len: u32,
|
||||
size: u64,
|
||||
};
|
||||
|
||||
/// The klog_status copy-out: the ring's live cursors plus the wall-clock time
|
||||
/// of boot — the anchor a log persister names its per-boot directory with and
|
||||
/// combines with record timestamps for wall-clock line stamps.
|
||||
pub const KlogStatus = extern struct {
|
||||
tail: u64, // oldest retained stream offset — always a record boundary
|
||||
head: u64, // next byte to be written (end of stream)
|
||||
next_sequence: u64, // the sequence the next record will get
|
||||
boot_unix_seconds: u64, // wall-clock time of boot (RTC anchor)
|
||||
};
|
||||
|
||||
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
|
||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||
pub const ServiceId = enum(u32) {
|
||||
vfs = 1,
|
||||
vfs = 1, // RETIRED: the router moved into the kernel (fs_resolve); the slot stays reserved
|
||||
input = 2,
|
||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||
@@ -183,6 +290,9 @@ pub const ServiceId = enum(u32) {
|
||||
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
||||
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
||||
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
||||
_,
|
||||
};
|
||||
|
||||
|
||||
+10
-9
@@ -37,6 +37,11 @@ pub const Framebuffer = extern struct {
|
||||
height: u32, // visible rows (e.g. 1080)
|
||||
pitch: u32, // bytes from the start of one row to the start of the next
|
||||
format: PixelFormat,
|
||||
/// The panel's refresh rate in Hz, computed from its EDID preferred timing (pixel
|
||||
/// clock / total pixels per frame) while GOP was still alive — the one moment it is
|
||||
/// readable (docs/gop.md). 0 = unknown (no EDID). The display service paces its
|
||||
/// frame clock by it; without vblank this fixes the *rate*, never the *phase*.
|
||||
refresh_hz: u32 = 0,
|
||||
|
||||
/// Whether a usable framebuffer was handed over.
|
||||
pub fn present(self: Framebuffer) bool {
|
||||
@@ -142,15 +147,11 @@ pub const BootInformation = extern struct {
|
||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||
acpi_rsdp: u64 = 0,
|
||||
/// The raw `/system/services/init` ELF image, read off the boot volume by the loader
|
||||
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||
/// kernel boots without user space. Grows into a full initial_ramdisk handoff later.
|
||||
init_base: u64 = 0,
|
||||
init_len: u64 = 0,
|
||||
/// The initial_ramdisk image (a bundle of extra user binaries — the VFS server and
|
||||
/// device drivers), read off the boot volume into memory that survives the
|
||||
/// handoff, same as `init` above. 0/0 = no initial_ramdisk. See system/initial-ramdisk.zig.
|
||||
/// The initial_ramdisk image: every user binary from the boot volume's /system
|
||||
/// tree (init included), packed by the loader into memory that survives the
|
||||
/// handoff (classified reserved, so the kernel identity-maps it and never
|
||||
/// allocates over it). Entries are named by full FHS path. 0/0 = no binaries
|
||||
/// found — the kernel boots without user space. See system/initial-ramdisk.zig.
|
||||
initial_ramdisk_base: u64 = 0,
|
||||
initial_ramdisk_len: u64 = 0,
|
||||
};
|
||||
|
||||
+28
-62
@@ -3,11 +3,13 @@
|
||||
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
||||
//! bootloader handed us) and translates the static tables into the generic
|
||||
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
||||
//! source. This is deliberately the *static-table* path: MADT (CPUs / interrupt
|
||||
//! controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET (timer), and FADT
|
||||
//! (power register map). The DSDT/SSDT bytecode is handed to the `aml` submodule
|
||||
//! only to extract the sleep-state (`_Sx`) values for power management; full AML namespace
|
||||
//! interpretation is a separate, larger subproject.
|
||||
//! source. This is deliberately the *static-table* path, and **only** that: MADT
|
||||
//! (CPUs / interrupt controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET
|
||||
//! (timer), and FADT (power register map). The DSDT/SSDT bytecode is *not*
|
||||
//! interpreted here — the kernel collects the blobs and publishes them on the
|
||||
//! acpi-tables node for the ring-3 acpi service to parse (device enumeration and
|
||||
//! soft-off). Keeping the ~0.5 MB AML interpretation out of kernel init keeps it
|
||||
//! off the single-core critical path (nothing else runs alongside it there).
|
||||
//!
|
||||
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||
@@ -19,7 +21,6 @@ const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const parameters = @import("parameters");
|
||||
const device_model = @import("device-model.zig");
|
||||
const aml = @import("aml/aml.zig");
|
||||
const DeviceTree = device_model.DeviceTree;
|
||||
const Hal = device_model.Hal;
|
||||
|
||||
@@ -37,8 +38,11 @@ pub const RegisterAccess = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
||||
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
||||
/// The power register map, extracted from the FADT during discovery. Populated by
|
||||
/// `discover`, read by `power` (kernel reboot). The **sleep-state (`_Sx`) values
|
||||
/// live in AML**, which the kernel no longer parses — soft-off (S5) is owned by the
|
||||
/// ring-3 acpi service (it re-parses the blobs on the published acpi-tables node and
|
||||
/// writes the PM1 control register itself). So this holds only the FADT scalars.
|
||||
pub const PowerInformation = struct {
|
||||
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||
@@ -54,10 +58,6 @@ pub const PowerInformation = struct {
|
||||
reset: RegisterAccess = .{},
|
||||
reset_value: u8 = 0,
|
||||
reset_supported: bool = false,
|
||||
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`)
|
||||
/// packages.
|
||||
s5: ?aml.SleepType = null,
|
||||
s3: ?aml.SleepType = null,
|
||||
};
|
||||
|
||||
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||
@@ -141,19 +141,6 @@ const maximum_cpus = parameters.maximum_cpus;
|
||||
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||
pub var cpu_information: CpuInformation = .{};
|
||||
|
||||
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
||||
/// walked every byte of the DSDT/SSDTs without desyncing.
|
||||
pub const AmlStats = struct {
|
||||
nodes: usize = 0,
|
||||
consumed: usize = 0,
|
||||
total: usize = 0,
|
||||
};
|
||||
pub var aml_stats: AmlStats = .{};
|
||||
|
||||
/// The ACPI namespace built from the DSDT/SSDTs, kept for sleep-state (`_Sx`) lookup now and
|
||||
/// device enumeration later. Null until `discover` runs successfully.
|
||||
pub var namespace: ?aml.Namespace = null;
|
||||
|
||||
/// Physical address of the DSDT the FADT points at, or 0.
|
||||
pub var dsdt_physical: u64 = 0;
|
||||
|
||||
@@ -165,8 +152,9 @@ var fadt_physical: u64 = 0;
|
||||
var fadt_length: u64 = 0;
|
||||
|
||||
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||
// address + length of each table's post-header bytecode. Scanned after the walk
|
||||
// for the sleep-state (`_Sx`) packages.
|
||||
// address + length of each table's post-header bytecode. The kernel does not
|
||||
// interpret them — it publishes them on the acpi-tables node for the ring-3 acpi
|
||||
// service to parse (device enumeration + soft-off). See publishAcpiTablesNode.
|
||||
var aml_block_physical: [32]u64 = undefined;
|
||||
var aml_block_len: [32]usize = undefined;
|
||||
var aml_block_count: usize = 0;
|
||||
@@ -373,7 +361,7 @@ const Hpet = extern struct {
|
||||
|
||||
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
||||
/// FADT into `power_information`, and publishes the AML blobs for the ring-3 acpi service.
|
||||
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
if (rsdp_physical == 0) return error.NoRsdp;
|
||||
boot_memory_regions = memory_regions;
|
||||
@@ -383,8 +371,6 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
||||
fadt_physical = 0;
|
||||
fadt_length = 0;
|
||||
platform_information = .{};
|
||||
aml_stats = .{};
|
||||
namespace = null;
|
||||
dsdt_physical = 0;
|
||||
aml_block_count = 0;
|
||||
|
||||
@@ -401,34 +387,20 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
|
||||
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||
}
|
||||
|
||||
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
||||
// read the sleep types from it.
|
||||
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
||||
for (0..aml_block_count) |i| {
|
||||
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
||||
}
|
||||
const active = blocks[0..aml_block_count];
|
||||
if (aml.parse(device_tree.allocator, active)) |pr| {
|
||||
namespace = pr.namespace;
|
||||
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
||||
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
||||
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
||||
// The namespace's Device objects are no longer folded into the kernel
|
||||
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
||||
// (published below), re-parses the same blobs, and registers + reports
|
||||
// the _HID devices itself. The kernel keeps the namespace only for the
|
||||
// \_S5 sleep type above.
|
||||
} else |_| {
|
||||
// AML parse failed (e.g. out of memory); power stays best-effort with
|
||||
// whatever the FADT alone provided.
|
||||
}
|
||||
// The kernel does **not** interpret the DSDT/SSDTs. Static-table discovery
|
||||
// above (MADT/HPET/FADT/MCFG) is all the kernel needs — CPUs, timers, PCIe,
|
||||
// and the power register map. The AML bytecode (device enumeration and the
|
||||
// sleep-state `_Sx` values for soft-off) is entirely the ring-3 acpi service's
|
||||
// job: it claims the acpi-tables node published below, parses the same blobs,
|
||||
// and both registers the `_HID` devices and owns S5. Not parsing ~0.5 MB of
|
||||
// AML in the kernel keeps boot latency off the critical, single-core path.
|
||||
|
||||
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||
// claimant. Kept even when the kernel-side device building (above) retires
|
||||
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
||||
// claimant — the sole path by which AML (devices + soft-off) reaches ring 3,
|
||||
// now that the kernel keeps only the *static* tables for itself.
|
||||
publishAcpiTablesNode(device_tree) catch {};
|
||||
}
|
||||
|
||||
@@ -461,12 +433,6 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
||||
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||
}
|
||||
|
||||
/// The number of Device objects in the namespace built during discovery, or 0.
|
||||
pub fn amlDeviceCount() usize {
|
||||
if (namespace) |*ns| return aml.deviceCount(ns);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
@@ -502,7 +468,7 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||
parseDmar(hal, header);
|
||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
||||
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||
addAmlBlock(sdt_physical);
|
||||
}
|
||||
// Any other signature is recognised but left opaque for now.
|
||||
@@ -740,8 +706,8 @@ const fadt_x_pm_tmr_blk = 208; // GAS
|
||||
const flag_reset_register_supported = 1 << 10;
|
||||
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||
|
||||
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
||||
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
||||
/// FADT -> the power register map (into `power_information`) and the DSDT address,
|
||||
/// whose bytecode is collected for the ring-3 parse. No AML interpretation here.
|
||||
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const len: usize = header.length;
|
||||
|
||||
@@ -38,6 +38,13 @@ pub const DeviceClass = enum(u32) {
|
||||
/// resources; the (class, subclass, protocol) triple that says what it is
|
||||
/// travels in the bus report's identity, not here.
|
||||
usb_device,
|
||||
/// A scanout framebuffer: a linear region of pixel memory the display service
|
||||
/// claims and maps. Unlike the other classes this one is not firmware-discovered
|
||||
/// — the kernel seeds it from the loader's [[boot-handoff]] framebuffer
|
||||
/// (`devices_broker.seedDisplay`). Its one `memory` resource is the framebuffer,
|
||||
/// flagged write-combining; the geometry to interpret it travels in
|
||||
/// `DeviceDescriptor.display`.
|
||||
display,
|
||||
unknown,
|
||||
};
|
||||
|
||||
@@ -59,10 +66,40 @@ pub const ResourceDescriptor = extern struct {
|
||||
kind: u64, // a ResourceKind value
|
||||
start: u64,
|
||||
len: u64,
|
||||
/// A bitmask of `resource_flag_*` hints. Zero for a plain register/RAM window;
|
||||
/// the kernel reads it when it maps the resource. Defaulted so every existing
|
||||
/// literal (which never set flags) keeps compiling and lays out identically.
|
||||
flags: u64 = 0,
|
||||
};
|
||||
|
||||
/// `ResourceDescriptor.flags`: map this `memory` resource **write-combining** rather
|
||||
/// than strong-uncacheable — for a framebuffer, where batched bursts to pixel memory
|
||||
/// are the whole point (an uncacheable framebuffer blit is glacial). See
|
||||
/// `mmio_map` (system/kernel/process.zig) and `setupPat` (…/x86_64/paging.zig).
|
||||
pub const resource_flag_write_combining: u64 = 1 << 0;
|
||||
|
||||
pub const maximum_device_resources = 8;
|
||||
|
||||
/// The byte order of a display's pixels — mirrors the loader's `PixelFormat`
|
||||
/// ([[boot-handoff]]) with the same numeric values, but lives here so user space
|
||||
/// (which must never import the loader↔kernel handoff) can name it. Only the two
|
||||
/// linear 32-bpp layouts a console can paint into exist; see docs/gop.md.
|
||||
pub const DisplayFormat = enum(u32) {
|
||||
rgbx = 0, // byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved
|
||||
bgrx = 1, // byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved
|
||||
};
|
||||
|
||||
/// The geometry of a `display` device's framebuffer, carried in its descriptor so a
|
||||
/// claiming driver knows how to interpret the pixel bytes its `memory` resource maps.
|
||||
/// `pitch` is bytes per row (may exceed `width * 4`; see docs/framebuffer.md).
|
||||
pub const DisplayInfo = extern struct {
|
||||
width: u32 = 0, // visible pixels per row
|
||||
height: u32 = 0, // visible rows
|
||||
pitch: u32 = 0, // bytes from one row's start to the next
|
||||
format: u32 = 0, // a DisplayFormat value
|
||||
refresh_hz: u32 = 0, // panel refresh rate from EDID (0 = unknown); see boot-handoff
|
||||
};
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||
pub const no_parent: u64 = ~@as(u64, 0);
|
||||
|
||||
@@ -92,4 +129,9 @@ pub const DeviceDescriptor = extern struct {
|
||||
resource_count: u64,
|
||||
hid: [8]u8,
|
||||
resources: [maximum_device_resources]ResourceDescriptor,
|
||||
// Framebuffer geometry, meaningful only when `class` is `DeviceClass.display`
|
||||
// (zeroed otherwise). Kept here — a class-specific field on the shared descriptor —
|
||||
// the same way `pci_class` is meaningful only for `pci_device` and `hid` only for
|
||||
// `acpi_device`.
|
||||
display: DisplayInfo = .{},
|
||||
};
|
||||
|
||||
@@ -22,13 +22,14 @@ pub const Resource = device_model.Resource;
|
||||
pub const ResourceKind = device_model.ResourceKind;
|
||||
pub const Hal = device_model.Hal;
|
||||
pub const PowerInformation = acpi.PowerInformation;
|
||||
pub const AmlStats = acpi.AmlStats;
|
||||
pub const PlatformInformation = acpi.PlatformInformation;
|
||||
pub const RegisterAccess = acpi.RegisterAccess;
|
||||
pub const IsoEntry = acpi.IsoEntry;
|
||||
pub const Cpu = acpi.Cpu;
|
||||
|
||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||
/// The FADT power register map discovery extracted (PM1 control, reset register),
|
||||
/// for kernel reboot and diagnostics. Sleep-state values are userspace's (S5 is
|
||||
/// owned by the ring-3 acpi service), so they are not here.
|
||||
pub fn powerInformation() PowerInformation {
|
||||
return acpi.power_information;
|
||||
}
|
||||
@@ -39,18 +40,6 @@ pub fn platformInformation() PlatformInformation {
|
||||
return acpi.platform_information;
|
||||
}
|
||||
|
||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
||||
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
||||
/// count against this.
|
||||
pub fn amlDeviceCount() usize {
|
||||
return acpi.amlDeviceCount();
|
||||
}
|
||||
|
||||
pub fn amlStats() AmlStats {
|
||||
return acpi.aml_stats;
|
||||
}
|
||||
|
||||
/// The usable logical processors discovered during enumeration — one entry per
|
||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||
@@ -92,12 +81,8 @@ pub fn discover(
|
||||
}
|
||||
|
||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls. Soft-off
|
||||
/// (S5) is not a kernel operation — the ring-3 acpi service owns it (docs/power.md).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
power.reboot(hal);
|
||||
}
|
||||
|
||||
/// Power the machine off (ACPI S5). Never returns on success.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
power.shutdown(hal);
|
||||
}
|
||||
|
||||
+10
-63
@@ -1,34 +1,17 @@
|
||||
//! Machine power control: enter ACPI mode, reboot, and power off (ACPI S5).
|
||||
//! Machine reboot: restart via the FADT reset register, with legacy fallbacks.
|
||||
//!
|
||||
//! Built entirely on the register map `acpi` extracted from the FADT plus the
|
||||
//! sleep-state (`_Sx`) types the AML submodule pulled from the DSDT, driven through the
|
||||
//! injected `Hal` (port I/O and MMIO). Nothing here is x86-specific beyond the
|
||||
//! well-known legacy reset fallbacks, which are guarded behind the ACPI methods.
|
||||
//!
|
||||
//! S3 (suspend-to-RAM) is stubbed: it needs a wake trampoline and device
|
||||
//! re-initialisation, a milestone of its own.
|
||||
//! Built on the register map `acpi` extracted from the FADT, driven through the
|
||||
//! injected `Hal` (port I/O and MMIO). Soft-off (ACPI S5) and suspend (S3) are
|
||||
//! **not** here: they need the AML sleep-state (`_Sx`) values, which the kernel no
|
||||
//! longer parses — the ring-3 acpi service owns power management (it re-parses the
|
||||
//! blobs and writes the PM1 control register itself). See docs/power.md. Reboot
|
||||
//! stays in the kernel because it needs no AML — only the FADT reset register and
|
||||
//! the well-known legacy fallbacks — so it survives as a last-resort restart.
|
||||
|
||||
const acpi = @import("acpi.zig");
|
||||
const device_model = @import("device-model.zig");
|
||||
const Hal = device_model.Hal;
|
||||
|
||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
||||
|
||||
/// Switch the platform into ACPI mode if it isn't already, so the PM1 control
|
||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
||||
pub fn enable(hal: Hal) void {
|
||||
const pi = acpi.power_information;
|
||||
if (!pi.pm1a_cnt.present()) return;
|
||||
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
||||
|
||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
||||
var spins: usize = 0;
|
||||
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||
}
|
||||
|
||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
@@ -48,42 +31,6 @@ pub fn reboot(hal: Hal) void {
|
||||
delay();
|
||||
}
|
||||
|
||||
/// Power the machine off via ACPI S5. Requires the soft-off (`_S5`) sleep type; if
|
||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
enable(hal);
|
||||
const pi = acpi.power_information;
|
||||
const s5 = pi.s5 orelse return;
|
||||
|
||||
if (pi.pm1a_cnt.present()) {
|
||||
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||
}
|
||||
if (pi.pm1b_cnt.present()) {
|
||||
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||
}
|
||||
delay();
|
||||
}
|
||||
|
||||
/// S3 suspend-to-RAM — not implemented (needs a wake path + device re-init).
|
||||
pub fn sleepS3(hal: Hal) error{Unsupported}!void {
|
||||
_ = hal;
|
||||
return error.Unsupported;
|
||||
}
|
||||
|
||||
/// The PM1 control write that requests sleep type `slp_typ`: SLP_TYP in bits
|
||||
/// [12:10], SLP_EN in bit 13.
|
||||
fn sleepValue(slp_typ: u8) u32 {
|
||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
||||
}
|
||||
|
||||
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
return p.*;
|
||||
}
|
||||
return hal.pioRead(register.width, @intCast(register.address));
|
||||
}
|
||||
|
||||
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
@@ -93,8 +40,8 @@ fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
||||
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||
/// A short busy-wait so a reset takes effect before we fall through to the next
|
||||
/// method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||
/// from being optimised away.
|
||||
fn delay() void {
|
||||
var i: usize = 0;
|
||||
|
||||
@@ -70,6 +70,104 @@ pub const Class = enum(u8) {
|
||||
_,
|
||||
};
|
||||
|
||||
/// A human-readable name for a device/interface class code, for logs. Unknown
|
||||
/// codes fall through to "class 0xNN".
|
||||
pub fn className(class: u8) []const u8 {
|
||||
return switch (@as(Class, @enumFromInt(class))) {
|
||||
.per_interface => "per-interface",
|
||||
.audio => "Audio",
|
||||
.communications => "Communications",
|
||||
.hid => "HID",
|
||||
.physical => "Physical",
|
||||
.image => "Image",
|
||||
.printer => "Printer",
|
||||
.mass_storage => "Mass Storage",
|
||||
.hub => "Hub",
|
||||
.cdc_data => "CDC Data",
|
||||
.smart_card => "Smart Card",
|
||||
.content_security => "Content Security",
|
||||
.video => "Video",
|
||||
.personal_healthcare => "Personal Healthcare",
|
||||
.audio_video => "Audio/Video",
|
||||
.billboard => "Billboard",
|
||||
.type_c_bridge => "Type-C Bridge",
|
||||
.bulk_display => "Bulk Display",
|
||||
.mctp => "MCTP",
|
||||
.i3c => "I3C",
|
||||
.diagnostic => "Diagnostic",
|
||||
.wireless_controller => "Wireless Controller",
|
||||
.miscellaneous => "Miscellaneous",
|
||||
.application_specific => "Application-specific",
|
||||
.vendor_specific => "Vendor-specific",
|
||||
_ => "Unknown",
|
||||
};
|
||||
}
|
||||
|
||||
/// The USB speed class (as xHCI reports it in PORTSC/slot contexts) named.
|
||||
pub fn speedName(speed: u32) []const u8 {
|
||||
return switch (speed) {
|
||||
1 => "Full-speed",
|
||||
2 => "Low-speed",
|
||||
3 => "High-speed",
|
||||
4 => "SuperSpeed",
|
||||
5 => "SuperSpeedPlus",
|
||||
else => "unknown-speed",
|
||||
};
|
||||
}
|
||||
|
||||
/// A USB3 Port Link State (xHCI PORTSC PLS field) named.
|
||||
pub fn linkStateName(pls: u32) []const u8 {
|
||||
return switch (pls) {
|
||||
0 => "U0",
|
||||
1 => "U1",
|
||||
2 => "U2",
|
||||
3 => "U3-suspended",
|
||||
4 => "Disabled",
|
||||
5 => "RxDetect",
|
||||
6 => "Inactive",
|
||||
7 => "Polling",
|
||||
8 => "Recovery",
|
||||
9 => "HotReset",
|
||||
10 => "Compliance",
|
||||
11 => "Test",
|
||||
15 => "Resume",
|
||||
else => "reserved",
|
||||
};
|
||||
}
|
||||
|
||||
/// The most useful readable name for an interface's (class, subclass, protocol)
|
||||
/// triple, decoding the well-known combinations recognizable in a log — e.g.
|
||||
/// "HID boot keyboard", "Mass Storage SCSI Bulk-Only", "Bluetooth". Falls back
|
||||
/// to the class name (and then "Unknown") for codes without a spelled-out combo.
|
||||
pub fn interfaceName(class: u8, subclass: u8, protocol: u8) []const u8 {
|
||||
return switch (@as(Class, @enumFromInt(class))) {
|
||||
.hid => if (subclass == @intFromEnum(hid.SubClass.boot)) switch (@as(hid.Protocol, @enumFromInt(protocol))) {
|
||||
.keyboard => "HID boot keyboard",
|
||||
.mouse => "HID boot mouse",
|
||||
else => "HID boot device",
|
||||
} else "HID",
|
||||
.mass_storage => switch (@as(mass_storage.Protocol, @enumFromInt(protocol))) {
|
||||
.bulk_only => "Mass Storage (Bulk-Only)",
|
||||
.uas => "Mass Storage (UAS)",
|
||||
else => "Mass Storage",
|
||||
},
|
||||
.hub => switch (@as(hub.Protocol, @enumFromInt(protocol))) {
|
||||
.super_speed => "Hub (SuperSpeed)",
|
||||
.hi_speed_multi_tt => "Hub (Hi-Speed multi-TT)",
|
||||
.hi_speed_single_tt => "Hub (Hi-Speed single-TT)",
|
||||
else => "Hub",
|
||||
},
|
||||
.wireless_controller => if (subclass == @intFromEnum(wireless_controller.SubClass.radio_frequency))
|
||||
wireless_controller.protocolName(protocol)
|
||||
else
|
||||
"Wireless Controller",
|
||||
.communications => communications.subclassName(subclass),
|
||||
.application_specific => application_specific.subclassName(subclass),
|
||||
.miscellaneous => "Miscellaneous",
|
||||
else => className(class),
|
||||
};
|
||||
}
|
||||
|
||||
// Subclass and protocol codes qualified by Class.hub. Hubs have no subclass codes; the
|
||||
// protocol distinguishes the hub's transaction-translator arrangement.
|
||||
pub const hub = struct {
|
||||
@@ -84,6 +182,16 @@ pub const hub = struct {
|
||||
super_speed = 0x03,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.full_speed => "full-speed",
|
||||
.hi_speed_single_tt => "Hi-Speed single-TT",
|
||||
.hi_speed_multi_tt => "Hi-Speed multi-TT",
|
||||
.super_speed => "SuperSpeed",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.hid.
|
||||
@@ -104,6 +212,23 @@ pub const hid = struct {
|
||||
mouse = 0x02,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.none => "none",
|
||||
.boot => "boot",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.none => "none",
|
||||
.keyboard => "keyboard",
|
||||
.mouse => "mouse",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.mass_storage. The subclass identifies the
|
||||
@@ -147,6 +272,33 @@ pub const mass_storage = struct {
|
||||
vendor_specific = 0xFF,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.not_reported => "SCSI (not reported)",
|
||||
.rbc => "RBC",
|
||||
.atapi => "ATAPI",
|
||||
.qic_157 => "QIC-157",
|
||||
.ufi => "UFI",
|
||||
.sff_8070i => "SFF-8070i",
|
||||
.scsi => "SCSI",
|
||||
.lsd_fs => "LSD FS",
|
||||
.ieee_1667 => "IEEE 1667",
|
||||
.vendor_specific => "vendor-specific",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.cbi_completion_interrupt => "CBI",
|
||||
.cbi => "CBI (no completion IRQ)",
|
||||
.bulk_only => "Bulk-Only",
|
||||
.uas => "UAS",
|
||||
.vendor_specific => "vendor-specific",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.communications (CDC). The protocol codes
|
||||
@@ -182,6 +334,25 @@ pub const communications = struct {
|
||||
network_control = 0x0D,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.direct_line => "Direct Line",
|
||||
.abstract_control => "Abstract Control (modem/serial)",
|
||||
.telephone => "Telephone",
|
||||
.multi_channel => "Multi-Channel",
|
||||
.capi => "CAPI",
|
||||
.ethernet => "Ethernet",
|
||||
.atm => "ATM",
|
||||
.wireless_handset => "Wireless Handset",
|
||||
.device_management => "Device Management",
|
||||
.mobile_direct_line => "Mobile Direct Line",
|
||||
.obex => "OBEX",
|
||||
.ethernet_emulation => "Ethernet Emulation",
|
||||
.network_control => "Network Control",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.wireless_controller.
|
||||
@@ -204,6 +375,23 @@ pub const wireless_controller = struct {
|
||||
bluetooth_amp = 0x04,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.radio_frequency => "RF",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.bluetooth => "Bluetooth",
|
||||
.ultra_wideband => "Ultra-Wideband",
|
||||
.remote_ndis => "Remote NDIS",
|
||||
.bluetooth_amp => "Bluetooth AMP",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.miscellaneous.
|
||||
@@ -221,6 +409,20 @@ pub const miscellaneous = struct {
|
||||
interface_association = 0x01,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.common => "common",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.interface_association => "Interface Association",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.application_specific.
|
||||
@@ -234,6 +436,15 @@ pub const application_specific = struct {
|
||||
test_and_measurement = 0x03,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.firmware_upgrade => "Device Firmware Upgrade",
|
||||
.irda_bridge => "IrDA Bridge",
|
||||
.test_and_measurement => "Test & Measurement",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// Pack a (class, subclass, protocol) triple into one 0xCCSSPP value — the
|
||||
@@ -280,6 +491,30 @@ test "class codes match the USB-IF assignments" {
|
||||
_ = application_specific.SubClass.firmware_upgrade;
|
||||
}
|
||||
|
||||
test "readable names decode the well-known triples" {
|
||||
const std = @import("std");
|
||||
const eql = std.testing.expectEqualStrings;
|
||||
|
||||
try eql("Hub", className(0x09));
|
||||
try eql("Unknown", className(0x42));
|
||||
|
||||
// interfaceName decodes the combos we log.
|
||||
try eql("HID boot keyboard", interfaceName(0x03, 0x01, 0x01));
|
||||
try eql("HID boot mouse", interfaceName(0x03, 0x01, 0x02));
|
||||
try eql("Mass Storage (Bulk-Only)", interfaceName(0x08, 0x06, 0x50));
|
||||
try eql("Hub (SuperSpeed)", interfaceName(0x09, 0x00, 0x03));
|
||||
try eql("Bluetooth", interfaceName(0xE0, 0x01, 0x01));
|
||||
|
||||
// The per-enum name functions.
|
||||
try eql("Bulk-Only", mass_storage.protocolName(0x50));
|
||||
try eql("SCSI", mass_storage.subclassName(0x06));
|
||||
try eql("Bluetooth", wireless_controller.protocolName(0x01));
|
||||
try eql("SuperSpeed", hub.protocolName(0x03));
|
||||
try eql("keyboard", hid.protocolName(0x01));
|
||||
try eql("SuperSpeed", speedName(4));
|
||||
try eql("Polling", linkStateName(7));
|
||||
}
|
||||
|
||||
test "packTriple / unpackTriple round-trip the identity a bus driver reports" {
|
||||
const std = @import("std");
|
||||
const expectEqual = std.testing.expectEqual;
|
||||
|
||||
@@ -16,11 +16,6 @@ const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Log a discovered function with its (class / subclass / prog-IF) triple decoded
|
||||
/// to human names — the boot-log breadcrumb that says *what* the hardware is, so
|
||||
/// "class 0x01 (Mass Storage Controller) subclass 0x06 (Serial ATA Controller)
|
||||
@@ -74,7 +69,7 @@ fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) voi
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(bridge_id)) {
|
||||
writeLine("/system/drivers/pci-bus: unable to claim bridge device {d}\n", .{bridge_id});
|
||||
std.log.info("unable to claim bridge device {d}", .{bridge_id});
|
||||
return false;
|
||||
}
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
@@ -85,7 +80,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == bridge_id) break d;
|
||||
} else {
|
||||
writeLine("/system/drivers/pci-bus: device {d} not in the device tree\n", .{bridge_id});
|
||||
std.log.info("device {d} not in the device tree", .{bridge_id});
|
||||
return false;
|
||||
};
|
||||
// Resource 0 is the ECAM window (1 MiB of config space per bus); the bus
|
||||
@@ -159,7 +154,7 @@ fn scan() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
writeLine("/system/drivers/pci-bus: {d} functions found\n", .{found});
|
||||
std.log.info("{d} functions found", .{found});
|
||||
}
|
||||
|
||||
/// Register one function under the bridge and report it to the manager. The
|
||||
@@ -229,7 +224,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
}
|
||||
|
||||
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||
writeLine("/system/drivers/pci-bus: register refused for {d}:{d}.{d}\n", .{ bus, dev, function });
|
||||
std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function });
|
||||
return;
|
||||
};
|
||||
const report = protocol.ChildAdded{
|
||||
@@ -240,7 +235,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("/system/drivers/pci-bus: child report for {d}:{d}.{d} failed\n", .{ bus, dev, function });
|
||||
std.log.info("child report for {d}:{d}.{d} failed", .{ bus, dev, function });
|
||||
};
|
||||
}
|
||||
|
||||
@@ -255,7 +250,7 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
||||
bridge_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/pci-bus: malformed bridge device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed bridge device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
@@ -263,8 +258,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -23,11 +23,6 @@ const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
const protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
|
||||
/// registering) is still coming up.
|
||||
fn lookupBus() ?ipc.Handle {
|
||||
@@ -75,14 +70,14 @@ pub fn main(init: runtime.process.Init) void {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: no HID argument\n");
|
||||
return;
|
||||
}
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: starting for hid {s}\n", .{hid});
|
||||
std.log.info("starting for hid {s}", .{hid});
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (device.findDeviceDescriptorByHid(buffer, hid) == null) {
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: no device for hid {s}\n", .{hid});
|
||||
std.log.info("no device for hid {s}", .{hid});
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -90,7 +85,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
// absent (as today) it defaults to us.
|
||||
const layout_name = init.arguments.get(2) orelse "us";
|
||||
const layout = xkb.byName(layout_name) orelse xkb.us;
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: layout {s}\n", .{layout.name});
|
||||
std.log.info("layout {s}", .{layout.name});
|
||||
|
||||
// Attach to the bus: hand it our endpoint, and it forwards every byte the
|
||||
// keyboard sends (it owns the controller; we own the decoding).
|
||||
@@ -177,8 +172,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -22,11 +22,6 @@ const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
const protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
|
||||
/// registering) is still coming up.
|
||||
fn lookupBus() ?ipc.Handle {
|
||||
@@ -54,14 +49,14 @@ pub fn main(init: runtime.process.Init) void {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: no HID argument\n");
|
||||
return;
|
||||
}
|
||||
writeLine("/system/drivers/ps2-bus/mouse: starting for hid {s}\n", .{hid});
|
||||
std.log.info("starting for hid {s}", .{hid});
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (device.findDeviceDescriptorByHid(buffer, hid) == null) {
|
||||
writeLine("/system/drivers/ps2-bus/mouse: no device for hid {s}\n", .{hid});
|
||||
if (ps2.findMouseDescriptor(buffer) == null) {
|
||||
std.log.info("no device for hid {s}", .{hid});
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -137,8 +132,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -16,13 +16,6 @@ const ps2 = @import("ps2-library.zig");
|
||||
const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so output
|
||||
/// from the child drivers (which run concurrently) can never interleave with it.
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Ask the device on `port` what it is, then spawn the matching driver from the
|
||||
/// initial-ramdisk, handing it the device's HID as argv[1]. The driver is chosen
|
||||
/// from what the device reports, not from the port number. Returns the identified
|
||||
@@ -30,19 +23,19 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
/// attaches, or null if nothing was spawned.
|
||||
fn spawnIdentifiedDriver(controller: ps2.Controller, port: ps2.Port) ?ps2.DeviceType {
|
||||
const device_type = controller.identifyDevice(port) orelse {
|
||||
writeLine("/system/drivers/ps2-bus: identify timed out on port {s}\n", .{@tagName(port)});
|
||||
std.log.info("identify timed out on port {s}", .{@tagName(port)});
|
||||
return null;
|
||||
};
|
||||
const driver_name = device_type.driverName() orelse {
|
||||
writeLine("/system/drivers/ps2-bus: unrecognized device on port {s}\n", .{@tagName(port)});
|
||||
std.log.info("unrecognized device on port {s}", .{@tagName(port)});
|
||||
return null;
|
||||
};
|
||||
const hid = device_type.hid() orelse "";
|
||||
if (runtime.system.spawnWithArguments(driver_name, &.{hid}) != null) {
|
||||
writeLine("/system/drivers/ps2-bus: port {s} is a {s}, spawned {s}\n", .{ @tagName(port), hid, driver_name });
|
||||
std.log.info("port {s} is a {s}, spawned {s}", .{ @tagName(port), hid, driver_name });
|
||||
return device_type;
|
||||
}
|
||||
writeLine("/system/drivers/ps2-bus: failed to spawn {s}\n", .{driver_name});
|
||||
std.log.info("failed to spawn {s}", .{driver_name});
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -83,7 +76,7 @@ fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
const device_type = maybe_type orelse continue;
|
||||
if (@intFromEnum(device_type) != request.device_type) continue;
|
||||
port_endpoints[port_index] = endpoint;
|
||||
writeLine("/system/drivers/ps2-bus: {s} driver attached\n", .{@tagName(device_type)});
|
||||
std.log.info("{s} driver attached", .{@tagName(device_type)});
|
||||
return reply.write(out, .ok);
|
||||
}
|
||||
return reply.write(out, .no_such_device);
|
||||
@@ -244,7 +237,7 @@ pub fn main() void {
|
||||
// to the *port*, whatever device identify found on it.
|
||||
var maybe_auxiliary_interrupt: ?struct { device_id: u64, interrupt_index: u64, gsi: u64 } = null;
|
||||
if (port_device_types[@intFromEnum(ps2.Port.two)] != null) {
|
||||
if (device.findDeviceDescriptorByHid(buffer, acpi_ids.HardwareId.ps2_mouse.hid())) |descriptor| {
|
||||
if (ps2.findMouseDescriptor(buffer)) |descriptor| {
|
||||
if (findInterruptResourceIndex(descriptor)) |auxiliary_index| {
|
||||
if (device.claim(descriptor.id) and device.irqBind(descriptor.id, auxiliary_index, endpoint)) {
|
||||
maybe_auxiliary_interrupt = .{
|
||||
@@ -308,8 +301,3 @@ pub fn main() void {
|
||||
reply_len = handleAttach(receive[0..got.len], got, &reply_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -270,6 +270,22 @@ pub const DeviceType = enum(u32) {
|
||||
}
|
||||
};
|
||||
|
||||
/// The `_HID`s a PS/2 pointing device (the controller's aux channel) can enumerate
|
||||
/// under. It is the same 8042 mouse channel whichever id the firmware chose:
|
||||
/// QEMU/OVMF report the generic `.ps2_mouse` (PNP0F13), VirtualBox reports
|
||||
/// `.microsoft_ps2_mouse` (PNP0F03). Both mean "the mouse on port two".
|
||||
pub const mouse_hardware_ids = [_]acpi_ids.HardwareId{ .ps2_mouse, .microsoft_ps2_mouse };
|
||||
|
||||
/// Find the aux (mouse) device's ACPI node, whichever of the PS/2-mouse `_HID`s the
|
||||
/// firmware used — the bus needs it to bind IRQ12, and the mouse driver to confirm
|
||||
/// its device is present. Returns the first match, or null if none is reported.
|
||||
pub fn findMouseDescriptor(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
for (mouse_hardware_ids) |id| {
|
||||
if (device.findDeviceDescriptorByHid(buffer, id.hid())) |descriptor| return descriptor;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- the bus <-> child-driver forwarding protocol -----------------------------
|
||||
//
|
||||
// The 8042's ports and IRQ1 live on the PNP0303 node that only the ps2-bus driver
|
||||
|
||||
@@ -23,11 +23,6 @@ const ipc = runtime.ipc;
|
||||
const process = runtime.process;
|
||||
const input_protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The modifier state a character lookup needs — derived from the report's
|
||||
// modifier byte, plus the driver-tracked caps-lock toggle.
|
||||
const ModifierSnapshot = struct {
|
||||
@@ -71,7 +66,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
const device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-hid/keyboard: malformed device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
const layout = xkb.byName(init.arguments.get(2) orelse "us") orelse xkb.us;
|
||||
@@ -82,7 +77,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
}
|
||||
var device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-hid/keyboard: could not open device {d}\n", .{device_id});
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return;
|
||||
};
|
||||
const endpoint = device.findEndpoint(runtime.usb.transfer_type_interrupt, true) orelse {
|
||||
@@ -104,7 +99,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
_ = process.bindSignals(device.endpoint);
|
||||
writeLine("/system/drivers/usb-hid/keyboard: ok (device {d}, interface {d}, layout {s})\n", .{ device_id, device.interface_number, layout.name });
|
||||
std.log.info("ok (device {d}, interface {d}, layout {s})", .{ device_id, device.interface_number, layout.name });
|
||||
|
||||
var decoder = hid.KeyboardDecoder{};
|
||||
var caps_lock = false;
|
||||
@@ -153,6 +148,15 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.character = character,
|
||||
.modifiers = modifier_word,
|
||||
});
|
||||
// Echo the character to the log — a simple end-to-end
|
||||
// keyboard check on real hardware: type a known phrase,
|
||||
// then read it back from usb-hid-keyboard.log (or watch
|
||||
// it appear live on screen in a -Ddiagnose boot, where
|
||||
// the kernel console is a log sink). Printable ASCII and
|
||||
// newline only; other keys are left to the input service.
|
||||
if (character == '\n' or (character >= 0x20 and character < 0x7F)) {
|
||||
_ = runtime.system.write(&[1]u8{@intCast(character)});
|
||||
}
|
||||
}
|
||||
},
|
||||
.released => {
|
||||
@@ -167,8 +171,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -18,11 +18,6 @@ const ipc = runtime.ipc;
|
||||
const process = runtime.process;
|
||||
const input_protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The current pressed-button bitmask in input-protocol terms.
|
||||
fn buttonMask(buttons: u8) u32 {
|
||||
var mask: u32 = 0;
|
||||
@@ -38,7 +33,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
const device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-hid/mouse: malformed device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -47,7 +42,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
}
|
||||
var device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-hid/mouse: could not open device {d}\n", .{device_id});
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return;
|
||||
};
|
||||
const endpoint = device.findEndpoint(runtime.usb.transfer_type_interrupt, true) orelse {
|
||||
@@ -67,7 +62,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
_ = process.bindSignals(device.endpoint);
|
||||
writeLine("/system/drivers/usb-hid/mouse: ok (device {d}, interface {d})\n", .{ device_id, device.interface_number });
|
||||
std.log.info("ok (device {d}, interface {d})", .{ device_id, device.interface_number });
|
||||
|
||||
var previous_buttons: u8 = 0;
|
||||
var receive: [64]u8 = undefined;
|
||||
@@ -133,8 +128,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@ const op_inquiry: u8 = 0x12;
|
||||
const op_read_capacity_10: u8 = 0x25;
|
||||
const op_read_10: u8 = 0x28;
|
||||
const op_write_10: u8 = 0x2A;
|
||||
const op_synchronize_cache_10: u8 = 0x35;
|
||||
|
||||
/// INQUIRY: standard device data (36 bytes: peripheral type, removable, vendor
|
||||
/// and product strings).
|
||||
@@ -57,6 +58,16 @@ pub fn write10(lba: u32, blocks: u16) [10]u8 {
|
||||
return cdb;
|
||||
}
|
||||
|
||||
/// SYNCHRONIZE CACHE(10): commit the device's write cache to stable media. LBA 0
|
||||
/// and block count 0 mean "the whole medium". No data stage. Without this a write
|
||||
/// can sit in the USB flash controller's cache and be lost if power is cut right
|
||||
/// after — which is exactly what a shutdown-time log flush hits on real hardware.
|
||||
pub fn synchronizeCache10() [10]u8 {
|
||||
var cdb = [_]u8{0} ** 10;
|
||||
cdb[0] = op_synchronize_cache_10;
|
||||
return cdb;
|
||||
}
|
||||
|
||||
/// Decode an 8-byte READ CAPACITY(10) reply.
|
||||
pub fn parseCapacity(bytes: [8]u8) struct { last_lba: u32, block_size: u32 } {
|
||||
return .{
|
||||
|
||||
@@ -18,11 +18,6 @@ const bot = @import("bulk-only-transport.zig");
|
||||
const block_protocol = @import("block-protocol");
|
||||
const dma = runtime.dma;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
var device_id: u64 = 0;
|
||||
var device: runtime.usb.Device = undefined;
|
||||
var bulk_in: runtime.usb.Endpoint = undefined;
|
||||
@@ -66,6 +61,12 @@ fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length
|
||||
return status.status == @intFromEnum(bot.CommandStatus.passed);
|
||||
}
|
||||
|
||||
/// Set when bring-up failed with the device PRESENT (an opened device that then
|
||||
/// failed a step): main exits nonzero, and the device manager restarts us with
|
||||
/// backoff — a transient failure heals instead of leaving storage down forever.
|
||||
/// Device-absent paths stay clean exits: nothing to serve, nothing to retry.
|
||||
var bring_up_failed = false;
|
||||
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!runtime.usb.helloManager(device_id)) {
|
||||
@@ -73,15 +74,17 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
return false;
|
||||
}
|
||||
device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-storage: could not open device {d}\n", .{device_id});
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return false;
|
||||
};
|
||||
bulk_in = device.findEndpoint(runtime.usb.transfer_type_bulk, true) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-storage: no bulk-IN endpoint\n");
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
};
|
||||
bulk_out = device.findEndpoint(runtime.usb.transfer_type_bulk, false) orelse {
|
||||
_ = runtime.system.write("/system/drivers/usb-storage: no bulk-OUT endpoint\n");
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
};
|
||||
command_wrapper = dma.alloc(4096, dma.coherent) orelse return false;
|
||||
@@ -104,6 +107,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
const capacity_command = scsi.readCapacity10();
|
||||
if (!transact(&capacity_command, true, command_data.physical, 8)) {
|
||||
_ = runtime.system.write("/system/drivers/usb-storage: READ CAPACITY failed\n");
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
}
|
||||
var capacity_bytes: [8]u8 = undefined;
|
||||
@@ -112,14 +116,14 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
const capacity = scsi.parseCapacity(capacity_bytes);
|
||||
block_size = capacity.block_size;
|
||||
block_count = @as(u64, capacity.last_lba) + 1;
|
||||
writeLine("/system/drivers/usb-storage: ready ({d} blocks x {d} bytes)\n", .{ block_count, block_size });
|
||||
std.log.info("ready ({d} blocks x {d} bytes)", .{ block_count, block_size });
|
||||
|
||||
// Self-check: read block 0 and log its trailing signature (0x55AA for a boot
|
||||
// sector) — proof READ(10) works end to end over the bulk path.
|
||||
const read0 = scsi.read10(0, 1);
|
||||
if (block_size <= 4096 and transact(&read0, true, command_data.physical, block_size)) {
|
||||
const sector: [*]const u8 = @ptrFromInt(command_data.virtual);
|
||||
writeLine("/system/drivers/usb-storage: block 0 signature 0x{x:0>2}{x:0>2}\n", .{ sector[510], sector[511] });
|
||||
std.log.info("block 0 signature 0x{x:0>2}{x:0>2}", .{ sector[510], sector[511] });
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -147,6 +151,14 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.flush) => {
|
||||
// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data
|
||||
// stage. Makes prior writes durable before a caller (init at shutdown)
|
||||
// cuts power. A device without a volatile cache reports success anyway.
|
||||
const cdb = scsi.synchronizeCache10();
|
||||
const ok = transact(&cdb, false, 0, 0);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = 0 });
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
@@ -163,7 +175,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-storage: malformed device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(block_protocol.message_maximum, .{
|
||||
@@ -171,9 +183,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
// A failure exit (nonzero -> .aborted) tells the device manager to restart
|
||||
// us with backoff; a clean return means there was nothing to serve.
|
||||
if (bring_up_failed) runtime.system.exit(1);
|
||||
}
|
||||
|
||||
@@ -64,13 +64,6 @@ fn reportEndpointFor(device_token: u64) ?usize {
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so
|
||||
/// concurrent instances (one per controller) can never interleave mid-line.
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
var controller_id: u64 = protocol.no_device;
|
||||
|
||||
/// Claim the assigned controller, find its register window, and hello the
|
||||
@@ -79,7 +72,7 @@ var controller_id: u64 = protocol.no_device;
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
if (!device.claim(controller_id)) {
|
||||
writeLine("/system/drivers/usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
||||
std.log.info("unable to claim controller device {d}", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -92,7 +85,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
writeLine("/system/drivers/usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
||||
std.log.info("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
@@ -105,10 +98,10 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
break resource;
|
||||
}
|
||||
} else {
|
||||
writeLine("/system/drivers/usb-xhci-bus: controller device {d} has no register BAR\n", .{controller_id});
|
||||
std.log.info("controller device {d} has no register BAR", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
||||
std.log.info("claimed controller device {d} (registers at 0x{x}, {d} bytes)", .{
|
||||
controller_id,
|
||||
register_window.start,
|
||||
register_window.len,
|
||||
@@ -124,7 +117,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: controller reset/bring-up failed\n");
|
||||
return false;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: controller running ({d} slots, {d}-byte contexts)\n", .{
|
||||
std.log.info("controller running ({d} slots, {d}-byte contexts)", .{
|
||||
controller.?.max_slots,
|
||||
controller.?.context_size,
|
||||
});
|
||||
@@ -150,6 +143,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: no device manager to hello\n");
|
||||
return false;
|
||||
};
|
||||
manager_handle = h; // the tick's hot-plug dispatch reports through this
|
||||
const hello = protocol.Hello{ .role = @intFromEnum(protocol.Role.bus), .device_id = controller_id };
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = runtime.ipc.call(h, std.mem.asBytes(&hello), &reply) catch {
|
||||
@@ -193,46 +187,179 @@ fn speedName(speed: u32) []const u8 {
|
||||
/// report one child per interface — carrying the interface's (class, subclass,
|
||||
/// protocol) triple as identity, which is what the device manager matches a
|
||||
/// class driver against.
|
||||
var manager_handle: ?runtime.ipc.Handle = null;
|
||||
|
||||
// Per-root-port connected state from the previous tick, so the poll acts on
|
||||
// empty->connected transitions (edge), never re-attempting a level every tick.
|
||||
var prev_connected: [64]bool = [_]bool{false} ** 64;
|
||||
|
||||
fn scanPorts(manager: runtime.ipc.Handle) void {
|
||||
const engine = if (controller) |*c| c else {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: controller not initialised\n");
|
||||
return;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: {d} root-hub ports\n", .{engine.max_ports});
|
||||
std.log.info("{d} root-hub ports", .{engine.max_ports});
|
||||
|
||||
var port: u32 = 1;
|
||||
var connected: u32 = 0;
|
||||
while (port <= engine.max_ports) : (port += 1) {
|
||||
const port_status = engine.portStatus(port);
|
||||
if (port_status & 1 == 0) continue; // CCS: nothing connected
|
||||
if (!engine.portConnected(port)) continue;
|
||||
if (port < prev_connected.len) prev_connected[port] = true; // don't re-fire the poll for these
|
||||
connected += 1;
|
||||
const speed = (port_status >> 10) & 0xF; // the PORTSC port-speed class
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} connected — {s} (speed class {d})\n", .{ port, speedName(speed), speed });
|
||||
bringUpPort(manager, engine, port);
|
||||
}
|
||||
if (connected == 0) {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: no devices connected\n");
|
||||
engine.dumpPortTopology(); // help diagnose an empty scan: the xECP map + raw PORTSC
|
||||
}
|
||||
}
|
||||
|
||||
const usb_device = engine.setupDevice(port, speed) orelse {
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} device setup failed\n", .{port});
|
||||
continue;
|
||||
};
|
||||
if (!engine.enumerate(usb_device)) {
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} enumeration failed\n", .{port});
|
||||
continue;
|
||||
}
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} device vendor 0x{x:0>4} product 0x{x:0>4}, {d} interface(s)\n", .{
|
||||
port,
|
||||
usb_device.device_descriptor.vendor_id,
|
||||
usb_device.device_descriptor.product_id,
|
||||
usb_device.interface_count,
|
||||
});
|
||||
/// Bring up whatever is on `port`: setup + enumerate + register/report one child
|
||||
/// per interface. Shared by the boot scan and hot-plug (a port-change event with
|
||||
/// the port now connected).
|
||||
fn bringUpPort(manager: runtime.ipc.Handle, engine: *library.Controller, port: u32) void {
|
||||
const speed = (engine.portStatus(port) >> 10) & 0xF; // the PORTSC port-speed class
|
||||
std.log.info("port {d} connected — {s} (speed class {d})", .{ port, speedName(speed), speed });
|
||||
|
||||
for (usb_device.interfaces[0..usb_device.interface_count]) |*interface| {
|
||||
// Record the id each interface was registered as, so a class driver
|
||||
// opening the interface (by that id) resolves to it.
|
||||
if (reportInterface(manager, port, interface.*)) |registered| {
|
||||
interface.registered_device_id = registered;
|
||||
}
|
||||
const usb_device = engine.setupDevice(port, speed) orelse {
|
||||
std.log.info("port {d} device setup failed", .{port});
|
||||
return;
|
||||
};
|
||||
if (!engine.enumerate(usb_device)) {
|
||||
std.log.info("port {d} enumeration failed", .{port});
|
||||
return;
|
||||
}
|
||||
var maker_buffer: [64]u8 = undefined;
|
||||
var product_buffer: [64]u8 = undefined;
|
||||
const maker = engine.readString(usb_device, @intFromEnum(usb_device.device_descriptor.manufacturer_index), &maker_buffer) orelse "?";
|
||||
const product = engine.readString(usb_device, @intFromEnum(usb_device.device_descriptor.product_index), &product_buffer) orelse "?";
|
||||
std.log.info("port {d} device: {s} \"{s} {s}\" (0x{x:0>4}:0x{x:0>4}), {d} interface(s)", .{
|
||||
port,
|
||||
usb_ids.className(usb_device.device_descriptor.device_class),
|
||||
maker,
|
||||
product,
|
||||
usb_device.device_descriptor.vendor_id,
|
||||
usb_device.device_descriptor.product_id,
|
||||
usb_device.interface_count,
|
||||
});
|
||||
|
||||
for (usb_device.interfaces[0..usb_device.interface_count]) |*interface| {
|
||||
// Record the id each interface was registered as, so a class driver
|
||||
// opening the interface (by that id) resolves to it.
|
||||
if (reportInterface(manager, port, interface.*)) |registered| {
|
||||
interface.registered_device_id = registered;
|
||||
}
|
||||
}
|
||||
if (connected == 0) _ = runtime.system.write("/system/drivers/usb-xhci-bus: no devices connected\n");
|
||||
|
||||
// A hub (class 9) is bus infrastructure the bus drives itself: configure it
|
||||
// and power its downstream ports (docs/usb-hub.md). Its interfaces are still
|
||||
// reported above, but no external class driver binds it.
|
||||
if (deviceIsHub(usb_device)) _ = engine.setupHub(usb_device);
|
||||
}
|
||||
|
||||
// A compact topology-unique port key for a hub downstream port: 1000 + slot*100
|
||||
// + port. Stays a few digits (the id tag "P<key>I<iface>" has an 8-byte cap)
|
||||
// while never colliding with a root port (1..N) or another (hub, port).
|
||||
fn hubPortKey(hub_slot: u8, port: u16) u32 {
|
||||
return 1000 + @as(u32, hub_slot) * 100 + port;
|
||||
}
|
||||
|
||||
/// Service a change on hub downstream `port`: dispatch a connect (enumerate the
|
||||
/// new device) or a disconnect (tear the old one down). Recurses for a hub
|
||||
/// behind a hub — a nested hub is set up on connect and its downstream devices
|
||||
/// torn down first on disconnect.
|
||||
fn bringUpBehindHub(manager: runtime.ipc.Handle, engine: *library.Controller, hub: *library.Device, port: u16) void {
|
||||
const status = engine.hubPortStatusAck(hub, port) orelse return;
|
||||
const connected = library.Controller.hubPortConnected(status);
|
||||
const existing = engine.deviceOnHubPort(hub, port);
|
||||
if (connected != (existing != null)) // only when a device appears or leaves — not empty seed-sweep ports
|
||||
std.log.info("hub slot {d} port {d}: {s} (status 0x{x:0>4})", .{ hub.slot_id, port, if (connected) "device connected" else "device removed", status & 0xFFFF });
|
||||
|
||||
if (!connected) {
|
||||
if (existing) |dev| tearDownHubDevice(manager, engine, dev);
|
||||
return;
|
||||
}
|
||||
if (existing != null) return; // already up
|
||||
|
||||
const usb_device = engine.serviceHubPort(hub, port) orelse return;
|
||||
if (!engine.enumerate(usb_device)) {
|
||||
std.log.info("hub slot {d} port {d}: enumeration failed", .{ hub.slot_id, port });
|
||||
return;
|
||||
}
|
||||
var maker_buffer: [64]u8 = undefined;
|
||||
var product_buffer: [64]u8 = undefined;
|
||||
const maker = engine.readString(usb_device, @intFromEnum(usb_device.device_descriptor.manufacturer_index), &maker_buffer) orelse "?";
|
||||
const product = engine.readString(usb_device, @intFromEnum(usb_device.device_descriptor.product_index), &product_buffer) orelse "?";
|
||||
std.log.info("hub slot {d} port {d} device: {s} \"{s} {s}\" (0x{x:0>4}:0x{x:0>4}), {d} interface(s)", .{
|
||||
hub.slot_id, port,
|
||||
usb_ids.className(usb_device.device_descriptor.device_class),
|
||||
maker,
|
||||
product,
|
||||
usb_device.device_descriptor.vendor_id,
|
||||
usb_device.device_descriptor.product_id,
|
||||
usb_device.interface_count,
|
||||
});
|
||||
for (usb_device.interfaces[0..usb_device.interface_count]) |*interface| {
|
||||
if (reportInterface(manager, hubPortKey(hub.slot_id, port), interface.*)) |registered| {
|
||||
interface.registered_device_id = registered;
|
||||
}
|
||||
}
|
||||
if (deviceIsHub(usb_device)) _ = engine.setupHub(usb_device);
|
||||
}
|
||||
|
||||
/// Tear down a device that disconnected from a hub: recursively tear down its
|
||||
/// own downstream devices first if it is a hub, report each interface removed,
|
||||
/// then Disable Slot. Mirrors tearDownPort for a hub-attached device.
|
||||
fn tearDownHubDevice(manager: runtime.ipc.Handle, engine: *library.Controller, dev: *library.Device) void {
|
||||
// A hub that left takes its whole subtree with it — tear children down first.
|
||||
if (dev.is_hub) {
|
||||
while (engine.nextChildOf(dev.slot_id, 0)) |child| tearDownHubDevice(manager, engine, child);
|
||||
}
|
||||
std.log.info("hub device slot {d} disconnected", .{dev.slot_id});
|
||||
const key = hubPortKey(dev.parent_slot, dev.parent_port);
|
||||
for (dev.interfaces[0..dev.interface_count]) |*interface| {
|
||||
if (interface.registered_device_id == 0) continue;
|
||||
const event = protocol.ChildRemoved{
|
||||
.parent = controller_id,
|
||||
.bus_address = (@as(u64, key) << 8) | interface.number,
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager, std.mem.asBytes(&event), &reply) catch {};
|
||||
interface.registered_device_id = 0;
|
||||
}
|
||||
engine.tearDownDevice(dev);
|
||||
}
|
||||
|
||||
/// Whether an enumerated device is a hub — class 9 at the device or the
|
||||
/// interface level (a hub's single interface is class 9/0/0).
|
||||
fn deviceIsHub(usb_device: *const library.Device) bool {
|
||||
if (usb_device.device_descriptor.device_class == @intFromEnum(usb_ids.Class.hub)) return true;
|
||||
for (usb_device.interfaces[0..usb_device.interface_count]) |interface| {
|
||||
if (interface.class == @intFromEnum(usb_ids.Class.hub)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Tear down whatever was on `port` after an unplug: report each registered
|
||||
/// interface as removed (the manager prunes the node, notifies watchers, and
|
||||
/// stops the class driver's world honestly), then release the controller-side
|
||||
/// device state (Disable Slot).
|
||||
fn tearDownPort(manager: runtime.ipc.Handle, engine: *library.Controller, port: u32) void {
|
||||
const usb_device = engine.deviceOnPort(port) orelse return;
|
||||
std.log.info("port {d} disconnected", .{port});
|
||||
for (usb_device.interfaces[0..usb_device.interface_count]) |*interface| {
|
||||
if (interface.registered_device_id == 0) continue;
|
||||
const event = protocol.ChildRemoved{
|
||||
.parent = controller_id,
|
||||
.bus_address = (@as(u64, port) << 8) | interface.number,
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager, std.mem.asBytes(&event), &reply) catch {
|
||||
std.log.info("child-removed report for port {d} interface {d} failed", .{ port, interface.number });
|
||||
};
|
||||
interface.registered_device_id = 0;
|
||||
}
|
||||
engine.tearDownDevice(usb_device);
|
||||
}
|
||||
|
||||
/// Register one interface as a resource-less child of the controller and report
|
||||
@@ -259,7 +386,7 @@ fn reportInterface(manager: runtime.ipc.Handle, port: u32, interface: library.In
|
||||
descriptor.hid_len = hid_text.len;
|
||||
@memcpy(descriptor.hid[0..hid_text.len], hid_text);
|
||||
const registered = device.register(controller_id, &descriptor) orelse {
|
||||
writeLine("/system/drivers/usb-xhci-bus: register refused for port {d} interface {d}\n", .{ port, interface.number });
|
||||
std.log.info("register refused for port {d} interface {d}", .{ port, interface.number });
|
||||
return null;
|
||||
};
|
||||
|
||||
@@ -271,12 +398,13 @@ fn reportInterface(manager: runtime.ipc.Handle, port: u32, interface: library.In
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("/system/drivers/usb-xhci-bus: child report for port {d} interface {d} failed\n", .{ port, interface.number });
|
||||
std.log.info("child report for port {d} interface {d} failed", .{ port, interface.number });
|
||||
return null;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} interface {d} class {d}/{d}/{d} registered as device {d}\n", .{
|
||||
std.log.info("port {d} interface {d}: {s} ({d}/{d}/{d}) registered as device {d}", .{
|
||||
port,
|
||||
interface.number,
|
||||
usb_ids.interfaceName(interface.class, interface.subclass, interface.protocol),
|
||||
interface.class,
|
||||
interface.subclass,
|
||||
interface.protocol,
|
||||
@@ -387,6 +515,52 @@ fn onNotification(badge: u64) void {
|
||||
if (badge & runtime.ipc.notify_timer_bit == 0) return;
|
||||
if (controller) |*engine| {
|
||||
engine.pump();
|
||||
// Poll every root port and reconcile — a device present but not yet
|
||||
// enumerated is brought up; a device gone is torn down. This does NOT
|
||||
// depend on a Port Status Change EVENT firing: the boot scan runs ~3 ms
|
||||
// after the controller reset, far too early for a USB2 connection to
|
||||
// debounce (~100 ms), and the SuperSpeed devices that DO show up early
|
||||
// proved the event path unreliable for the late USB2 companion hub on
|
||||
// real hardware. Polling catches it on the next tick regardless.
|
||||
if (manager_handle) |manager| {
|
||||
var port: u32 = 1;
|
||||
while (port <= engine.max_ports and port <= prev_connected.len) : (port += 1) {
|
||||
const connected = engine.portConnected(port);
|
||||
const was = prev_connected[port];
|
||||
prev_connected[port] = connected;
|
||||
if (connected and !was and engine.deviceOnPort(port) == null) {
|
||||
// Rising edge the boot scan missed (it ran before the USB2
|
||||
// connection debounced): bring the device up now.
|
||||
std.log.info("root port {d}: device appeared (PORTSC 0x{x:0>8})", .{ port, engine.portStatus(port) });
|
||||
bringUpPort(manager, engine, port);
|
||||
} else if (!connected and was and engine.deviceOnPort(port) != null) {
|
||||
tearDownPort(manager, engine, port);
|
||||
}
|
||||
}
|
||||
}
|
||||
while (engine.takePortChange()) |port| {
|
||||
const manager = manager_handle orelse break;
|
||||
const connected = engine.portConnected(port);
|
||||
std.log.info("root port {d} change: {s} (PORTSC 0x{x:0>8})", .{ port, if (connected) "connected" else "empty", engine.portStatus(port) });
|
||||
if (connected) {
|
||||
if (engine.deviceOnPort(port) == null) bringUpPort(manager, engine, port);
|
||||
} else {
|
||||
tearDownPort(manager, engine, port);
|
||||
}
|
||||
}
|
||||
// Downstream hub-port changes (docs/usb-hub.md): a device connected on a
|
||||
// hub's downstream port is enumerated and registered here, so a keyboard
|
||||
// behind a hub reaches its class driver like one on a root port.
|
||||
// Cap per tick: even if a hub's change bits refuse to clear, the driver
|
||||
// must not spin here — it services a bounded batch and yields to the
|
||||
// next tick (and to storage, input, everything else).
|
||||
var serviced: u32 = 0;
|
||||
while (engine.takeHubChange()) |change| {
|
||||
const manager = manager_handle orelse break;
|
||||
bringUpBehindHub(manager, engine, change.hub, change.port);
|
||||
serviced += 1;
|
||||
if (serviced >= 32) break;
|
||||
}
|
||||
while (engine.takeReport()) |report| {
|
||||
var message = transfer.InterruptReport{
|
||||
.device_token = report.device_token,
|
||||
@@ -407,7 +581,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed controller device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(transfer.message_maximum, .{
|
||||
@@ -417,8 +591,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -22,6 +22,7 @@ const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const mmio = @import("mmio");
|
||||
const usb_abi = @import("usb-abi");
|
||||
const usb_ids = @import("usb-ids");
|
||||
const dma = runtime.dma;
|
||||
const system = runtime.system;
|
||||
|
||||
@@ -85,6 +86,7 @@ pub const TrbType = enum(u6) {
|
||||
status_stage = 4,
|
||||
link = 6,
|
||||
enable_slot = 9,
|
||||
disable_slot = 10,
|
||||
address_device = 11,
|
||||
configure_endpoint = 12,
|
||||
evaluate_context = 13,
|
||||
@@ -99,6 +101,7 @@ pub const TrbType = enum(u6) {
|
||||
pub const CompletionCode = enum(u8) {
|
||||
invalid = 0,
|
||||
success = 1,
|
||||
usb_transaction_error = 4,
|
||||
short_packet = 13,
|
||||
_,
|
||||
};
|
||||
@@ -279,6 +282,21 @@ pub const Device = struct {
|
||||
// Transfer rings configured for this device's interrupt/bulk endpoints.
|
||||
endpoint_ring_count: u8 = 0,
|
||||
endpoint_rings: [max_configured_endpoints]ConfiguredEndpoint = [_]ConfiguredEndpoint{.{}} ** max_configured_endpoints,
|
||||
|
||||
// Hub topology (docs/usb-hub.md). A device behind a hub is addressed with a
|
||||
// route string; these carry the fields buildAddressInputContext needs.
|
||||
route: u32 = 0, // Slot Context route string (5 tiers x 4 bits); 0 = on a root port
|
||||
root_port: u32 = 0, // the ROOT-hub port the chain hangs off (inherited down a chain)
|
||||
parent_slot: u8 = 0, // the parent hub's slot id (0 = on a root port) — the TT hub
|
||||
parent_port: u8 = 0, // the parent hub's downstream port this device sits on
|
||||
|
||||
// Set when this device IS a hub, after setupHub configures it.
|
||||
is_hub: bool = false,
|
||||
hub_ports: u8 = 0, // downstream port count from the hub descriptor
|
||||
hub_multi_tt: bool = false,
|
||||
// Downstream ports with a pending change to service (bit P = port P), set
|
||||
// by the status-change endpoint (and by an initial sweep in setupHub).
|
||||
hub_change_mask: u32 = 0,
|
||||
};
|
||||
|
||||
// A standing interrupt-IN subscription: the endpoint's ring is kept armed with a
|
||||
@@ -297,6 +315,9 @@ const Subscription = struct {
|
||||
// device token and the endpoint handle its reports are sent to.
|
||||
device_token: u64 = 0,
|
||||
report_endpoint: usize = 0,
|
||||
// When set, this is an IN-PROCESS hub status-change subscription: completions
|
||||
// set the hub's pending-change mask instead of queuing a class-driver report.
|
||||
hub: ?*Device = null,
|
||||
};
|
||||
|
||||
// One interrupt report waiting for the bus layer to push it to a subscriber.
|
||||
@@ -314,6 +335,109 @@ const max_devices = 8;
|
||||
const max_subscriptions = 8;
|
||||
const report_queue_capacity = 16;
|
||||
|
||||
// --- USB hub class requests + constants (docs/usb-hub.md) ------------------
|
||||
//
|
||||
// A hub is bus infrastructure the CONTROLLER driver handles in-process: the
|
||||
// route strings and slot contexts a downstream device needs only exist here.
|
||||
// These are the class-specific control requests to a hub device.
|
||||
|
||||
// A USB2 hub port's wPortStatus speed bits (bit 9 = low-speed, bit 10 =
|
||||
// high-speed; neither = full-speed) mapped to the xHCI speed id.
|
||||
fn mapHubPortSpeed(port_speed_bits: u32) u32 {
|
||||
if (port_speed_bits & 0x1 != 0) return 2; // low-speed (wPortStatus bit 9)
|
||||
if (port_speed_bits & 0x2 != 0) return 3; // high-speed
|
||||
return 1; // full-speed
|
||||
}
|
||||
|
||||
fn routeDepth(route: u32) u16 {
|
||||
// Tiers used by a route string: each nonzero 4-bit nibble is one tier.
|
||||
var depth: u16 = 0;
|
||||
var r = route;
|
||||
while (r != 0) : (r >>= 4) {
|
||||
if (r & 0xF != 0) depth += 1;
|
||||
}
|
||||
return depth;
|
||||
}
|
||||
|
||||
const hubreq = struct {
|
||||
// Hub descriptor types (GET_DESCRIPTOR value high byte).
|
||||
const descriptor_usb2: u8 = 0x29;
|
||||
const descriptor_usb3: u8 = 0x2A;
|
||||
|
||||
// Hub/port feature selectors (SET_FEATURE / CLEAR_FEATURE value).
|
||||
const feature_port_reset: u16 = 4;
|
||||
const feature_port_power: u16 = 8;
|
||||
const feature_c_port_connection: u16 = 16;
|
||||
const feature_c_port_reset: u16 = 20;
|
||||
const feature_c_port_link_state: u16 = 25; // SS
|
||||
const feature_bh_port_reset: u16 = 28; // SS
|
||||
const feature_c_bh_port_reset: u16 = 29; // SS
|
||||
|
||||
// Port status (wPortStatus, first 16 bits of the 4-byte GET_STATUS result).
|
||||
const status_connection: u16 = 1 << 0;
|
||||
const status_enable: u16 = 1 << 1;
|
||||
const status_reset: u16 = 1 << 4;
|
||||
// Port-status change bits (wPortChange, the high 16 bits).
|
||||
const change_connection: u16 = 1 << 0;
|
||||
const change_reset: u16 = 1 << 4;
|
||||
|
||||
// wHubCharacteristics bit 7: multiple transaction translators.
|
||||
const characteristics_multi_tt: u16 = 1 << 7;
|
||||
|
||||
// SET_HUB_DEPTH (SuperSpeed hubs, so they can compose route strings).
|
||||
const request_set_hub_depth: u8 = 12;
|
||||
|
||||
fn getDescriptor(kind: u8, length: u16) usb_abi.Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .device, .kind = .class, .direction = .device_to_host },
|
||||
.request_code = .get_descriptor,
|
||||
.value = @as(u16, kind) << 8,
|
||||
.index = 0,
|
||||
.length = length,
|
||||
};
|
||||
}
|
||||
|
||||
fn setPortFeature(feature: u16, port: u16) usb_abi.Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .other, .kind = .class, .direction = .host_to_device },
|
||||
.request_code = .set_feature,
|
||||
.value = feature,
|
||||
.index = port,
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
fn clearPortFeature(feature: u16, port: u16) usb_abi.Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .other, .kind = .class, .direction = .host_to_device },
|
||||
.request_code = .clear_feature,
|
||||
.value = feature,
|
||||
.index = port,
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
fn getPortStatus(port: u16) usb_abi.Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .other, .kind = .class, .direction = .device_to_host },
|
||||
.request_code = .get_status,
|
||||
.value = 0,
|
||||
.index = port,
|
||||
.length = 4,
|
||||
};
|
||||
}
|
||||
|
||||
fn setHubDepth(depth: u16) usb_abi.Request {
|
||||
return .{
|
||||
.request_type = .{ .recipient = .device, .kind = .class, .direction = .host_to_device },
|
||||
.request_code = @enumFromInt(request_set_hub_depth),
|
||||
.value = depth,
|
||||
.index = 0,
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
pub const Controller = struct {
|
||||
register_base: usize,
|
||||
op_base: usize,
|
||||
@@ -330,6 +454,8 @@ pub const Controller = struct {
|
||||
subscriptions: [max_subscriptions]Subscription = [_]Subscription{.{}} ** max_subscriptions,
|
||||
report_queue: [report_queue_capacity]Report = [_]Report{.{}} ** report_queue_capacity,
|
||||
report_count: usize = 0,
|
||||
port_changes: [16]u32 = undefined,
|
||||
port_change_count: usize = 0,
|
||||
// Transferred length of the most recent awaited transfer (requested minus the
|
||||
// event residual); read right after a control or bulk transfer returns true.
|
||||
last_transfer_length: u32 = 0,
|
||||
@@ -360,6 +486,73 @@ pub const Controller = struct {
|
||||
}
|
||||
|
||||
// PORTSC for 1-based port `port`.
|
||||
/// Diagnostic: walk the xECP list and log each Supported Protocol capability
|
||||
/// (USB 2.0 vs 3.x, the compatible root-port range), then dump every port's
|
||||
/// raw PORTSC. Reveals where the USB2 root ports are and their state — for
|
||||
/// finding a USB2 companion hub that isn't presenting a connection.
|
||||
pub fn dumpPortTopology(self: *const Controller) void {
|
||||
const hccparams1 = read32(self.register_base + cap_hccparams1);
|
||||
var offset: usize = (hccparams1 >> 16) & 0xFFFF; // xECP: dword offset from register_base
|
||||
var guard: u32 = 0;
|
||||
while (offset != 0 and guard < 64) : (guard += 1) {
|
||||
const cap_base = self.register_base + offset * 4;
|
||||
const dw0 = read32(cap_base);
|
||||
const id = dw0 & 0xFF;
|
||||
if (id == 2) { // Supported Protocol
|
||||
const dw2 = read32(cap_base + 8);
|
||||
const major = (dw0 >> 24) & 0xFF;
|
||||
const minor = (dw0 >> 16) & 0xFF;
|
||||
const port_offset = dw2 & 0xFF;
|
||||
const port_count = (dw2 >> 8) & 0xFF;
|
||||
std.log.info("xECP USB {d}.{d}: root ports {d}..{d}", .{ major, minor, port_offset, port_offset + port_count - 1 });
|
||||
}
|
||||
const next = (dw0 >> 8) & 0xFF;
|
||||
if (next == 0) break;
|
||||
offset += next;
|
||||
}
|
||||
var port: u32 = 1;
|
||||
while (port <= self.max_ports) : (port += 1) {
|
||||
const portsc = self.portStatus(port);
|
||||
std.log.info("PORTSC[{d}] 0x{x:0>8}: {s}, {s}, link={s}, power={s}, {s}", .{
|
||||
port,
|
||||
portsc,
|
||||
if (portsc & 1 != 0) "connected" else "empty",
|
||||
if (portsc & 2 != 0) "enabled" else "disabled",
|
||||
usb_ids.linkStateName((portsc >> 5) & 0xF),
|
||||
if (portsc & (1 << 9) != 0) "on" else "off",
|
||||
usb_ids.speedName((portsc >> 10) & 0xF),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Read USB STRING descriptor `index` (English, langid 0x0409) into `out` as
|
||||
/// ASCII, returning the slice — for logging manufacturer/product names.
|
||||
/// Null for index 0 (no string) or a failed transfer. Non-ASCII code units
|
||||
/// become '?'.
|
||||
pub fn readString(self: *Controller, device: *Device, index: u8, out: []u8) ?[]const u8 {
|
||||
if (index == 0) return null;
|
||||
var raw: [256]u8 = undefined;
|
||||
const request = usb_abi.Request{
|
||||
.request_type = .{ .recipient = .device, .kind = .standard, .direction = .device_to_host },
|
||||
.request_code = .get_descriptor,
|
||||
.value = (@as(u16, 3) << 8) | index, // STRING descriptor
|
||||
.index = 0x0409, // English (US)
|
||||
.length = raw.len,
|
||||
};
|
||||
if (!self.controlTransfer(device, request, raw[0..], true)) return null;
|
||||
const length = raw[0]; // bLength; the UTF-16LE payload is bytes 2..length
|
||||
if (length < 2) return null;
|
||||
const chars = (@min(length, raw.len) - 2) / 2;
|
||||
var n: usize = 0;
|
||||
var i: usize = 0;
|
||||
while (i < chars and n < out.len) : (i += 1) {
|
||||
const unit = @as(u16, raw[2 + i * 2]) | (@as(u16, raw[2 + i * 2 + 1]) << 8);
|
||||
out[n] = if (unit >= 0x20 and unit < 0x7F) @intCast(unit) else '?';
|
||||
n += 1;
|
||||
}
|
||||
return out[0..n];
|
||||
}
|
||||
|
||||
pub fn portStatus(self: *const Controller, port: u32) u32 {
|
||||
return read32(self.op_base + op_portsc_base + op_portsc_stride * (port - 1));
|
||||
}
|
||||
@@ -429,11 +622,30 @@ pub const Controller = struct {
|
||||
write64(self.interrupter(event_ring_dequeue_pointer), self.event_ring.segment.physical);
|
||||
write64(self.interrupter(event_ring_segment_table_base), self.event_ring.table.physical);
|
||||
write32(self.interrupter(interrupter_moderation), 0);
|
||||
|
||||
// Run. (Interrupts are left disabled — the event ring is polled.)
|
||||
// Enable the interrupter (IMAN.IE) and USBCMD.INTE. We still POLL the
|
||||
// event ring — no interrupt is wired — but some controllers (QEMU's
|
||||
// qemu-xhci among them) only WRITE runtime events to the ring when the
|
||||
// interrupter is enabled, so a hot-plug port-change event is silently
|
||||
// dropped otherwise. Enabling it is harmless to a polling driver.
|
||||
write32(self.interrupter(interrupter_management), 1 << 1); // IE
|
||||
mmio.wmb();
|
||||
write32(self.operational(op_usbcmd), read32(self.operational(op_usbcmd)) | usbcmd_run);
|
||||
|
||||
// Run.
|
||||
mmio.wmb();
|
||||
write32(self.operational(op_usbcmd), read32(self.operational(op_usbcmd)) | usbcmd_run | usbcmd_interrupter_enable);
|
||||
if (!waitClear(self.operational(op_usbsts), usbsts_halted)) return null;
|
||||
|
||||
// Power EVERY port — including empty ones — so a later hot-plug can
|
||||
// signal a connect (an unpowered port reports nothing: PP=0 is why a
|
||||
// device added after boot never raised a port-change event). Boot-time
|
||||
// devices are on already-powered ports; this just extends power to the
|
||||
// rest. Write PP without disturbing the write-1-to-clear bits.
|
||||
var port: u32 = 1;
|
||||
while (port <= self.max_ports) : (port += 1) {
|
||||
const status = self.portStatus(port);
|
||||
if (status & portsc_power == 0)
|
||||
self.writePortStatus(port, (status & ~portsc_write_1_to_clear) | portsc_power);
|
||||
}
|
||||
return self;
|
||||
}
|
||||
|
||||
@@ -601,10 +813,19 @@ pub const Controller = struct {
|
||||
const base = device.input_context.virtual;
|
||||
// Input Control Context (index 0): Add flags in dword 1 = A0 | A1.
|
||||
contextDword(base, 0, 1, cs).* = 0b11;
|
||||
// Slot Context (index 1): speed[23:20], Context Entries[31:27] = 1.
|
||||
contextDword(base, 1, 0, cs).* = (device.speed << 20) | (@as(u32, 1) << 27);
|
||||
// Root Hub Port Number[23:16].
|
||||
contextDword(base, 1, 1, cs).* = device.port << 16;
|
||||
// Slot Context dword 0: Route String[19:0], Speed[23:20], Context
|
||||
// Entries[31:27] = 1.
|
||||
contextDword(base, 1, 0, cs).* = (device.route & 0xFFFFF) | (device.speed << 20) | (@as(u32, 1) << 27);
|
||||
// Slot Context dword 1: Root Hub Port Number[23:16] — the ROOT port the
|
||||
// hub chain hangs off (inherited down a chain), not the device's own
|
||||
// downstream hub port.
|
||||
contextDword(base, 1, 1, cs).* = device.root_port << 16;
|
||||
// Slot Context dword 2: the transaction translator — a full/low-speed
|
||||
// device behind a high-speed hub routes split transactions through the
|
||||
// parent hub's TT. Parent Hub Slot ID[7:0], Parent Port Number[13:8].
|
||||
if (device.parent_slot != 0 and device.speed < 3) {
|
||||
contextDword(base, 1, 2, cs).* = @as(u32, device.parent_slot) | (@as(u32, device.parent_port) << 8);
|
||||
}
|
||||
// EP0 Context (index 2): CErr[2:1]=3, EP Type[5:3]=Control(4), MPS[31:16].
|
||||
contextDword(base, 2, 1, cs).* = (@as(u32, 3) << 1) | (@as(u32, 4) << 3) | (device.max_packet_size_0 << 16);
|
||||
// TR Dequeue Pointer (dwords 2:3) with Dequeue Cycle State = 1.
|
||||
@@ -616,12 +837,31 @@ pub const Controller = struct {
|
||||
}
|
||||
|
||||
fn addressDeviceCommand(self: *Controller, device: *Device) bool {
|
||||
const physical = self.submitCommand(.{
|
||||
.parameter = device.input_context.physical,
|
||||
.control = trbControl(.address_device, @as(u32, device.slot_id) << 24),
|
||||
});
|
||||
const code = self.awaitCommand(physical) orelse return false;
|
||||
return code == @intFromEnum(CompletionCode.success);
|
||||
// Retry on a USB Transaction Error (code 4): a freshly-reset device can
|
||||
// miss the first SET_ADDRESS; re-reset the port and try again (xHCI
|
||||
// 4.6.5). Up to 3 attempts.
|
||||
var attempt: u32 = 0;
|
||||
while (attempt < 3) : (attempt += 1) {
|
||||
const physical = self.submitCommand(.{
|
||||
.parameter = device.input_context.physical,
|
||||
.control = trbControl(.address_device, @as(u32, device.slot_id) << 24),
|
||||
});
|
||||
const code = self.awaitCommand(physical) orelse {
|
||||
std.log.info("port {d} setup: Address Device timed out (attempt {d})", .{ device.port, attempt + 1 });
|
||||
return false;
|
||||
};
|
||||
if (code == @intFromEnum(CompletionCode.success)) return true;
|
||||
std.log.info("port {d} setup: Address Device completion code {d} (attempt {d})", .{ device.port, code, attempt + 1 });
|
||||
if (code != @intFromEnum(CompletionCode.usb_transaction_error)) return false;
|
||||
// Re-reset a root-port device and wait the recovery interval before
|
||||
// retrying. (A device behind a hub is reset through the hub — not
|
||||
// retried here; its port was reset in serviceHubPort.)
|
||||
if (device.parent_slot == 0) {
|
||||
if (!self.resetPort(device.port)) return false;
|
||||
system.sleep(10);
|
||||
} else return false;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Reset the port, enable a slot, and address the device on it: after this the
|
||||
@@ -629,15 +869,45 @@ pub const Controller = struct {
|
||||
/// or null on any failure. The EP0 MPS is taken from the speed default and
|
||||
/// corrected from the device descriptor by `refreshMaxPacketSize0` if needed.
|
||||
pub fn setupDevice(self: *Controller, port: u32, speed: u32) ?*Device {
|
||||
if (!self.resetPort(port)) return null;
|
||||
const slot_id = self.enableSlot() orelse return null;
|
||||
const device = self.allocateDevice() orelse return null;
|
||||
// A SuperSpeed port that has trained its link is ALREADY enabled — the
|
||||
// xHCI advances USB3 ports to Enabled with no reset (spec 4.3). Driving
|
||||
// a hot reset into a live SS link drops PED mid-reset on real silicon
|
||||
// (observed: "setup failed" in the same millisecond as "connected").
|
||||
// Only a not-yet-enabled port — every USB2 device, or a stuck SS link —
|
||||
// needs the reset to enable.
|
||||
const already_enabled = speed >= 4 and self.portStatus(port) & portsc_enabled != 0;
|
||||
var effective_speed = speed;
|
||||
if (!already_enabled) {
|
||||
if (!self.resetPort(port)) {
|
||||
std.log.info("port {d} setup: port reset failed (PORTSC 0x{x:0>8})", .{ port, self.portStatus(port) });
|
||||
return null;
|
||||
}
|
||||
// A USB2 port's PORTSC speed field is only meaningful once the port
|
||||
// is enabled by the reset — sample it NOW, not at connect time
|
||||
// (pre-reset reads misreport on real controllers; M20).
|
||||
effective_speed = (self.portStatus(port) >> 10) & 0xF;
|
||||
if (effective_speed == 0) effective_speed = speed; // defensive: keep the caller's read
|
||||
// USB 2.0 spec 7.1.7.5: a device needs a reset-recovery interval
|
||||
// (TRSTRCY, 10 ms) after reset before it answers SET_ADDRESS.
|
||||
// Addressing immediately gives a USB Transaction Error (code 4) on
|
||||
// real full-speed devices; QEMU tolerates the omission.
|
||||
system.sleep(10);
|
||||
}
|
||||
const slot_id = self.enableSlot() orelse {
|
||||
std.log.info("port {d} setup: Enable Slot failed", .{port});
|
||||
return null;
|
||||
};
|
||||
const device = self.allocateDevice() orelse {
|
||||
std.log.info("port {d} setup: no free device slot", .{port});
|
||||
return null;
|
||||
};
|
||||
device.* = .{
|
||||
.used = true,
|
||||
.slot_id = slot_id,
|
||||
.port = port,
|
||||
.speed = speed,
|
||||
.max_packet_size_0 = defaultMaxPacketSize0(speed),
|
||||
.speed = effective_speed,
|
||||
.max_packet_size_0 = defaultMaxPacketSize0(effective_speed),
|
||||
.root_port = port, // a root-port device: the chain root IS this port
|
||||
};
|
||||
device.input_context = dma.alloc(page_size, dma.coherent) orelse return self.abandon(device);
|
||||
device.device_context = dma.alloc(page_size, dma.coherent) orelse return self.abandon(device);
|
||||
@@ -652,22 +922,262 @@ pub const Controller = struct {
|
||||
return device;
|
||||
}
|
||||
|
||||
|
||||
/// Configure an enumerated class-9 device as a hub (docs/usb-hub.md): read
|
||||
/// the hub descriptor for the downstream port count, tell the controller the
|
||||
/// slot is a hub (so it routes downstream traffic), SET_HUB_DEPTH for a
|
||||
/// SuperSpeed hub, and power every downstream port. Downstream enumeration
|
||||
/// (status-change handling) is B4b. Returns false on a control-transfer
|
||||
/// failure; the hub is still registered, just inert.
|
||||
pub fn setupHub(self: *Controller, device: *Device) bool {
|
||||
const is_usb3 = device.speed >= 4;
|
||||
var descriptor: [16]u8 = undefined;
|
||||
const kind: u8 = if (is_usb3) hubreq.descriptor_usb3 else hubreq.descriptor_usb2;
|
||||
if (!self.controlTransfer(device, hubreq.getDescriptor(kind, descriptor.len), descriptor[0..], true)) {
|
||||
std.log.info("hub slot {d}: hub descriptor read failed", .{device.slot_id});
|
||||
return false;
|
||||
}
|
||||
device.is_hub = true;
|
||||
device.hub_ports = descriptor[2]; // bNbrPorts
|
||||
const characteristics = @as(u16, descriptor[3]) | (@as(u16, descriptor[4]) << 8);
|
||||
device.hub_multi_tt = !is_usb3 and (characteristics & hubreq.characteristics_multi_tt != 0);
|
||||
|
||||
// A SuperSpeed hub needs its depth (tiers from the root) to compose the
|
||||
// route strings of devices below it.
|
||||
if (is_usb3) {
|
||||
const depth = routeDepth(device.route);
|
||||
_ = self.controlTransfer(device, hubreq.setHubDepth(depth), &.{}, false);
|
||||
}
|
||||
|
||||
// Tell the controller the slot is a hub — Hub bit, Number of Ports, and
|
||||
// (for a USB2 multi-TT hub) MTT + TT Think Time. Configure Endpoint with
|
||||
// only the slot add-flag (A0) evaluates these (xHCI 4.6.6).
|
||||
if (!self.configureSlotAsHub(device)) {
|
||||
std.log.info("hub slot {d}: could not configure slot as a hub", .{device.slot_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Power every downstream port.
|
||||
var port: u16 = 1;
|
||||
while (port <= device.hub_ports) : (port += 1) {
|
||||
_ = self.controlTransfer(device, hubreq.setPortFeature(hubreq.feature_port_power, port), &.{}, false);
|
||||
}
|
||||
// Seed every downstream port as pending: the bus tick GET_STATUSes each
|
||||
// and enumerates the connected ones. This makes a STATIC topology (a
|
||||
// device present at power-on) work without relying on the initial
|
||||
// status-change interrupt edge; the interrupt then handles later plugs.
|
||||
device.hub_change_mask = if (device.hub_ports >= 31) 0xFFFF_FFFE else (@as(u32, 1) << @intCast(device.hub_ports + 1)) - 2;
|
||||
|
||||
// Arm the status-change interrupt endpoint (in-process) for hot-plug.
|
||||
self.armHubStatus(device);
|
||||
|
||||
std.log.info("hub slot {d}: {d} downstream ports powered ({s})", .{
|
||||
device.slot_id,
|
||||
device.hub_ports,
|
||||
if (is_usb3) "SuperSpeed" else if (device.hub_multi_tt) "USB2 multi-TT" else "USB2 single-TT",
|
||||
});
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Arm the hub's interrupt-IN status-change endpoint with an in-process
|
||||
/// subscription: completions set the hub's pending-change mask (serviced on
|
||||
/// the bus tick). Best-effort — a hub with no interrupt endpoint (shouldn't
|
||||
/// happen) just relies on the initial sweep.
|
||||
fn armHubStatus(self: *Controller, device: *Device) void {
|
||||
for (device.interfaces[0..device.interface_count]) |interface| {
|
||||
for (interface.endpoints[0..interface.endpoint_count]) |endpoint| {
|
||||
const is_interrupt = endpoint.transfer_type == 3;
|
||||
const is_in = endpoint.address & 0x80 != 0;
|
||||
if (!is_interrupt or !is_in) continue;
|
||||
const ring = self.getOrConfigureEndpoint(device, endpoint) orelse return;
|
||||
const subscription = self.allocateSubscription() orelse return;
|
||||
const buffer = dma.alloc(page_size, dma.coherent) orelse return;
|
||||
const number: u8 = endpoint.address & 0x0F;
|
||||
subscription.* = .{
|
||||
.active = true,
|
||||
.slot_id = device.slot_id,
|
||||
.dci = doorbellContextIndex(number, true),
|
||||
.endpoint_address = endpoint.address,
|
||||
.ring = ring,
|
||||
.buffer = buffer,
|
||||
.max_length = endpoint.max_packet_size,
|
||||
.hub = device,
|
||||
};
|
||||
self.armInterrupt(subscription);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The next pending (hub, downstream-port) change to service, or null. Clears
|
||||
/// the returned port's bit. Called on the bus tick.
|
||||
pub fn takeHubChange(self: *Controller) ?struct { hub: *Device, port: u16 } {
|
||||
for (&self.devices) |*device| {
|
||||
if (!device.used or !device.is_hub or device.hub_change_mask == 0) continue;
|
||||
const bit: u5 = @intCast(@ctz(device.hub_change_mask));
|
||||
device.hub_change_mask &= ~(@as(u32, 1) << bit);
|
||||
if (bit == 0) continue; // bit 0 is the hub itself, not a downstream port
|
||||
return .{ .hub = device, .port = bit };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn readHubPortStatus(self: *Controller, hub: *Device, port: u16) ?u32 {
|
||||
var buffer: [4]u8 = undefined;
|
||||
if (!self.controlTransfer(hub, hubreq.getPortStatus(port), buffer[0..], true)) return null;
|
||||
return @as(u32, buffer[0]) | (@as(u32, buffer[1]) << 8) | (@as(u32, buffer[2]) << 16) | (@as(u32, buffer[3]) << 24);
|
||||
}
|
||||
|
||||
/// Bring up (or note the disconnect of) a device on hub downstream `port`:
|
||||
/// read the port status, acknowledge the change bits, and on a fresh connect
|
||||
/// reset the port, read the speed, and setup+address the downstream device
|
||||
/// (route string + TT). Returns the addressed device for the bus to enumerate
|
||||
/// and register, or null (empty port, disconnect, or a failure).
|
||||
/// Read a downstream hub port's status and acknowledge its latched change
|
||||
/// bits (so it can signal again). Returns the wPortStatus word.
|
||||
pub fn hubPortStatusAck(self: *Controller, hub: *Device, port: u16) ?u32 {
|
||||
const status = self.readHubPortStatus(hub, port) orelse return null;
|
||||
// Acknowledge every latched change bit. A SuperSpeed hub has extra ones
|
||||
// (link-state, BH-reset) beyond a USB2 hub's connection/reset — leaving
|
||||
// any set makes the hub's status-change endpoint re-report the same port
|
||||
// forever, spinning the driver (a real SuperSpeed hub hung boot here;
|
||||
// QEMU's USB2 hub has none of these). Clearing an inapplicable feature
|
||||
// is harmless (the hub STALLs it and we move on).
|
||||
_ = self.controlTransfer(hub, hubreq.clearPortFeature(hubreq.feature_c_port_connection, port), &.{}, false);
|
||||
_ = self.controlTransfer(hub, hubreq.clearPortFeature(hubreq.feature_c_port_reset, port), &.{}, false);
|
||||
if (hub.speed >= 4) {
|
||||
_ = self.controlTransfer(hub, hubreq.clearPortFeature(hubreq.feature_c_port_link_state, port), &.{}, false);
|
||||
_ = self.controlTransfer(hub, hubreq.clearPortFeature(hubreq.feature_c_bh_port_reset, port), &.{}, false);
|
||||
}
|
||||
return status;
|
||||
}
|
||||
|
||||
pub fn hubPortConnected(status: u32) bool {
|
||||
return status & hubreq.status_connection != 0;
|
||||
}
|
||||
|
||||
/// Reset + address a device on a connected, empty downstream hub `port` (the
|
||||
/// bus has confirmed connect and no existing device): reset the port, read
|
||||
/// the speed, and setup+address the downstream device (route string + TT).
|
||||
/// Returns the addressed device for the bus to enumerate + register.
|
||||
pub fn serviceHubPort(self: *Controller, hub: *Device, port: u16) ?*Device {
|
||||
const status = self.readHubPortStatus(hub, port) orelse return null;
|
||||
|
||||
// Reset the port if not yet enabled, then wait (bounded) for enable.
|
||||
if (status & hubreq.status_enable == 0) {
|
||||
_ = self.controlTransfer(hub, hubreq.setPortFeature(hubreq.feature_port_reset, port), &.{}, false);
|
||||
var tries: u32 = 0;
|
||||
while (tries < 200) : (tries += 1) {
|
||||
system.sleep(5);
|
||||
const s = self.readHubPortStatus(hub, port) orelse return null;
|
||||
if (s & hubreq.status_enable != 0) break;
|
||||
}
|
||||
_ = self.controlTransfer(hub, hubreq.clearPortFeature(hubreq.feature_c_port_reset, port), &.{}, false);
|
||||
}
|
||||
const enabled = self.readHubPortStatus(hub, port) orelse return null;
|
||||
if (enabled & hubreq.status_enable == 0) {
|
||||
std.log.info("hub slot {d} port {d}: reset did not enable", .{ hub.slot_id, port });
|
||||
return null;
|
||||
}
|
||||
const downstream_speed = (enabled >> 9) & 0x3; // wPortStatus: bit9 low-speed, bit10 high-speed
|
||||
|
||||
return self.setupDeviceBehindHub(hub, port, downstream_speed);
|
||||
}
|
||||
|
||||
pub fn deviceOnHubPort(self: *Controller, hub: *Device, port: u16) ?*Device {
|
||||
for (&self.devices) |*device| {
|
||||
if (device.used and device.parent_slot == hub.slot_id and device.parent_port == port) return device;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Enable a slot and Address a device behind `hub` on downstream `port`, with
|
||||
/// the composed route string, inherited root port, and TT fields (so the
|
||||
/// controller routes split transactions through this hub's TT for a
|
||||
/// full/low-speed device). Mirrors setupDevice for a root-port device.
|
||||
fn setupDeviceBehindHub(self: *Controller, hub: *Device, port: u16, speed: u32) ?*Device {
|
||||
const slot_id = self.enableSlot() orelse {
|
||||
std.log.info("hub slot {d} port {d}: Enable Slot failed", .{ hub.slot_id, port });
|
||||
return null;
|
||||
};
|
||||
const device = self.allocateDevice() orelse return null;
|
||||
const child_speed = mapHubPortSpeed(speed);
|
||||
device.* = .{
|
||||
.used = true,
|
||||
.slot_id = slot_id,
|
||||
.port = hub.root_port,
|
||||
.speed = child_speed,
|
||||
.max_packet_size_0 = defaultMaxPacketSize0(child_speed),
|
||||
.route = (hub.route << 4) | (port & 0xF),
|
||||
.root_port = hub.root_port,
|
||||
.parent_slot = hub.slot_id,
|
||||
.parent_port = @intCast(port),
|
||||
};
|
||||
device.input_context = dma.alloc(page_size, dma.coherent) orelse return self.abandon(device);
|
||||
device.device_context = dma.alloc(page_size, dma.coherent) orelse return self.abandon(device);
|
||||
device.ep0_ring = .{ .region = dma.alloc(page_size, dma.coherent) orelse return self.abandon(device) };
|
||||
device.ep0_ring.installLink();
|
||||
device.control_buffer = dma.alloc(page_size, dma.coherent) orelse return self.abandon(device);
|
||||
|
||||
self.buildAddressInputContext(device);
|
||||
const array: [*]volatile u64 = @ptrFromInt(self.device_context_array.virtual);
|
||||
array[device.slot_id] = device.device_context.physical;
|
||||
if (!self.addressDeviceCommand(device)) return self.abandon(device);
|
||||
return device;
|
||||
}
|
||||
|
||||
/// Configure Endpoint with only A0 (slot) set: rebuild the slot context with
|
||||
/// the Hub bit, Number of Ports, and MTT/TT-Think-Time, so the controller
|
||||
/// treats this slot as a hub.
|
||||
fn configureSlotAsHub(self: *Controller, device: *Device) bool {
|
||||
const cs = self.context_size;
|
||||
const base = device.input_context.virtual;
|
||||
@memset(@as([*]u8, @ptrFromInt(base))[0 .. 2 * cs], 0);
|
||||
contextDword(base, 0, 1, cs).* = 0b1; // Input Control Context add flags: A0 (slot)
|
||||
// Slot Context dword 0: route, speed, context entries, plus Hub[26] and
|
||||
// (USB2 multi-TT) MTT[25].
|
||||
var dword0: u32 = (device.route & 0xFFFFF) | (device.speed << 20) | (@as(u32, 1) << 27) | (@as(u32, 1) << 26);
|
||||
if (device.hub_multi_tt) dword0 |= (@as(u32, 1) << 25);
|
||||
contextDword(base, 1, 0, cs).* = dword0;
|
||||
// Slot Context dword 1: Root Hub Port Number[23:16], Number of Ports[31:24].
|
||||
contextDword(base, 1, 1, cs).* = (device.root_port << 16) | (@as(u32, device.hub_ports) << 24);
|
||||
// Slot Context dword 2: TT Think Time[17:16] = 0 (8 FS bit times); the
|
||||
// parent-TT fields (if this hub is itself behind a hub) carry over.
|
||||
if (device.parent_slot != 0 and device.speed < 3) {
|
||||
contextDword(base, 1, 2, cs).* = @as(u32, device.parent_slot) | (@as(u32, device.parent_port) << 8);
|
||||
}
|
||||
return self.configureEndpointCommand(device);
|
||||
}
|
||||
|
||||
fn abandon(self: *Controller, device: *Device) ?*Device {
|
||||
_ = self;
|
||||
device.used = false;
|
||||
return null;
|
||||
}
|
||||
|
||||
fn awaitTransfer(self: *Controller, requested_length: u32) ?u8 {
|
||||
/// Await the completion of OUR transfer — identified by the event's slot id
|
||||
/// (control[31:24]) and endpoint DCI (control[20:16]). Any other transfer
|
||||
/// event is either a subscription's report (serviced) or foreign noise (an
|
||||
/// interrupt endpoint's error/stale completion whose TRB pointer no longer
|
||||
/// matches the armed one) — DROPPED, never misattributed: claiming a foreign
|
||||
/// event as our completion desynchronized the mass-storage bulk protocol in
|
||||
/// a way that survived every driver restart (the 1-in-3 READ CAPACITY
|
||||
/// failure at boot, with a USB keyboard and mouse polling concurrently).
|
||||
fn awaitTransfer(self: *Controller, slot_id: u8, dci: u32, requested_length: u32) ?u8 {
|
||||
const deadline = system.clock() + 1_000_000_000;
|
||||
while (true) {
|
||||
const event = self.nextEvent(deadline) orelse return null;
|
||||
if (trbType(event.control) == @intFromEnum(TrbType.transfer_event)) {
|
||||
if (self.serviceInterruptEvent(event)) continue; // a subscription's report
|
||||
const residual = event.status & 0xFFFFFF;
|
||||
self.last_transfer_length = if (residual >= requested_length) 0 else requested_length - residual;
|
||||
return completionCode(event.status); // our transfer's completion (or error)
|
||||
if (trbType(event.control) != @intFromEnum(TrbType.transfer_event)) continue;
|
||||
if (self.serviceInterruptEvent(event)) continue; // a subscription's report
|
||||
const event_slot: u8 = @truncate(event.control >> 24);
|
||||
const event_dci: u32 = (event.control >> 16) & 0x1F;
|
||||
if (event_slot != slot_id or event_dci != dci) {
|
||||
std.log.info("dropped foreign transfer event (slot {d} dci {d}, code {d})", .{ event_slot, event_dci, completionCode(event.status) });
|
||||
continue;
|
||||
}
|
||||
const residual = event.status & 0xFFFFFF;
|
||||
self.last_transfer_length = if (residual >= requested_length) 0 else requested_length - residual;
|
||||
return completionCode(event.status); // our transfer's completion (or error)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -708,7 +1218,7 @@ pub const Controller = struct {
|
||||
|
||||
mmio.wmb();
|
||||
self.ringDoorbell(device.slot_id, 1); // DCI 1 = EP0
|
||||
const code = self.awaitTransfer(@intCast(data.len)) orelse return false;
|
||||
const code = self.awaitTransfer(device.slot_id, 1, @intCast(data.len)) orelse return false;
|
||||
if (code != @intFromEnum(CompletionCode.success) and code != @intFromEnum(CompletionCode.short_packet)) return false;
|
||||
|
||||
if (has_data and direction_in) {
|
||||
@@ -731,7 +1241,56 @@ pub const Controller = struct {
|
||||
/// endpoints into `device`, and select the configuration. After this the
|
||||
/// device is in the configured state and its interfaces are ready to match a
|
||||
/// class driver. Returns false on any control-transfer failure.
|
||||
/// Correct EP0's max packet size from the device itself. The context starts
|
||||
/// with the SPEED-DEFAULT (full-speed: 8, but the true value may be 8/16/32/
|
||||
/// 64 — byte 7 of the device descriptor). Read just the descriptor's first
|
||||
/// 8 bytes (always deliverable at any legal MPS0), and when the device
|
||||
/// disagrees with the context, issue Evaluate Context to fix EP0 before any
|
||||
/// longer transfer. Real controllers fault the full 18-byte read on a wrong
|
||||
/// MPS0; QEMU forgives it — the classic full-speed-mouse-on-real-hardware
|
||||
/// failure (M20).
|
||||
fn refreshMaxPacketSize0(self: *Controller, device: *Device) bool {
|
||||
// SuperSpeed (and above) fix EP0's max packet size at 512, and encode
|
||||
// bMaxPacketSize0 as an EXPONENT (9 = 2^9 = 512), not a literal size —
|
||||
// the default is already correct and byte 7 must NOT be read as a size.
|
||||
// Only full/low/high speed carry a literal 8/16/32/64 that can differ
|
||||
// from the speed default and need this correction. (Reading the SS
|
||||
// exponent as a size set EP0 to 9 bytes and broke every following
|
||||
// transfer — a SuperSpeed hub failing to enumerate on real hardware.)
|
||||
if (device.speed >= 4) return true;
|
||||
|
||||
var head: [8]u8 = undefined;
|
||||
const request = usb_abi.getDescriptor(.device, 0, 0, 8);
|
||||
if (!self.controlTransfer(device, request, head[0..], true)) return false;
|
||||
const actual: u32 = head[7];
|
||||
if (actual == 0 or actual == device.max_packet_size_0) return true;
|
||||
|
||||
// Input Control Context: add-flag A1 (EP0 only); EP0 context rebuilt
|
||||
// with the corrected MPS. Fields not being changed stay zero (the
|
||||
// controller evaluates only the added context).
|
||||
const cs = self.context_size;
|
||||
const base = device.input_context.virtual;
|
||||
@memset(@as([*]u8, @ptrFromInt(base))[0 .. 3 * cs], 0);
|
||||
contextDword(base, 0, 1, cs).* = 0b10; // A1
|
||||
contextDword(base, 2, 1, cs).* = (@as(u32, 3) << 1) | (@as(u32, 4) << 3) | (actual << 16);
|
||||
const physical = self.submitCommand(.{
|
||||
.parameter = device.input_context.physical,
|
||||
.control = trbControl(.evaluate_context, @as(u32, device.slot_id) << 24),
|
||||
});
|
||||
const code = self.awaitCommand(physical) orelse {
|
||||
std.log.info("slot {d}: Evaluate Context (MPS0 {d} -> {d}) timed out", .{ device.slot_id, device.max_packet_size_0, actual });
|
||||
return false;
|
||||
};
|
||||
if (code != @intFromEnum(CompletionCode.success)) {
|
||||
std.log.info("slot {d}: Evaluate Context (MPS0 {d} -> {d}) completion code {d}", .{ device.slot_id, device.max_packet_size_0, actual, code });
|
||||
return false;
|
||||
}
|
||||
device.max_packet_size_0 = actual;
|
||||
return true;
|
||||
}
|
||||
|
||||
pub fn enumerate(self: *Controller, device: *Device) bool {
|
||||
if (!self.refreshMaxPacketSize0(device)) return false;
|
||||
device.device_descriptor = self.getDeviceDescriptor(device) orelse return false;
|
||||
|
||||
// The configuration descriptor's own 9 bytes carry the total length of
|
||||
@@ -901,8 +1460,9 @@ pub const Controller = struct {
|
||||
mmio.wmb();
|
||||
const number: u8 = endpoint.address & 0x0F;
|
||||
const direction_in = endpoint.address & 0x80 != 0;
|
||||
self.ringDoorbell(device.slot_id, doorbellContextIndex(number, direction_in));
|
||||
const code = self.awaitTransfer(length) orelse return null;
|
||||
const dci = doorbellContextIndex(number, direction_in);
|
||||
self.ringDoorbell(device.slot_id, dci);
|
||||
const code = self.awaitTransfer(device.slot_id, dci, length) orelse return null;
|
||||
if (code != @intFromEnum(CompletionCode.success) and code != @intFromEnum(CompletionCode.short_packet)) return null;
|
||||
return self.last_transfer_length;
|
||||
}
|
||||
@@ -960,9 +1520,20 @@ pub const Controller = struct {
|
||||
if (!subscription.active or subscription.armed_trb_physical != trb_pointer) continue;
|
||||
const code = completionCode(event.status);
|
||||
if (code == @intFromEnum(CompletionCode.success) or code == @intFromEnum(CompletionCode.short_packet)) {
|
||||
const residual = event.status & 0xFFFFFF;
|
||||
const transferred: u16 = if (residual >= subscription.max_length) 0 else @intCast(subscription.max_length - residual);
|
||||
self.enqueueReport(subscription, transferred);
|
||||
if (subscription.hub) |hub_device| {
|
||||
// Hub status-change report: OR the changed-port bitmap into
|
||||
// the hub's pending mask (bit 0 = the hub itself, ignored;
|
||||
// bit P = downstream port P). The control transfers to
|
||||
// service it run on the bus tick, not here.
|
||||
const bytes: [*]const u8 = @ptrFromInt(subscription.buffer.virtual);
|
||||
var i: usize = 0;
|
||||
while (i < subscription.max_length and i < 4) : (i += 1)
|
||||
hub_device.hub_change_mask |= @as(u32, bytes[i]) << @intCast(i * 8);
|
||||
} else {
|
||||
const residual = event.status & 0xFFFFFF;
|
||||
const transferred: u16 = if (residual >= subscription.max_length) 0 else @intCast(subscription.max_length - residual);
|
||||
self.enqueueReport(subscription, transferred);
|
||||
}
|
||||
}
|
||||
self.armInterrupt(subscription); // keep polling
|
||||
return true;
|
||||
@@ -993,12 +1564,81 @@ pub const Controller = struct {
|
||||
return report;
|
||||
}
|
||||
|
||||
/// Drain any events currently on the event ring, dispatching interrupt reports
|
||||
/// into the queue. Non-blocking — called on the driver's timer tick.
|
||||
/// Drain any events currently on the event ring: interrupt reports into the
|
||||
/// report queue, PORT STATUS CHANGES into the port-change queue (hot-plug —
|
||||
/// these were silently dropped before M20). Non-blocking — called on the
|
||||
/// driver's timer tick.
|
||||
pub fn pump(self: *Controller) void {
|
||||
while (true) {
|
||||
const event = self.nextEvent(system.clock()) orelse return; // deadline=now: null when empty
|
||||
if (trbType(event.control) == @intFromEnum(TrbType.transfer_event)) _ = self.serviceInterruptEvent(event);
|
||||
const kind = trbType(event.control);
|
||||
if (kind == @intFromEnum(TrbType.transfer_event)) {
|
||||
_ = self.serviceInterruptEvent(event);
|
||||
} else if (kind == @intFromEnum(TrbType.port_status_change_event)) {
|
||||
// Port ID rides bits 31:24 of the TRB's first dword.
|
||||
const port: u32 = @intCast((event.parameter >> 24) & 0xFF);
|
||||
if (port == 0 or port > self.max_ports) continue;
|
||||
// Acknowledge the change bits so the port can signal again.
|
||||
const status = self.portStatus(port);
|
||||
self.writePortStatus(port, (status & ~portsc_write_1_to_clear) | (status & portsc_change_mask));
|
||||
if (self.port_change_count < self.port_changes.len) {
|
||||
self.port_changes[self.port_change_count] = port;
|
||||
self.port_change_count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Dequeue the oldest pending port change (a port whose connect state may
|
||||
/// have flipped), or null. The bus layer reads PORTSC to decide plug/unplug.
|
||||
pub fn takePortChange(self: *Controller) ?u32 {
|
||||
if (self.port_change_count == 0) return null;
|
||||
const port = self.port_changes[0];
|
||||
var i: usize = 1;
|
||||
while (i < self.port_change_count) : (i += 1) self.port_changes[i - 1] = self.port_changes[i];
|
||||
self.port_change_count -= 1;
|
||||
return port;
|
||||
}
|
||||
|
||||
/// Whether a port currently has a device connected (PORTSC.CCS).
|
||||
pub fn portConnected(self: *const Controller, port: u32) bool {
|
||||
return self.portStatus(port) & portsc_connected != 0;
|
||||
}
|
||||
|
||||
/// The tracked device on `port`, or null.
|
||||
pub fn deviceOnPort(self: *Controller, port: u32) ?*Device {
|
||||
for (&self.devices) |*device| {
|
||||
if (device.used and device.port == port) return device;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The next used device whose parent hub is `hub_slot` and slot id > `after`
|
||||
/// (for recursive teardown when a hub itself disconnects), or null.
|
||||
pub fn nextChildOf(self: *Controller, hub_slot: u8, after: u8) ?*Device {
|
||||
for (&self.devices) |*device| {
|
||||
if (device.used and device.parent_slot == hub_slot and device.slot_id > after) return device;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Tear a device down after unplug: cancel its interrupt subscriptions,
|
||||
/// Disable Slot (frees the controller's slot state), clear its context-array
|
||||
/// entry, and release the tracking slot. DMA regions leak (as elsewhere) —
|
||||
/// bounded by the device-slot count.
|
||||
pub fn tearDownDevice(self: *Controller, device: *Device) void {
|
||||
for (&self.subscriptions) |*subscription| {
|
||||
if (subscription.active and subscription.slot_id == device.slot_id) subscription.active = false;
|
||||
}
|
||||
const physical = self.submitCommand(.{
|
||||
.control = trbControl(.disable_slot, @as(u32, device.slot_id) << 24),
|
||||
});
|
||||
if (self.awaitCommand(physical)) |code| {
|
||||
if (code != @intFromEnum(CompletionCode.success))
|
||||
std.log.info("slot {d}: Disable Slot completion code {d}", .{ device.slot_id, code });
|
||||
} else std.log.info("slot {d}: Disable Slot timed out", .{device.slot_id});
|
||||
const array: [*]volatile u64 = @ptrFromInt(self.device_context_array.virtual);
|
||||
array[device.slot_id] = 0;
|
||||
device.used = false;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
//! The virtio-gpu control protocol — the command/response structs the driver exchanges with
|
||||
//! the device over its control virtqueue (virtio spec, "GPU Device"). `extern` structs, so
|
||||
//! the layout matches the little-endian wire format exactly. Host-tested for size. See
|
||||
//! docs/display-v2.md.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Control command / response types (virtio_gpu_ctrl_type). Commands are 0x01xx, responses
|
||||
/// 0x11xx (ok) / 0x12xx (error).
|
||||
pub const CmdType = enum(u32) {
|
||||
get_display_info = 0x0100,
|
||||
resource_create_2d = 0x0101,
|
||||
resource_unref = 0x0102,
|
||||
set_scanout = 0x0103,
|
||||
resource_flush = 0x0104,
|
||||
transfer_to_host_2d = 0x0105,
|
||||
resource_attach_backing = 0x0106,
|
||||
resource_detach_backing = 0x0107,
|
||||
get_edid = 0x010a,
|
||||
|
||||
resp_ok_nodata = 0x1100,
|
||||
resp_ok_display_info = 0x1101,
|
||||
resp_ok_edid = 0x1104,
|
||||
resp_err_unspec = 0x1200,
|
||||
_,
|
||||
};
|
||||
|
||||
/// Set in a command's `flags` to request a fence; the device echoes `fence_id` in the
|
||||
/// response and does not report completion until the command's effects are visible.
|
||||
pub const flag_fence: u32 = 1 << 0;
|
||||
|
||||
/// VIRTIO_GPU_F_EDID — device feature bit 1 (the low feature word): the device answers the
|
||||
/// `get_edid` command. Negotiate it only when the device offers it.
|
||||
pub const feature_edid: u32 = 1 << 1;
|
||||
|
||||
/// virtio_gpu_ctrl_hdr — the header on every command and response.
|
||||
pub const CtrlHdr = extern struct {
|
||||
type: u32,
|
||||
flags: u32 = 0,
|
||||
fence_id: u64 = 0,
|
||||
ctx_id: u32 = 0,
|
||||
ring_idx: u8 = 0,
|
||||
padding: [3]u8 = .{ 0, 0, 0 },
|
||||
};
|
||||
|
||||
pub const Rect = extern struct {
|
||||
x: u32,
|
||||
y: u32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
};
|
||||
|
||||
/// 2D pixel formats. QEMU's virtio-gpu host default is B8G8R8X8 (matches our bgrx).
|
||||
pub const format_b8g8r8x8_unorm: u32 = 2;
|
||||
pub const format_r8g8b8x8_unorm: u32 = 134;
|
||||
|
||||
pub const ResourceCreate2d = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
resource_id: u32,
|
||||
format: u32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
};
|
||||
|
||||
/// One scatter-gather entry of a resource's guest backing (a physical span).
|
||||
pub const MemEntry = extern struct {
|
||||
addr: u64,
|
||||
length: u32,
|
||||
padding: u32 = 0,
|
||||
};
|
||||
|
||||
/// Header for RESOURCE_ATTACH_BACKING; `nr_entries` `MemEntry` follow it inline.
|
||||
pub const ResourceAttachBacking = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
resource_id: u32,
|
||||
nr_entries: u32,
|
||||
};
|
||||
|
||||
pub const SetScanout = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
rect: Rect,
|
||||
scanout_id: u32,
|
||||
resource_id: u32,
|
||||
};
|
||||
|
||||
pub const ResourceFlush = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
rect: Rect,
|
||||
resource_id: u32,
|
||||
padding: u32 = 0,
|
||||
};
|
||||
|
||||
/// Copy the guest backing into the host resource for `rect` (2D resources must transfer
|
||||
/// before a flush shows the update).
|
||||
pub const TransferToHost2d = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
rect: Rect,
|
||||
offset: u64,
|
||||
resource_id: u32,
|
||||
padding: u32 = 0,
|
||||
};
|
||||
|
||||
pub const max_scanouts = 16;
|
||||
|
||||
pub const DisplayOne = extern struct {
|
||||
rect: Rect,
|
||||
enabled: u32,
|
||||
flags: u32,
|
||||
};
|
||||
|
||||
pub const RespDisplayInfo = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
pmodes: [max_scanouts]DisplayOne,
|
||||
};
|
||||
|
||||
pub const GetEdid = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
scanout: u32,
|
||||
padding: u32 = 0,
|
||||
};
|
||||
|
||||
pub const RespEdid = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
size: u32,
|
||||
padding: u32 = 0,
|
||||
edid: [1024]u8,
|
||||
};
|
||||
|
||||
test "virtio-gpu struct sizes match the wire layout" {
|
||||
try std.testing.expectEqual(@as(usize, 24), @sizeOf(CtrlHdr));
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(Rect));
|
||||
try std.testing.expectEqual(@as(usize, 40), @sizeOf(ResourceCreate2d));
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(MemEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(ResourceAttachBacking));
|
||||
try std.testing.expectEqual(@as(usize, 48), @sizeOf(SetScanout));
|
||||
try std.testing.expectEqual(@as(usize, 48), @sizeOf(ResourceFlush));
|
||||
try std.testing.expectEqual(@as(usize, 56), @sizeOf(TransferToHost2d));
|
||||
try std.testing.expectEqual(@as(usize, 24 + 4 + 4 + 1024), @sizeOf(RespEdid));
|
||||
}
|
||||
@@ -0,0 +1,625 @@
|
||||
//! /system/drivers/virtio-gpu — the virtio-gpu (virtio 1.0, modern PCI) display driver.
|
||||
//! The device manager spawns it for the display/other PCI function (class 0x0380) whose
|
||||
//! config space says vendor 0x1AF4 / device 0x1050; this instance claims that device and
|
||||
//! brings up a single 2D scanout.
|
||||
//!
|
||||
//! V3 (this increment): the whole path end to end, proven from serial without a screenshot.
|
||||
//! Claim the function, map its config space (resource 0) and the BAR that carries the
|
||||
//! virtio structures, walk the vendor capabilities to find common-config / notify, reset
|
||||
//! and negotiate VERSION_1, stand up the control virtqueue in DMA memory, then drive the
|
||||
//! GPU: RESOURCE_CREATE_2D → ATTACH_BACKING (a coherent DMA buffer) → SET_SCANOUT, paint a
|
||||
//! known test pattern, TRANSFER_TO_HOST_2D → RESOURCE_FLUSH, and **wait for the device's
|
||||
//! used-ring ack**. Reading the backing back confirms it is CPU-visible; the ack confirms
|
||||
//! the device consumed the frame. The compositor backend, hot-attach, mode-set/EDID, and
|
||||
//! restart/re-attach are V4–V6. See docs/display-v2.md.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const mmio = @import("mmio");
|
||||
const device = runtime.device;
|
||||
const dma = runtime.dma;
|
||||
const shared_memory = runtime.shared_memory;
|
||||
const system = runtime.system;
|
||||
const ipc = runtime.ipc;
|
||||
const dp = runtime.display_protocol;
|
||||
const sp = runtime.scanout_protocol;
|
||||
const dm = runtime.device_manager_protocol;
|
||||
const vp = @import("virtio-pci.zig");
|
||||
const vg = @import("virtio-gpu-protocol.zig");
|
||||
|
||||
/// The DisplayFormat (device-abi) our B8G8R8X8 scanout resource presents: bgrx = 1. Handed to
|
||||
/// the compositor in the announce so it packs colours in the surface's byte order.
|
||||
const display_format_bgrx: u32 = 1;
|
||||
|
||||
/// The PCI vendor/device ids of a modern virtio-gpu (Red Hat / virtio; GPU is a
|
||||
/// virtio-1.0-only device, so the id is always the modern 0x1050 — no legacy variant).
|
||||
const virtio_vendor: u16 = 0x1AF4;
|
||||
const virtio_gpu_device: u16 = 0x1050;
|
||||
|
||||
/// The scanout resource + shared surface are sized to the *largest* mode we offer; a mode
|
||||
/// change (V5) re-points the scanout rectangle within it, so the resource, its backing, and
|
||||
/// the shared surface never churn — and the surface's row stride is always `max_width`, which
|
||||
/// the compositor is told in the announce. Kept modest so the backing is an easy contiguous run.
|
||||
const max_width: u32 = 800;
|
||||
const max_height: u32 = 600;
|
||||
const scanout_bytes: usize = @as(usize, max_width) * max_height * 4;
|
||||
const resource_id: u32 = 1;
|
||||
|
||||
/// The modes this scanout offers (all ≤ max). The first is the mode it comes up in.
|
||||
const Mode = struct { width: u32, height: u32 };
|
||||
const offered_modes = [_]Mode{ .{ .width = 640, .height = 480 }, .{ .width = 800, .height = 600 } };
|
||||
|
||||
/// The active mode — the scanout rectangle within the max-sized surface. Changed by `set_mode`.
|
||||
var current_width: u32 = offered_modes[0].width;
|
||||
var current_height: u32 = offered_modes[0].height;
|
||||
|
||||
/// Monotonic fence id for fenced flushes; the device signals the fence when the flush is
|
||||
/// complete, which its used-ring ack already gates our synchronous present on. Completion
|
||||
/// feedback, not vblank — nothing here is paced to the display's refresh.
|
||||
var fence_next: u64 = 1;
|
||||
|
||||
/// Whether the device offered VIRTIO_GPU_F_EDID, so `get_edid` is worth issuing.
|
||||
var edid_available = false;
|
||||
|
||||
/// The panel refresh rate parsed from the EDID preferred timing (0 = unknown). Carried to
|
||||
/// the compositor in the announce so its frame clock paces to the panel, not a guess.
|
||||
var edid_refresh_hz: u32 = 0;
|
||||
|
||||
/// The control virtqueue. We drive it synchronously — one command, notify, poll the used
|
||||
/// ring — so a depth of 16 is ample; we ask the device to shrink to it (virtio 1.0 lets the
|
||||
/// driver reduce queue_size), keeping the whole ring inside one page.
|
||||
const queue_size: u16 = 16;
|
||||
const desc_offset: usize = 0; // 16 * 16 = 256 bytes
|
||||
const avail_offset: usize = 256; // flags + idx + ring[16] + used_event = 38 bytes
|
||||
const used_offset: usize = 1024; // flags + idx + ring[16] + avail_event = 134 bytes
|
||||
|
||||
/// The command scratch: the request the device reads, then its response, in one DMA page.
|
||||
const request_offset: usize = 0;
|
||||
const response_offset: usize = 2048;
|
||||
|
||||
var device_id: u64 = 0;
|
||||
|
||||
// Mapped virtio structures (virtual addresses into the device's BAR).
|
||||
var common_base: usize = 0;
|
||||
var notify_base: usize = 0;
|
||||
var notify_multiplier: u32 = 0;
|
||||
var notify_addr: usize = 0;
|
||||
|
||||
// Per-BAR mapping cache: several capabilities usually share one BAR, and mmio_map must not
|
||||
// be asked to map the same resource twice.
|
||||
var bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 };
|
||||
|
||||
// DMA memory: the virtqueue rings and the command scratch.
|
||||
var ring: dma.Region = undefined;
|
||||
var command: dma.Region = undefined;
|
||||
|
||||
// The scanout backing is a **shared** (shared-memory) region, not DMA: cacheable so the compositor
|
||||
// composites into it cheaply (x86 DMA is coherent, so the device still sees the writes), and
|
||||
// shareable so the same physical pages the device scans out of are the ones the compositor
|
||||
// paints. The driver keeps the capability to hand to the compositor in the announce.
|
||||
var surface: shared_memory.Region = undefined;
|
||||
|
||||
// Split-virtqueue producer/consumer shadows.
|
||||
var avail_shadow: u16 = 0;
|
||||
var used_shadow: u16 = 0;
|
||||
|
||||
// --- common-config register access (little-endian MMIO at `common_base`) ---------------
|
||||
|
||||
fn cfgRead(comptime T: type, comptime field: []const u8) T {
|
||||
return mmio.read(T, common_base + @offsetOf(vp.CommonCfg, field));
|
||||
}
|
||||
fn cfgWrite(comptime T: type, comptime field: []const u8, value: T) void {
|
||||
mmio.write(T, common_base + @offsetOf(vp.CommonCfg, field), value);
|
||||
}
|
||||
/// Write a 64-bit common-config register as two 32-bit halves (low then high) — the widest
|
||||
/// access every virtio-pci host is required to accept for the queue-address registers.
|
||||
fn cfgWrite64(comptime field: []const u8, value: u64) void {
|
||||
const at = common_base + @offsetOf(vp.CommonCfg, field);
|
||||
mmio.write(u32, at, @truncate(value));
|
||||
mmio.write(u32, at + 4, @truncate(value >> 32));
|
||||
}
|
||||
fn orStatus(bit: u8) void {
|
||||
cfgWrite(u8, "device_status", cfgRead(u8, "device_status") | bit);
|
||||
}
|
||||
|
||||
// --- PCI config-space capability walk (config space is resource 0) ---------------------
|
||||
|
||||
/// Map the BAR numbered `bar` (0..5) and return its virtual base, correlating the BAR's
|
||||
/// physical address (read from config space) with one of our device resources — because a
|
||||
/// virtio capability names a BAR *number*, while `mmio_map` takes a *resource index* (and
|
||||
/// resource 0 is config space, so BAR resources are re-numbered and gaps skipped).
|
||||
fn mapBar(config: usize, descriptor: *const device.DeviceDescriptor, bar: u8) ?usize {
|
||||
if (bar >= 6) return null;
|
||||
if (bar_virtual[bar] != 0) return bar_virtual[bar];
|
||||
|
||||
const low = mmio.read(u32, config + 0x10 + @as(usize, bar) * 4);
|
||||
if (low & 0x1 != 0) return null; // an I/O-space BAR — virtio structures are in memory BARs
|
||||
var base: u64 = low & 0xFFFF_FFF0;
|
||||
if ((low & 0x6) == 0x4) { // 64-bit memory BAR: the high half is the next dword
|
||||
const high = mmio.read(u32, config + 0x10 + (@as(usize, bar) + 1) * 4);
|
||||
base |= @as(u64, high) << 32;
|
||||
}
|
||||
|
||||
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and resource.start == base) {
|
||||
const v = device.mmioMap(device_id, index) orelse return null;
|
||||
bar_virtual[bar] = v;
|
||||
return v;
|
||||
}
|
||||
}
|
||||
std.log.info("BAR {d} (physical 0x{x}) is not a mapped resource", .{ bar, base });
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Walk the PCI capability list from mapped config space, recording the common-config and
|
||||
/// notify structures (the only two V3 needs). Returns false if either is missing.
|
||||
fn walkCapabilities(config: usize, descriptor: *const device.DeviceDescriptor) bool {
|
||||
if (mmio.read(u16, config + 0x06) & 0x10 == 0) { // Status bit 4: capabilities list present
|
||||
std.log.info("device has no PCI capability list", .{});
|
||||
return false;
|
||||
}
|
||||
var cap: u8 = @as(u8, @truncate(mmio.read(u8, config + 0x34))) & 0xFC;
|
||||
var guard: u32 = 0;
|
||||
while (cap != 0 and guard < 48) : (guard += 1) {
|
||||
const at = config + cap;
|
||||
const id = mmio.read(u8, at + 0);
|
||||
const next = mmio.read(u8, at + 1) & 0xFC;
|
||||
// Only map BARs for the structures V3 uses (common + notify). The other virtio
|
||||
// capabilities (isr, device, and especially the cfg_pci back-door, which carries a
|
||||
// placeholder bar=0/offset=0) reference BARs we never touch, so mapping them would
|
||||
// just log spurious "not a mapped resource" noise.
|
||||
if (id == vp.pci_cap_vendor) {
|
||||
const cfg_type = mmio.read(u8, at + 3);
|
||||
if (cfg_type == vp.cfg_common or cfg_type == vp.cfg_notify) {
|
||||
const bar = mmio.read(u8, at + 4);
|
||||
const offset = mmio.read(u32, at + 8);
|
||||
if (mapBar(config, descriptor, bar)) |bar_base| {
|
||||
if (cfg_type == vp.cfg_common) {
|
||||
common_base = bar_base + offset;
|
||||
} else {
|
||||
notify_base = bar_base + offset;
|
||||
notify_multiplier = mmio.read(u32, at + 16); // virtio_pci_notify_cap tail
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
cap = next;
|
||||
}
|
||||
if (common_base == 0 or notify_base == 0) {
|
||||
std.log.info("missing common-config or notify capability", .{});
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- the control virtqueue -------------------------------------------------------------
|
||||
|
||||
/// Publish the two-descriptor chain (request read by the device, response written by it),
|
||||
/// notify the control queue, and wait for the device to return the buffer on the used ring.
|
||||
fn submit(request_len: usize, response_len: usize) bool {
|
||||
const desc: [*]vp.Desc = @ptrFromInt(ring.virtual + desc_offset);
|
||||
desc[0] = .{
|
||||
.addr = command.physical + request_offset,
|
||||
.len = @intCast(request_len),
|
||||
.flags = vp.desc_flag_next,
|
||||
.next = 1,
|
||||
};
|
||||
desc[1] = .{
|
||||
.addr = command.physical + response_offset,
|
||||
.len = @intCast(response_len),
|
||||
.flags = vp.desc_flag_write,
|
||||
.next = 0,
|
||||
};
|
||||
|
||||
const avail_ring: [*]u16 = @ptrFromInt(ring.virtual + avail_offset + 4);
|
||||
avail_ring[avail_shadow % queue_size] = 0; // head of the chain is descriptor 0
|
||||
mmio.wmb();
|
||||
avail_shadow +%= 1;
|
||||
mmio.write(u16, ring.virtual + avail_offset + 2, avail_shadow); // avail.idx
|
||||
mmio.wmb();
|
||||
|
||||
mmio.write(u16, notify_addr, 0); // ring the control queue's doorbell
|
||||
return waitUsed();
|
||||
}
|
||||
|
||||
/// Spin, then sleep-poll, on the used-ring index until the device advances it. QEMU
|
||||
/// processes the notify on its own thread, so the ack usually lands immediately; the sleep
|
||||
/// fallback covers a device that defers it without burning the CPU.
|
||||
fn waitUsed() bool {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 2000) : (tries += 1) {
|
||||
mmio.rmb();
|
||||
const idx = mmio.read(u16, ring.virtual + used_offset + 2); // used.idx
|
||||
if (idx != used_shadow) {
|
||||
used_shadow = idx;
|
||||
return true;
|
||||
}
|
||||
if (tries > 8) system.sleep(1);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// The type field of the response the device wrote — `resp_ok_nodata` on success.
|
||||
fn responseType() u32 {
|
||||
const response: *vg.CtrlHdr = @ptrFromInt(command.virtual + response_offset);
|
||||
return response.type;
|
||||
}
|
||||
|
||||
/// Submit a command whose response is a bare header, returning its response type (0 if the
|
||||
/// device never acked).
|
||||
fn command_nodata(request_len: usize) u32 {
|
||||
if (!submit(request_len, @sizeOf(vg.CtrlHdr))) return 0;
|
||||
return responseType();
|
||||
}
|
||||
|
||||
const ok_nodata: u32 = @intFromEnum(vg.CmdType.resp_ok_nodata);
|
||||
|
||||
fn requestAt(comptime T: type) *T {
|
||||
return @ptrFromInt(command.virtual + request_offset);
|
||||
}
|
||||
|
||||
/// A deterministic, recognisable pixel so a read-back is a real check, not a tautology.
|
||||
fn testPixel(index: u32) u32 {
|
||||
return 0xFF00_0000 | (index *% 0x9E37_79B1);
|
||||
}
|
||||
|
||||
// --- bring-up --------------------------------------------------------------------------
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(device_id)) {
|
||||
std.log.info("unable to claim device {d}", .{device_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||
const total = device.enumerate(&descriptors);
|
||||
const descriptor = for (descriptors[0..@min(total, descriptors.len)]) |*d| {
|
||||
if (d.id == device_id) break d;
|
||||
} else {
|
||||
std.log.info("device {d} not in the device tree", .{device_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// Config space is resource 0. Confirm it really is a virtio-gpu, then enable memory-space
|
||||
// decode + bus mastering (the device DMAs the ring and backing out of RAM); pci-bus only
|
||||
// preserves whatever the firmware left, and a secondary display is often left disabled.
|
||||
const config = device.mmioMap(device_id, 0) orelse {
|
||||
std.log.info("config-space map failed", .{});
|
||||
return false;
|
||||
};
|
||||
const vendor = mmio.read(u16, config + 0x00);
|
||||
const dev = mmio.read(u16, config + 0x02);
|
||||
if (vendor != virtio_vendor or dev != virtio_gpu_device) {
|
||||
std.log.info("not a virtio-gpu (vendor 0x{x} device 0x{x})", .{ vendor, dev });
|
||||
return false;
|
||||
}
|
||||
mmio.write(u16, config + 0x04, mmio.read(u16, config + 0x04) | 0x06); // MEM + bus master
|
||||
|
||||
if (!walkCapabilities(config, descriptor)) return false;
|
||||
|
||||
// Reset, then the modern feature handshake: acknowledge, take driver ownership, require
|
||||
// VERSION_1 and offer nothing else, and confirm the device accepts that.
|
||||
cfgWrite(u8, "device_status", 0);
|
||||
orStatus(vp.status_acknowledge);
|
||||
orStatus(vp.status_driver);
|
||||
|
||||
// Low feature word (device-specific): note whether the device offers EDID (bit 1).
|
||||
cfgWrite(u32, "device_feature_select", 0);
|
||||
edid_available = cfgRead(u32, "device_feature") & vg.feature_edid != 0;
|
||||
// High feature word: VERSION_1 (bit 32) is required for a modern device.
|
||||
cfgWrite(u32, "device_feature_select", vp.feature_version_1_word);
|
||||
if (cfgRead(u32, "device_feature") & vp.feature_version_1_bit == 0) {
|
||||
std.log.info("device does not offer VERSION_1 (not a modern device)", .{});
|
||||
return false;
|
||||
}
|
||||
// Accept exactly VERSION_1, plus EDID when the device offered it (never a feature it didn't).
|
||||
cfgWrite(u32, "driver_feature_select", 0);
|
||||
cfgWrite(u32, "driver_feature", if (edid_available) vg.feature_edid else 0);
|
||||
cfgWrite(u32, "driver_feature_select", vp.feature_version_1_word);
|
||||
cfgWrite(u32, "driver_feature", vp.feature_version_1_bit);
|
||||
orStatus(vp.status_features_ok);
|
||||
if (cfgRead(u8, "device_status") & vp.status_features_ok == 0) {
|
||||
std.log.info("device rejected the negotiated features", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Stand up the control virtqueue (queue 0) in coherent DMA memory.
|
||||
cfgWrite(u16, "queue_select", 0);
|
||||
const device_qsize = cfgRead(u16, "queue_size");
|
||||
if (device_qsize < queue_size) {
|
||||
std.log.info("control queue too small ({d})", .{device_qsize});
|
||||
return false;
|
||||
}
|
||||
ring = dma.alloc(4096, dma.coherent) orelse {
|
||||
std.log.info("virtqueue allocation failed", .{});
|
||||
return false;
|
||||
};
|
||||
command = dma.alloc(4096, dma.coherent) orelse {
|
||||
std.log.info("command-buffer allocation failed", .{});
|
||||
return false;
|
||||
};
|
||||
mmio.write(u16, ring.virtual + avail_offset, 1); // VIRTQ_AVAIL_F_NO_INTERRUPT: we poll
|
||||
cfgWrite(u16, "queue_size", queue_size);
|
||||
cfgWrite64("queue_desc", ring.physical + desc_offset);
|
||||
cfgWrite64("queue_driver", ring.physical + avail_offset);
|
||||
cfgWrite64("queue_device", ring.physical + used_offset);
|
||||
cfgWrite(u16, "queue_msix_vector", 0xFFFF); // VIRTIO_MSI_NO_VECTOR
|
||||
cfgWrite(u16, "queue_enable", 1);
|
||||
|
||||
cfgWrite(u16, "queue_select", 0);
|
||||
notify_addr = notify_base + @as(usize, cfgRead(u16, "queue_notify_off")) * notify_multiplier;
|
||||
|
||||
orStatus(vp.status_driver_ok);
|
||||
|
||||
// Drive the GPU: create a 2D resource at the *max* mode, back it with a shared surface, and
|
||||
// scan out the current-mode rectangle within it.
|
||||
{
|
||||
const request = requestAt(vg.ResourceCreate2d);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_create_2d) },
|
||||
.resource_id = resource_id,
|
||||
.format = vg.format_b8g8r8x8_unorm,
|
||||
.width = max_width,
|
||||
.height = max_height,
|
||||
};
|
||||
if (command_nodata(@sizeOf(vg.ResourceCreate2d)) != ok_nodata) {
|
||||
std.log.info("resource_create_2d failed", .{});
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Back the resource with a shared (shared-memory) surface, so the compositor and the device work
|
||||
// the same physical pages. The device needs the guest-physical base for attach_backing.
|
||||
surface = shared_memory.create(scanout_bytes) orelse {
|
||||
std.log.info("scanout surface allocation failed", .{});
|
||||
return false;
|
||||
};
|
||||
const surface_physical = shared_memory.physical(surface.handle) orelse {
|
||||
std.log.info("could not resolve the scanout surface physical address", .{});
|
||||
return false;
|
||||
};
|
||||
{
|
||||
const request = requestAt(vg.ResourceAttachBacking);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_attach_backing) },
|
||||
.resource_id = resource_id,
|
||||
.nr_entries = 1,
|
||||
};
|
||||
const entry: *vg.MemEntry = @ptrFromInt(command.virtual + request_offset + @sizeOf(vg.ResourceAttachBacking));
|
||||
entry.* = .{ .addr = surface_physical, .length = @intCast(scanout_bytes) };
|
||||
if (command_nodata(@sizeOf(vg.ResourceAttachBacking) + @sizeOf(vg.MemEntry)) != ok_nodata) {
|
||||
std.log.info("resource_attach_backing failed", .{});
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!setScanoutRect()) {
|
||||
std.log.info("set_scanout failed", .{});
|
||||
return false;
|
||||
}
|
||||
std.log.info("scanout {d}x{d} online", .{ current_width, current_height });
|
||||
|
||||
// Hello the device manager so it counts us as up (and does not stop us at the hello
|
||||
// deadline). A restarted instance re-hellos here and re-announces below — the compositor
|
||||
// re-attaches to the fresh scanout (V6).
|
||||
helloManager();
|
||||
|
||||
// Read the monitor's EDID (best-effort, when the device offers it) — the mode list a real
|
||||
// driver derives from it; we log the preferred mode and keep our fixed offered list.
|
||||
readEdid();
|
||||
|
||||
// Paint a known pattern, present it, and read it back — the V3 self-test that proves the
|
||||
// whole path (virtqueue, resource, shared backing, transfer, flush) before a client attaches.
|
||||
const pixels: [*]u32 = @ptrCast(@alignCast(surface.ptr));
|
||||
const pixel_count: usize = @as(usize, max_width) * max_height;
|
||||
for (0..pixel_count) |i| pixels[i] = testPixel(@intCast(i));
|
||||
|
||||
if (!presentFull()) {
|
||||
std.log.info("initial present failed", .{});
|
||||
return false;
|
||||
}
|
||||
// The scanout surface is CPU-visible RAM: read the pattern back to prove the mapping,
|
||||
// which together with the flush ack above is the automated stand-in for "it's on screen".
|
||||
mmio.rmb();
|
||||
if (pixels[0] != testPixel(0) or pixels[pixel_count / 2] != testPixel(@intCast(pixel_count / 2))) {
|
||||
std.log.info("pixel read-back mismatch", .{});
|
||||
return false;
|
||||
}
|
||||
std.log.info("flush acked, pixel check ok", .{});
|
||||
|
||||
// Offer the shared surface to the compositor so it upgrades off the GOP floor (V4).
|
||||
announce();
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Point scanout 0 at the current-mode rectangle of the resource. Reused by initial bring-up
|
||||
/// and by `set_mode`.
|
||||
fn setScanoutRect() bool {
|
||||
const request = requestAt(vg.SetScanout);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.set_scanout) },
|
||||
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||
.scanout_id = 0,
|
||||
.resource_id = resource_id,
|
||||
};
|
||||
return command_nodata(@sizeOf(vg.SetScanout)) == ok_nodata;
|
||||
}
|
||||
|
||||
/// Read and log the monitor's preferred mode from its EDID (VIRTIO_GPU_F_EDID). Best-effort:
|
||||
/// a device that doesn't offer EDID, or a missing/short block, is logged and ignored.
|
||||
fn readEdid() void {
|
||||
if (!edid_available) {
|
||||
std.log.info("EDID not offered by device", .{});
|
||||
return;
|
||||
}
|
||||
const request = requestAt(vg.GetEdid);
|
||||
request.* = .{ .hdr = .{ .type = @intFromEnum(vg.CmdType.get_edid) }, .scanout = 0 };
|
||||
if (!submit(@sizeOf(vg.GetEdid), @sizeOf(vg.RespEdid))) {
|
||||
std.log.info("EDID request not acked", .{});
|
||||
return;
|
||||
}
|
||||
const response: *vg.RespEdid = @ptrFromInt(command.virtual + response_offset);
|
||||
if (response.hdr.type != @intFromEnum(vg.CmdType.resp_ok_edid) or response.size < 64) {
|
||||
std.log.info("EDID unavailable", .{});
|
||||
return;
|
||||
}
|
||||
// The first detailed timing descriptor (EDID base-block offset 54) is the preferred mode:
|
||||
// active pixels are 12-bit, low byte + high nibble (bytes 2/4 horizontal, 5/7 vertical).
|
||||
// The refresh rate is derived from the same descriptor: pixel clock (bytes 0-1, 10 kHz
|
||||
// units) over total (active + blanking) pixels per frame — the loader does the identical
|
||||
// computation for the boot framebuffer (boot/efi.zig edidNative).
|
||||
const e = &response.edid;
|
||||
const h_active = @as(u32, e[56]) | (@as(u32, e[58] & 0xF0) << 4);
|
||||
const v_active = @as(u32, e[59]) | (@as(u32, e[61] & 0xF0) << 4);
|
||||
const clock_hz = (@as(u64, e[54]) | (@as(u64, e[55]) << 8)) * 10_000;
|
||||
const h_blank = @as(u64, e[57]) | (@as(u64, e[58] & 0x0F) << 8);
|
||||
const v_blank = @as(u64, e[60]) | (@as(u64, e[61] & 0x0F) << 8);
|
||||
const total = (@as(u64, h_active) + h_blank) * (@as(u64, v_active) + v_blank);
|
||||
if (total != 0) edid_refresh_hz = @intCast((clock_hz + total / 2) / total);
|
||||
std.log.info("EDID preferred mode {d}x{d} @ {d} Hz", .{ h_active, v_active, edid_refresh_hz });
|
||||
}
|
||||
|
||||
/// Present the whole surface: copy the guest backing into the host resource, then flush it to
|
||||
/// the panel. Reused by the V3 self-test and by every compositor present over `.scanout`. V4
|
||||
/// presents the full surface; the damage-rect fast path is a later refinement.
|
||||
fn presentFull() bool {
|
||||
mmio.wmb(); // the surface writes must be visible before the device transfers them
|
||||
{
|
||||
// Transfer the current-mode rectangle from the guest backing to the host resource. The
|
||||
// device uses the resource's (max) width as the row stride, so the top-left rect at
|
||||
// offset 0 is exactly the visible area — the compositor composes at that same stride.
|
||||
const request = requestAt(vg.TransferToHost2d);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.transfer_to_host_2d) },
|
||||
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||
.offset = 0,
|
||||
.resource_id = resource_id,
|
||||
};
|
||||
if (command_nodata(@sizeOf(vg.TransferToHost2d)) != ok_nodata) return false;
|
||||
}
|
||||
{
|
||||
// A fenced flush: the device signals the fence once it has consumed the frame — which
|
||||
// its used-ring ack, what our synchronous submit waits on, already gates. Completion
|
||||
// feedback and a tear-free snapshot, not vblank pacing.
|
||||
const request = requestAt(vg.ResourceFlush);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_flush), .flags = vg.flag_fence, .fence_id = fence_next },
|
||||
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||
.resource_id = resource_id,
|
||||
};
|
||||
fence_next += 1;
|
||||
if (command_nodata(@sizeOf(vg.ResourceFlush)) != ok_nodata) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Hello the device manager (role: bus — we own a PCI function, though we report no children):
|
||||
/// the handshake that marks us up so the manager doesn't stop us at the hello deadline, and
|
||||
/// (as a supervised driver) restarts us if we die. Best-effort: without a manager we still run.
|
||||
fn helloManager() void {
|
||||
var tries: u32 = 0;
|
||||
const manager = while (tries < 100) : (tries += 1) {
|
||||
if (ipc.lookup(.device_manager)) |h| break h;
|
||||
system.sleep(20);
|
||||
} else {
|
||||
std.log.info("no device manager to hello", .{});
|
||||
return;
|
||||
};
|
||||
const hello = dm.Hello{ .role = @intFromEnum(dm.Role.bus), .device_id = device_id };
|
||||
var reply: [dm.reply_size]u8 = undefined;
|
||||
const n = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch {
|
||||
std.log.info("hello call failed", .{});
|
||||
return;
|
||||
};
|
||||
if (n < dm.reply_size or std.mem.bytesToValue(dm.HelloReply, reply[0..dm.reply_size]).status != 0) {
|
||||
std.log.info("hello refused", .{});
|
||||
return;
|
||||
}
|
||||
std.log.info("hello acknowledged", .{});
|
||||
}
|
||||
|
||||
/// Announce the scanout to the display service so it upgrades off the GOP framebuffer: hand it
|
||||
/// the shared surface as a capability plus the geometry. Best-effort and non-fatal — without a
|
||||
/// display service (the standalone virtio-gpu bring-up test) the driver is still a valid
|
||||
/// scanout service; it just serves no one. The display replies immediately (it defers its
|
||||
/// first present to a timer), so this returns before we start serving `.scanout` — no deadlock.
|
||||
fn announce() void {
|
||||
var tries: u32 = 0;
|
||||
const display = while (tries < 50) : (tries += 1) {
|
||||
if (ipc.lookup(.display)) |h| break h;
|
||||
system.sleep(20);
|
||||
} else {
|
||||
std.log.info("no display service to announce to (scanout-only)", .{});
|
||||
return;
|
||||
};
|
||||
var request = dp.Request{
|
||||
.operation = @intFromEnum(dp.Operation.attach_scanout),
|
||||
.x = max_width, // the shared surface's row stride in pixels (it is sized to the max mode)
|
||||
.y = edid_refresh_hz, // the panel refresh from EDID (0 = unknown) — the frame-clock seed
|
||||
.width = current_width,
|
||||
.height = current_height,
|
||||
.colour = display_format_bgrx,
|
||||
};
|
||||
var reply: [dp.reply_size]u8 = undefined;
|
||||
_ = ipc.callCap(display, std.mem.asBytes(&request), &reply, surface.handle) catch {
|
||||
std.log.info("announce to display failed", .{});
|
||||
return;
|
||||
};
|
||||
std.log.info("announced scanout to display", .{});
|
||||
}
|
||||
|
||||
/// A `sp.Reply{status}` written into `reply`.
|
||||
fn scanoutStatus(reply: []u8, ok: bool) usize {
|
||||
const response = sp.Reply{ .status = if (ok) 0 else -1 };
|
||||
@memcpy(reply[0..sp.reply_size], std.mem.asBytes(&response));
|
||||
return sp.reply_size;
|
||||
}
|
||||
|
||||
/// The `.scanout` service: the compositor drives present / mode queries here. The pixels are
|
||||
/// already in the shared surface, so a present is a transfer-to-host + fenced flush; a mode
|
||||
/// change just re-points the scanout rectangle (the surface is sized to the largest mode).
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
if (message.len < sp.request_size) return 0;
|
||||
const request = std.mem.bytesToValue(sp.Request, message[0..sp.request_size]);
|
||||
switch (request.operation) {
|
||||
@intFromEnum(sp.Operation.present) => return scanoutStatus(reply, presentFull()),
|
||||
@intFromEnum(sp.Operation.get_modes) => {
|
||||
var response = sp.ModesReply{ .status = 0, .count = offered_modes.len, .modes = undefined };
|
||||
for (0..sp.max_modes) |i| {
|
||||
response.modes[i] = if (i < offered_modes.len)
|
||||
.{ .width = offered_modes[i].width, .height = offered_modes[i].height }
|
||||
else
|
||||
.{ .width = 0, .height = 0 };
|
||||
}
|
||||
@memcpy(reply[0..sp.modes_reply_size], std.mem.asBytes(&response));
|
||||
return sp.modes_reply_size;
|
||||
},
|
||||
@intFromEnum(sp.Operation.set_mode) => {
|
||||
const w = request.width;
|
||||
const h = request.height;
|
||||
if (w == 0 or h == 0 or w > max_width or h > max_height) return scanoutStatus(reply, false);
|
||||
current_width = w;
|
||||
current_height = h;
|
||||
return scanoutStatus(reply, setScanoutRect());
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = system.write("virtio-gpu: missing device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(256, .{
|
||||
.service = .scanout,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
//! virtio 1.0 PCI transport — the vendor capabilities in PCI config space that point at the
|
||||
//! device's structures (common config, notify, ISR) in a BAR, the common-config register
|
||||
//! block, and the split-virtqueue layout. `extern` structs matching the spec. Host-tested
|
||||
//! for size. See docs/display-v2.md.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// PCI vendor-specific capability id (0x09) — virtio 1.0 structures are advertised as these.
|
||||
pub const pci_cap_vendor: u8 = 0x09;
|
||||
|
||||
/// virtio_pci_cap `cfg_type`: which structure a vendor capability points at.
|
||||
pub const cfg_common: u8 = 1;
|
||||
pub const cfg_notify: u8 = 2;
|
||||
pub const cfg_isr: u8 = 3;
|
||||
pub const cfg_device: u8 = 4;
|
||||
pub const cfg_pci: u8 = 5;
|
||||
|
||||
/// virtio_pci_cap — a vendor capability naming a structure at (bar, offset, length) within
|
||||
/// a PCI BAR. Read straight out of config space.
|
||||
pub const PciCap = extern struct {
|
||||
cap_vndr: u8, // 0x09
|
||||
cap_next: u8, // next capability's offset in config space (0 = end)
|
||||
cap_len: u8,
|
||||
cfg_type: u8, // cfg_common / cfg_notify / ...
|
||||
bar: u8, // which BAR the structure lives in
|
||||
padding: [3]u8,
|
||||
offset: u32, // offset within the BAR
|
||||
length: u32, // length of the structure
|
||||
};
|
||||
|
||||
/// virtio_pci_notify_cap: a notify capability carries a multiplier after the base cap; the
|
||||
/// per-queue notify address is `notify_base + queue_notify_off * notify_off_multiplier`.
|
||||
pub const NotifyCap = extern struct {
|
||||
cap: PciCap,
|
||||
notify_off_multiplier: u32,
|
||||
};
|
||||
|
||||
/// virtio_pci_common_cfg — the common configuration register block (little-endian MMIO).
|
||||
pub const CommonCfg = extern struct {
|
||||
device_feature_select: u32,
|
||||
device_feature: u32,
|
||||
driver_feature_select: u32,
|
||||
driver_feature: u32,
|
||||
msix_config: u16,
|
||||
num_queues: u16,
|
||||
device_status: u8,
|
||||
config_generation: u8,
|
||||
queue_select: u16,
|
||||
queue_size: u16,
|
||||
queue_msix_vector: u16,
|
||||
queue_enable: u16,
|
||||
queue_notify_off: u16,
|
||||
queue_desc: u64,
|
||||
queue_driver: u64,
|
||||
queue_device: u64,
|
||||
};
|
||||
|
||||
/// device_status bits (written to `CommonCfg.device_status` during bring-up).
|
||||
pub const status_acknowledge: u8 = 1;
|
||||
pub const status_driver: u8 = 2;
|
||||
pub const status_driver_ok: u8 = 4;
|
||||
pub const status_features_ok: u8 = 8;
|
||||
|
||||
/// VIRTIO_F_VERSION_1 — feature bit 32 (in the second 32-bit feature word). Required for a
|
||||
/// modern device; we negotiate exactly this bit and nothing else.
|
||||
pub const feature_version_1_word: u32 = 1; // device_feature_select value for bits 32..63
|
||||
pub const feature_version_1_bit: u32 = 1 << 0; // bit 32 within that word
|
||||
|
||||
// --- split virtqueue -------------------------------------------------------
|
||||
|
||||
pub const Desc = extern struct {
|
||||
addr: u64, // guest-physical
|
||||
len: u32,
|
||||
flags: u16,
|
||||
next: u16,
|
||||
};
|
||||
pub const desc_flag_next: u16 = 1; // buffer continues in `next`
|
||||
pub const desc_flag_write: u16 = 2; // device-writable (else driver-writable/device-readable)
|
||||
|
||||
/// The available ring's fixed header; a `[queue_size]u16` ring and a trailing `used_event`
|
||||
/// u16 follow it in memory (laid out by the driver).
|
||||
pub const AvailHdr = extern struct {
|
||||
flags: u16,
|
||||
idx: u16,
|
||||
};
|
||||
|
||||
/// One entry of the used ring.
|
||||
pub const UsedElem = extern struct {
|
||||
id: u32,
|
||||
len: u32,
|
||||
};
|
||||
|
||||
/// The used ring's fixed header; a `[queue_size]UsedElem` ring and a trailing `avail_event`
|
||||
/// u16 follow it.
|
||||
pub const UsedHdr = extern struct {
|
||||
flags: u16,
|
||||
idx: u16,
|
||||
};
|
||||
|
||||
test "virtio-pci struct sizes match the spec" {
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(PciCap));
|
||||
try std.testing.expectEqual(@as(usize, 56), @sizeOf(CommonCfg));
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(Desc));
|
||||
try std.testing.expectEqual(@as(usize, 8), @sizeOf(UsedElem));
|
||||
}
|
||||
@@ -1,7 +1,13 @@
|
||||
//! The initial_ramdisk (initial ramdisk) container format — shared by the build-time
|
||||
//! packer (tools/make-initial-ramdisk.py) and the kernel that unpacks it. Deliberately
|
||||
//! trivial: a header, a table of fixed-size entries, then the concatenated file
|
||||
//! blobs. We own both producer and consumer, so it need be no fancier.
|
||||
//! The initial_ramdisk (initial ramdisk) container format — built in RAM by the
|
||||
//! bootloader (boot/efi.zig walks the boot volume's /system tree) and unpacked by
|
||||
//! the kernel. Deliberately trivial: a header, a table of fixed-size entries, then
|
||||
//! the concatenated file blobs. We own both producer and consumer, so it need be
|
||||
//! no fancier.
|
||||
//!
|
||||
//! v2: entry names are full FHS paths ("/system/services/init"), 64 bytes — the
|
||||
//! same limit as a task name (abi.maximum_process_name), so a path-named task is
|
||||
//! never truncated. The boot volume's file tree is the single source of truth;
|
||||
//! this image is only the loader→kernel handoff snapshot of it.
|
||||
//!
|
||||
//! Layout:
|
||||
//! Header (magic, count)
|
||||
@@ -10,8 +16,14 @@
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// "DNRD" — identifies a danos initial_ramdisk image.
|
||||
pub const magic: u32 = 0x444E5244;
|
||||
/// "DNR2" — identifies a danos initial_ramdisk image, format v2 (path names).
|
||||
/// The v1 magic ("DNRD", basename entries) is rejected: a stale image should
|
||||
/// fail loudly at Reader.init, not misparse names.
|
||||
pub const magic: u32 = 0x32524E44;
|
||||
|
||||
/// Entry name capacity. Matches abi.maximum_process_name so a spawned task can
|
||||
/// always carry its full binary path as its name.
|
||||
pub const maximum_name = 64;
|
||||
|
||||
pub const Header = extern struct {
|
||||
magic: u32,
|
||||
@@ -19,11 +31,17 @@ pub const Header = extern struct {
|
||||
};
|
||||
|
||||
pub const Entry = extern struct {
|
||||
name: [32]u8, // NUL-padded file name (basename)
|
||||
name: [maximum_name]u8, // NUL-padded FHS path, e.g. "/system/services/init"
|
||||
offset: u64, // byte offset of the blob within the image
|
||||
len: u64, // blob length in bytes
|
||||
};
|
||||
|
||||
/// The basename of a path: the final component after the last '/'.
|
||||
pub fn basename(path: []const u8) []const u8 {
|
||||
const i = std.mem.lastIndexOfScalar(u8, path, '/') orelse return path;
|
||||
return path[i + 1 ..];
|
||||
}
|
||||
|
||||
/// A validated view over an initial_ramdisk image. `init` checks the magic and that the
|
||||
/// entry table fits; `entry` bounds-checks each blob against the image.
|
||||
pub const Reader = struct {
|
||||
@@ -48,11 +66,77 @@ pub const Reader = struct {
|
||||
if (e.offset > self.image.len or e.len > self.image.len - e.offset) return null;
|
||||
// The name is stored in the entry's fixed field; return a stable slice
|
||||
// into the image (not the value copy) up to the NUL terminator.
|
||||
const name_field = self.image[off .. off + 32];
|
||||
const name_field = self.image[off .. off + maximum_name];
|
||||
const nlen = std.mem.indexOfScalar(u8, name_field, 0) orelse name_field.len;
|
||||
return .{
|
||||
.name = name_field[0..nlen],
|
||||
.blob = self.image[@intCast(e.offset)..][0..@intCast(e.len)],
|
||||
};
|
||||
}
|
||||
|
||||
/// Look a binary up by name: an exact path match wins; otherwise a unique
|
||||
/// basename match ("fat" finds "/system/services/fat") keeps pre-path callers
|
||||
/// working. Comparisons are ASCII case-insensitive — the entries come from a
|
||||
/// FAT volume, whose name lookups are case-insensitive by definition (and
|
||||
/// whose short entries store uppercase). The returned Item's name is always
|
||||
/// the stored full path.
|
||||
pub fn find(self: Reader, name: []const u8) ?Item {
|
||||
var i: u32 = 0;
|
||||
while (i < self.count) : (i += 1) {
|
||||
const item = self.entry(i) orelse continue;
|
||||
if (std.ascii.eqlIgnoreCase(item.name, name)) return item;
|
||||
}
|
||||
i = 0;
|
||||
while (i < self.count) : (i += 1) {
|
||||
const item = self.entry(i) orelse continue;
|
||||
if (std.ascii.eqlIgnoreCase(basename(item.name), name)) return item;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
// --- tests (host) -----------------------------------------------------------
|
||||
|
||||
fn testImage(buffer: []u8, entries: []const struct { name: []const u8, blob: []const u8 }) []const u8 {
|
||||
const table_end = @sizeOf(Header) + entries.len * @sizeOf(Entry);
|
||||
var offset: usize = table_end;
|
||||
std.mem.bytesAsValue(Header, buffer[0..@sizeOf(Header)]).* = .{ .magic = magic, .count = @intCast(entries.len) };
|
||||
for (entries, 0..) |e, i| {
|
||||
var record = Entry{ .name = @splat(0), .offset = offset, .len = e.blob.len };
|
||||
@memcpy(record.name[0..e.name.len], e.name);
|
||||
std.mem.bytesAsValue(Entry, buffer[@sizeOf(Header) + i * @sizeOf(Entry) ..][0..@sizeOf(Entry)]).* = record;
|
||||
@memcpy(buffer[offset..][0..e.blob.len], e.blob);
|
||||
offset += e.blob.len;
|
||||
}
|
||||
return buffer[0..offset];
|
||||
}
|
||||
|
||||
test "find matches exact path, then unique basename; name is the stored path" {
|
||||
var buffer: [1024]u8 = undefined;
|
||||
const image = testImage(&buffer, &.{
|
||||
.{ .name = "/system/services/init", .blob = "INIT" },
|
||||
.{ .name = "/system/drivers/ps2-bus", .blob = "PS2" },
|
||||
});
|
||||
const rd = Reader.init(image).?;
|
||||
|
||||
const by_path = rd.find("/system/services/init").?;
|
||||
try std.testing.expectEqualStrings("/system/services/init", by_path.name);
|
||||
try std.testing.expectEqualStrings("INIT", by_path.blob);
|
||||
|
||||
const by_base = rd.find("ps2-bus").?;
|
||||
try std.testing.expectEqualStrings("/system/drivers/ps2-bus", by_base.name);
|
||||
try std.testing.expectEqualStrings("PS2", by_base.blob);
|
||||
|
||||
try std.testing.expect(rd.find("no-such-binary") == null);
|
||||
}
|
||||
|
||||
test "v1 magic is rejected" {
|
||||
var buffer: [64]u8 = @splat(0);
|
||||
std.mem.bytesAsValue(Header, buffer[0..@sizeOf(Header)]).* = .{ .magic = 0x444E5244, .count = 0 };
|
||||
try std.testing.expect(Reader.init(&buffer) == null);
|
||||
}
|
||||
|
||||
test "basename" {
|
||||
try std.testing.expectEqualStrings("fat", basename("/system/services/fat"));
|
||||
try std.testing.expectEqualStrings("fat", basename("fat"));
|
||||
}
|
||||
|
||||
@@ -529,8 +529,22 @@ var warp_ap_ready: u32 = 0;
|
||||
var warp_stop: u32 = 0;
|
||||
var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test)
|
||||
|
||||
const warp_rounds: u32 = 1 << 20; // locked reads on the BSP: ~1 ms at GHz rates
|
||||
const warp_spin_limit: u64 = 1 << 32; // bound every rendezvous wait so a lost core can't hang boot
|
||||
// The warp check is bounded by TIME, not iterations: a warp tick is a locked
|
||||
// read-modify-write on a cacheline two cores are fighting over — microseconds
|
||||
// under real contention, not the nanosecond an uncontended count assumes (a
|
||||
// 1<<20-round budget measured 2-21 SECONDS per core on a 16-core machine), and
|
||||
// a PAUSE costs ~140 cycles on modern Intel, so an iteration-counted await
|
||||
// mis-measures by two orders of magnitude too. ~5 ms of pairwise hammering per
|
||||
// core is plenty to catch a lagging TSC (Linux's check_tsc_warp budget), and
|
||||
// ~100 ms is a generous rendezvous window for a healthy core.
|
||||
const warp_check_ns: u64 = 5_000_000; // per-AP pairwise check duration
|
||||
const warp_await_ns: u64 = 100_000_000; // rendezvous wait before giving up
|
||||
|
||||
/// TSC ticks for `ns` nanoseconds (valid whenever the warp check runs: the TSC
|
||||
/// is the clocksource, so tsc_hz is calibrated).
|
||||
fn warpTicksFor(ns: u64) u64 {
|
||||
return @intCast(@as(u128, ns) * tsc_hz / 1_000_000_000);
|
||||
}
|
||||
|
||||
fn warpTick() void {
|
||||
while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause");
|
||||
@@ -545,11 +559,11 @@ fn warpTick() void {
|
||||
@atomicStore(u32, &warp_lock, 0, .release);
|
||||
}
|
||||
|
||||
/// Spin (bounded) until `flag` is nonzero; false on timeout.
|
||||
/// Spin (time-bounded) until `flag` is nonzero; false on timeout.
|
||||
fn warpAwait(flag: *u32) bool {
|
||||
var spins: u64 = 0;
|
||||
while (@atomicLoad(u32, flag, .acquire) == 0) : (spins += 1) {
|
||||
if (spins >= warp_spin_limit) return false;
|
||||
const deadline = rdtsc() +% warpTicksFor(warp_await_ns);
|
||||
while (@atomicLoad(u32, flag, .acquire) == 0) {
|
||||
if (rdtsc() -% deadline < (1 << 62)) return false; // past the deadline
|
||||
asm volatile ("pause");
|
||||
}
|
||||
return true;
|
||||
@@ -568,8 +582,8 @@ pub fn checkWarpSource() void {
|
||||
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||
return;
|
||||
}
|
||||
var i: u32 = 0;
|
||||
while (i < warp_rounds) : (i += 1) warpTick();
|
||||
const deadline = rdtsc() +% warpTicksFor(warp_check_ns);
|
||||
while (rdtsc() -% deadline >= (1 << 62)) warpTick(); // until the time budget is spent
|
||||
@atomicStore(u32, &warp_stop, 1, .release);
|
||||
@atomicStore(u32, &warp_bsp_ready, 0, .release);
|
||||
warp_checks += 1;
|
||||
@@ -583,9 +597,10 @@ pub fn checkWarpTarget() void {
|
||||
if (clock_source != .tsc) return;
|
||||
if (!warpAwait(&warp_bsp_ready)) return;
|
||||
@atomicStore(u32, &warp_ap_ready, 1, .release);
|
||||
var spins: u64 = 0;
|
||||
while (@atomicLoad(u32, &warp_stop, .acquire) == 0) : (spins += 1) {
|
||||
if (spins >= warp_spin_limit) return;
|
||||
// The BSP owns the budget; this bound only protects against a lost BSP.
|
||||
const deadline = rdtsc() +% warpTicksFor(2 * warp_check_ns + warp_await_ns);
|
||||
while (@atomicLoad(u32, &warp_stop, .acquire) == 0) {
|
||||
if (rdtsc() -% deadline < (1 << 62)) return; // past the deadline
|
||||
warpTick();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -106,6 +106,12 @@ pub fn serialWrite(bytes: []const u8) void {
|
||||
serial.write(bytes);
|
||||
}
|
||||
|
||||
/// Whether a working UART was detected (loopback probe). When false the serial
|
||||
/// sink is silently inert — a dead legacy COM1 costs nothing per byte.
|
||||
pub fn serialPresent() bool {
|
||||
return serial.present();
|
||||
}
|
||||
|
||||
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||
/// displays. The last-resort progress signal when there's no text output at all.
|
||||
@@ -162,10 +168,18 @@ pub fn mapUserPageInto(root: u64, virtual: u64, physical: u64, writable: bool, e
|
||||
paging.mapUserInto(root, virtual, physical, writable, executable);
|
||||
}
|
||||
|
||||
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
|
||||
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
|
||||
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
paging.mapUserDeviceInto(root, virtual, physical, len);
|
||||
/// Map a device MMIO window into address space `root`: RW+NX, and marked so teardown
|
||||
/// won't free the MMIO frames as RAM. `write_combining` picks the cache type —
|
||||
/// false = strong-uncacheable (registers), true = write-combining (a framebuffer).
|
||||
/// For IO passthrough.
|
||||
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64, write_combining: bool) void {
|
||||
paging.mapUserDeviceInto(root, virtual, physical, len, write_combining);
|
||||
}
|
||||
|
||||
/// Is the user leaf mapping `virtual` in address space `root` write-combining? Null if
|
||||
/// unmapped. For tests verifying the framebuffer map's cache type.
|
||||
pub fn userLeafIsWriteCombining(root: u64, virtual: u64) ?bool {
|
||||
return paging.leafIsWriteCombining(root, virtual);
|
||||
}
|
||||
|
||||
/// Map coherent DMA RAM into address space `root`: strong-uncacheable, RW+NX, but
|
||||
@@ -174,6 +188,13 @@ pub fn mapUserDmaInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
paging.mapUserDmaInto(root, virtual, physical, len);
|
||||
}
|
||||
|
||||
/// Map shared cacheable RAM into address space `root`: write-back cacheable, RW+NX, and
|
||||
/// marked so teardown won't free the frames (they're owned by a refcounted shared-memory object,
|
||||
/// freed when its last capability drops). For shared_memory_create/shared_memory_map.
|
||||
pub fn mapUserSharedInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
paging.mapUserSharedInto(root, virtual, physical, len);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
||||
paging.map(virtual, physical, writable);
|
||||
@@ -283,6 +304,17 @@ pub fn cpuLocal() usize {
|
||||
return pcpu.scheduler();
|
||||
}
|
||||
|
||||
const ia32_fs_base = 0xC000_0100;
|
||||
|
||||
/// Set the user-space TLS **thread pointer** — the arch-neutral name the generic scheduler
|
||||
/// calls (`architecture.setThreadPointer`). On x86_64 that is the FS-segment base
|
||||
/// (`IA32_FS_BASE`); an aarch64 port implements the same call against `TPIDR_EL0`. The
|
||||
/// kernel never touches FS, so this only affects the user task that runs next, which the
|
||||
/// scheduler restores per task across context switches (docs/threading-plan.md M10).
|
||||
pub fn setThreadPointer(base: u64) void {
|
||||
io.wrmsr(ia32_fs_base, base);
|
||||
}
|
||||
|
||||
// --- SMP: application-processor bring-up ----------------------------------
|
||||
|
||||
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
|
||||
@@ -666,6 +698,15 @@ pub fn jumpToUser(entry: u64, stack_top: u64) noreturn {
|
||||
jump_to_user(entry, stack_top);
|
||||
}
|
||||
|
||||
/// As `jumpToUser`, but delivers `arg0` in the user's `rdi` — how a fresh thread
|
||||
/// receives its closure pointer (docs/threading.md). A normal process is dropped
|
||||
/// with `arg0 = 0`, which its `_start` ignores (it reads argv off the stack).
|
||||
extern fn jump_to_user_arg(rip: u64, rsp: u64, arg0: u64) callconv(.c) noreturn;
|
||||
|
||||
pub fn jumpToUserArg(entry: u64, stack_top: u64, arg0: u64) noreturn {
|
||||
jump_to_user_arg(entry, stack_top, arg0);
|
||||
}
|
||||
|
||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||
/// return. Until set, faults just halt the core.
|
||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||
|
||||
@@ -123,6 +123,22 @@ jump_to_user:
|
||||
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||
iretq
|
||||
|
||||
# jump_to_user_arg(rdi = user rip, rsi = user rsp, rdx = user rdi/arg0): as
|
||||
# jump_to_user, but delivers arg0 in the user's rdi — how a fresh **thread**
|
||||
# receives its closure pointer (docs/threading.md). rdi carries the rip only until
|
||||
# it is pushed into the iretq frame, after which we overwrite it with the arg.
|
||||
.global jump_to_user_arg
|
||||
jump_to_user_arg:
|
||||
cli
|
||||
push $0x1B # user SS (0x18 | RPL 3)
|
||||
push %rsi # user RSP
|
||||
push $0x202 # RFLAGS: IF | reserved-1
|
||||
push $0x23 # user CS (0x20 | RPL 3)
|
||||
push %rdi # user RIP (consumes rdi)
|
||||
mov %rdx, %rdi # user rdi = arg0 (the thread's closure pointer)
|
||||
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||
iretq
|
||||
|
||||
# --- ring 3 entry/exit ------------------------------------------------------
|
||||
|
||||
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
||||
@@ -205,7 +221,17 @@ syscall_entry:
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # trap-frame pointer
|
||||
# Preserve the caller's SSE/x87 register file across the syscall — see the same
|
||||
# dance in isr_common. Without it a syscall (or a task the scheduler runs while
|
||||
# this one blocks) clobbers the caller's live XMM values, which the compiler is
|
||||
# free to hold across a syscall (its wrappers only clobber rcx/r11/memory).
|
||||
mov %rsp, %rbx
|
||||
and $-16, %rsp
|
||||
sub $512, %rsp
|
||||
fxsave (%rsp)
|
||||
call interruptDispatch
|
||||
fxrstor (%rsp)
|
||||
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
@@ -357,7 +383,21 @@ isr_common:
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
||||
# Save the interrupted SSE/x87 register file before any kernel code runs, and
|
||||
# restore it on the way out — the kernel and user both keep live values in XMM
|
||||
# (a 16-byte struct copy is a movdqu), and the kernel never otherwise preserves
|
||||
# them, so an interrupt handler (and whatever the scheduler runs in its place)
|
||||
# would silently clobber the interrupted task's vector registers. rbx bridges the
|
||||
# exact rsp across the call: it is callee-saved (interruptDispatch and every
|
||||
# context switch preserve it), so it survives even a blocking dispatch, and the
|
||||
# `and`/`sub` gives fxsave its required 16-byte-aligned scratch on the kernel stack.
|
||||
mov %rsp, %rbx
|
||||
and $-16, %rsp
|
||||
sub $512, %rsp
|
||||
fxsave (%rsp)
|
||||
call interruptDispatch
|
||||
fxrstor (%rsp)
|
||||
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
|
||||
@@ -7,8 +7,13 @@
|
||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||
//! which the kernel heap will build on.
|
||||
//!
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
//! The physmap (the permanent window onto all physical RAM) is built with 2 MiB
|
||||
//! huge pages wherever the range is 2 MiB-aligned, falling back to 4 KiB for the
|
||||
//! unaligned edges. On a big machine that is the difference between ~16.7M page-
|
||||
//! table entries (128 MiB of tables) and ~32K — it makes both the build and the
|
||||
//! footprint scale sanely with RAM. Everything else (kernel segments, heap, user
|
||||
//! space, on-demand MMIO) stays 4 KiB: precise, and the table memory is
|
||||
//! negligible there.
|
||||
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
@@ -22,10 +27,22 @@ const writable: u64 = 1 << 1;
|
||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||
const pwt: u64 = 1 << 3; // page write-through
|
||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||
const page_size_bit: u64 = 1 << 7; // PS: this PDPT/PD entry is a 1 GiB/2 MiB leaf, not a pointer to the next table
|
||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// The PAT-index bit. In a 4 KiB PTE it is bit 7; in a huge leaf (2 MiB PDE / 1 GiB
|
||||
// PDPTE) bit 7 is PS, so the PAT bit moves to bit 12. With PCD=PWT=0 this selects
|
||||
// PAT entry 4, which `setupPat` programs to write-combining (see mapRangePhysmap).
|
||||
const pte_pat: u64 = 1 << 7;
|
||||
const huge_pat: u64 = 1 << 12;
|
||||
const ia32_pat: u32 = 0x277;
|
||||
|
||||
/// The physmap's page size for 2 MiB-aligned RAM: one PD leaf covers this instead
|
||||
/// of 512 PT entries. 4 KiB pages fill the unaligned edges (see mapRangePhysmap).
|
||||
const huge_page_size: u64 = 2 << 20; // 2 MiB
|
||||
|
||||
// ELF segment flags (p_flags).
|
||||
const pf_x: u32 = 1;
|
||||
const pf_w: u32 = 2;
|
||||
@@ -74,7 +91,15 @@ fn allocTable() u64 {
|
||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||
/// writable only if every level is; non-executable if any level is).
|
||||
fn descend(entry: *u64) u64 {
|
||||
if (entry.* & present != 0) return entry.* & address_mask;
|
||||
if (entry.* & present != 0) {
|
||||
// A present-but-huge entry is a leaf, not a table: descending would read
|
||||
// its 2 MiB/1 GiB data frame as a page table and corrupt RAM. This only
|
||||
// fires on a bug — a 4 KiB map landing inside a physmap huge page — and a
|
||||
// loud panic beats silent corruption. (The physmap and the 4 KiB regions
|
||||
// live in disjoint PML4 slots, so it should never happen.)
|
||||
if (entry.* & page_size_bit != 0) @panic("paging: descend through a huge-page leaf");
|
||||
return entry.* & address_mask;
|
||||
}
|
||||
const frame = allocTable();
|
||||
entry.* = frame | present | writable;
|
||||
return frame;
|
||||
@@ -96,15 +121,39 @@ fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||
}
|
||||
|
||||
/// Map one 2 MiB huge page `virtual` -> `physical` with `flags` — a leaf at the PD
|
||||
/// level (PS bit set), with no PT beneath it. Both addresses must be 2 MiB-aligned.
|
||||
/// One of these replaces 512 `mapPage`s (and the PT frame they'd need).
|
||||
fn mapHugePage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||
@panic("paging: new higher-half PML4 entry after init");
|
||||
const pdpt = descend(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte);
|
||||
tableAt(pd)[(virtual >> 21) & 0x1FF] = (physical & address_mask) | flags | present | page_size_bit;
|
||||
}
|
||||
|
||||
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||
/// window onto physical memory once the low identity map goes away.
|
||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
||||
/// window onto physical memory once the low identity map goes away. The 2 MiB-
|
||||
/// aligned interior is mapped with huge pages; the unaligned head/tail with 4 KiB.
|
||||
/// `write_combining` selects the WC memory type (setupPat's PAT entry 4) via the
|
||||
/// PAT bit — bit 7 in a 4 KiB PTE, bit 12 in a huge leaf — for the framebuffer.
|
||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64, write_combining: bool) void {
|
||||
const pte_flags = if (write_combining) flags | pte_pat else flags;
|
||||
const huge_flags = if (write_combining) flags | huge_pat else flags;
|
||||
var address = physical_base & ~@as(u64, page_size - 1);
|
||||
const end = physical_base + len;
|
||||
while (address < end) : (address += page_size) {
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, flags);
|
||||
}
|
||||
// Head: 4 KiB pages up to the next 2 MiB boundary.
|
||||
while (address < end and address & (huge_page_size - 1) != 0) : (address += page_size)
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||
// Interior: 2 MiB huge pages while a whole one still fits.
|
||||
while (address + huge_page_size <= end) : (address += huge_page_size)
|
||||
mapHugePage(pml4, boot_handoff.physicalToVirtual(address), address, huge_flags);
|
||||
// Tail: 4 KiB pages for whatever is left.
|
||||
while (address < end) : (address += page_size)
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, pte_flags);
|
||||
}
|
||||
|
||||
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
||||
@@ -118,11 +167,26 @@ fn enableNx() void {
|
||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||
}
|
||||
|
||||
/// Program this core's PAT so entry 4 (selected by the PAT bit with PCD=PWT=0) is
|
||||
/// **write-combining**, leaving the other seven at their reset types. Nothing else
|
||||
/// in danos sets the PAT bit, so this changes no existing mapping — it only gives
|
||||
/// the framebuffer a write-combining type, which turns its full-screen clear from
|
||||
/// glacial (uncached writes to a GPU BAR, the real-hardware default via MTRRs) into
|
||||
/// a batched burst. Must run on **every** core (PAT is per-logical-processor) — the
|
||||
/// framebuffer mapping lives in the shared kernel half, so a core with the reset
|
||||
/// PAT would see it as write-back and alias. Called from `init` (BSP) and each AP.
|
||||
pub fn setupPat() void {
|
||||
// Reset PAT is PA0=WB PA1=WT PA2=UC- PA3=UC PA4=WB PA5=WT PA6=UC- PA7=UC; flip
|
||||
// PA4 from WB (0x06) to WC (0x01). Type codes: UC=0 WC=1 WT=4 WP=5 WB=6 UC-=7.
|
||||
io.wrmsr(ia32_pat, 0x0007_0401_0007_0406);
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
||||
alloc_frame = allocFrame;
|
||||
free_frame = freeFrame;
|
||||
enableNx();
|
||||
setupPat(); // BSP: PAT entry 4 = write-combining, for the framebuffer window
|
||||
const pml4 = allocTable();
|
||||
|
||||
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||
@@ -130,13 +194,15 @@ pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot
|
||||
// mapped on demand (mapMmio) or explicitly below.
|
||||
for (regions(boot_information.memory_map)) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute, false);
|
||||
}
|
||||
|
||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||
// the kernel touches directly), RW + NX.
|
||||
// the kernel touches directly), RW + NX. The framebuffer is **write-
|
||||
// combining** (see setupPat) so the console's full-screen clear is a burst,
|
||||
// not millions of uncached single-word writes.
|
||||
const fb = boot_information.framebuffer;
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute, true);
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||
@@ -254,8 +320,12 @@ pub fn mapUserInto(pml4: u64, virtual: u64, physical: u64, writable_page: bool,
|
||||
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
|
||||
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
|
||||
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
|
||||
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
|
||||
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64, write_combining: bool) void {
|
||||
// Registers are strong-uncacheable (PCD|PWT). A framebuffer instead wants
|
||||
// write-combining — the PAT bit (bit 7 in a 4 KiB PTE) with PCD=PWT=0 selects PAT
|
||||
// entry 4, which `setupPat` programs to WC — so pixel writes batch into bursts.
|
||||
const cache: u64 = if (write_combining) pte_pat else (pcd | pwt);
|
||||
const flags: u64 = present | user | writable | no_execute | device_grant | cache;
|
||||
const first = physical & ~@as(u64, page_size - 1);
|
||||
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||
var off: u64 = 0;
|
||||
@@ -296,6 +366,57 @@ pub fn mapUserDmaInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// The raw leaf entry mapping `virtual` in the address space rooted at `pml4`, or null
|
||||
/// if any level of the walk is absent. **Read-only** — never allocates or descends into
|
||||
/// a missing table (unlike the `map*` paths' `descendUser`). Stops at the first huge
|
||||
/// leaf. For tests and introspection that need a page's actual flag bits.
|
||||
pub fn leafEntryOf(pml4: u64, virtual: u64) ?u64 {
|
||||
const l4 = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (l4 & present == 0) return null;
|
||||
const l3 = tableAt(l4 & address_mask)[(virtual >> 30) & 0x1FF];
|
||||
if (l3 & present == 0) return null;
|
||||
if (l3 & page_size_bit != 0) return l3; // 1 GiB leaf
|
||||
const l2 = tableAt(l3 & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (l2 & present == 0) return null;
|
||||
if (l2 & page_size_bit != 0) return l2; // 2 MiB leaf
|
||||
const l1 = tableAt(l2 & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (l1 & present == 0) return null;
|
||||
return l1;
|
||||
}
|
||||
|
||||
/// Is the 4 KiB leaf mapping `virtual` write-combining — the PAT bit set with PCD and
|
||||
/// PWT clear, which `setupPat` makes PAT entry 4 (WC)? Null if unmapped. The device
|
||||
/// mapping path (`mapUserDeviceInto`) always uses 4 KiB leaves, so bit 7 (`pte_pat`)
|
||||
/// is the PAT selector in play.
|
||||
pub fn leafIsWriteCombining(pml4: u64, virtual: u64) ?bool {
|
||||
const e = leafEntryOf(pml4, virtual) orelse return null;
|
||||
return (e & pte_pat != 0) and (e & pcd == 0) and (e & pwt == 0);
|
||||
}
|
||||
|
||||
/// Map `[physical, physical+len)` into the user half rooted at `pml4` as **shared cacheable
|
||||
/// RAM**: write-back cacheable (RW + NX) for CPU compositing, and carrying `device_grant`
|
||||
/// so teardown (`freeSubtree`) does **not** return the frames to the allocator. The frames
|
||||
/// are owned by a refcounted shared-memory object (system/kernel/ipc-synchronous.zig) and
|
||||
/// freed only when its last capability drops — not when one sharer's address space dies, or
|
||||
/// the other sharers would be left mapping freed RAM. The caller aligns `virtual`/`physical`.
|
||||
pub fn mapUserSharedInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
const flags: u64 = present | user | writable | no_execute | device_grant; // WB cacheable
|
||||
const first = physical & ~@as(u64, page_size - 1);
|
||||
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||
var off: u64 = 0;
|
||||
while (first + off <= last) : (off += page_size) {
|
||||
const v = virtual + off;
|
||||
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
|
||||
const pdpt = descendUser(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
|
||||
const pd = descendUser(pdpte);
|
||||
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
|
||||
const pt = descendUser(pde);
|
||||
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
|
||||
invalidate(v);
|
||||
}
|
||||
}
|
||||
|
||||
/// Create a new address space: a fresh PML4 with an empty user half and the
|
||||
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
||||
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
||||
@@ -338,8 +459,8 @@ fn freeSubtree(physical: u64, level: u32) void {
|
||||
}
|
||||
|
||||
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
||||
/// clear. Walks the 4-level tables, stopping at a 2 MiB huge-page leaf (the physmap
|
||||
/// uses them). Returns false if unmapped. Used for W^X checks in tests.
|
||||
pub fn isExecutable(virtual: u64) bool {
|
||||
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return false;
|
||||
@@ -347,6 +468,7 @@ pub fn isExecutable(virtual: u64) bool {
|
||||
if (pdpte & present == 0) return false;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return false;
|
||||
if (pde & page_size_bit != 0) return pde & no_execute == 0; // 2 MiB huge leaf
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return false;
|
||||
return pte & no_execute == 0;
|
||||
@@ -385,9 +507,9 @@ pub fn unmapInto(pml4: u64, virtual: u64) void {
|
||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
||||
/// needs the frame behind a user vaddr to free it).
|
||||
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
||||
/// resolving the offset within it. The foundation for cross-address-space copies
|
||||
/// and for munmap (which needs the frame behind a user virtual_address to free it).
|
||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return null;
|
||||
@@ -395,6 +517,8 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
if (pdpte & present == 0) return null;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return null;
|
||||
if (pde & page_size_bit != 0) // 2 MiB huge leaf: frame base is bits 51:21
|
||||
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return null;
|
||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||
|
||||
@@ -17,6 +17,12 @@ const Access = enum { port, mmio };
|
||||
var access: Access = .port;
|
||||
var base: u64 = 0x3F8; // COM1
|
||||
|
||||
/// Whether `init`/`reconfigure` found a *working* UART at `base`. False on a
|
||||
/// legacy-free machine whose COM1 is decoded but dead: writing to it is then a
|
||||
/// no-op, so `write` never spins waiting for a transmit register that will never
|
||||
/// drain. Cleared until proven by the loopback probe.
|
||||
var uart_present: bool = false;
|
||||
|
||||
fn portOut(p: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[p]"
|
||||
:
|
||||
@@ -57,6 +63,34 @@ pub fn init() void {
|
||||
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||
setRegister(4, 0x0B); // RTS/DSR set
|
||||
uart_present = probe();
|
||||
}
|
||||
|
||||
/// Detect a *working* UART by internal loopback: route the transmitter back to
|
||||
/// the receiver (MCR bit 4), send a byte, and check it comes back. A port that is
|
||||
/// merely decoded but has nothing behind it (the common case on a legacy-free
|
||||
/// board that still answers I/O at 0x3F8) never echoes, so this returns false.
|
||||
///
|
||||
/// This matters for speed, not just correctness: a dead UART's line-status
|
||||
/// register reads back 0x00, so its transmit-holding-empty bit never sets, and
|
||||
/// `writeByte` would otherwise spin its full guard — tens of milliseconds — on
|
||||
/// *every* logged byte. On real hardware that alone can add ~a minute to boot.
|
||||
fn probe() bool {
|
||||
const saved_mcr = register(4);
|
||||
setRegister(4, 0x1E); // MCR: LOOP | OUT2 | OUT1 | RTS — internal loopback
|
||||
setRegister(0, 0xAE); // push a distinctive byte into the loopback path
|
||||
var guard: u32 = 0;
|
||||
while (register(5) & 0x01 == 0 and guard < 10_000) : (guard += 1) {} // await Data Ready
|
||||
const echo = register(0);
|
||||
setRegister(4, saved_mcr); // restore the modem-control lines
|
||||
return echo == 0xAE;
|
||||
}
|
||||
|
||||
/// Whether a working UART was detected (see `probe`). The log sink stays
|
||||
/// registered regardless — it simply does nothing until this is true — so a UART
|
||||
/// that only `reconfigure` discovers (via SPCR) still starts logging.
|
||||
pub fn present() bool {
|
||||
return uart_present;
|
||||
}
|
||||
|
||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||
@@ -70,15 +104,19 @@ pub fn reconfigure(is_mmio: bool, address: u64) void {
|
||||
}
|
||||
|
||||
fn writeByte(c: u8) void {
|
||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
||||
// Wait for the transmit-holding register to empty. `write` only reaches here
|
||||
// for a UART the loopback probe proved live, so this bounds a momentary stall
|
||||
// (e.g. deasserted flow control), not an absent port: ~5000 legacy-port reads
|
||||
// is a few ms — comfortably longer than one 38400-baud byte-time (~260 µs).
|
||||
var guard: u32 = 0;
|
||||
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
||||
while (register(5) & 0x20 == 0 and guard < 5_000) : (guard += 1) {}
|
||||
setRegister(0, c);
|
||||
}
|
||||
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up. A no-op
|
||||
/// when no working UART was detected, so a dead COM1 costs nothing per byte.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
if (!uart_present) return;
|
||||
for (bytes) |c| {
|
||||
if (c == '\n') writeByte('\r');
|
||||
writeByte(c);
|
||||
|
||||
@@ -173,6 +173,7 @@ fn delayMicros(us: u64) void {
|
||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
const cpu = boot_index;
|
||||
paging.setupPat(); // this core's PAT: entry 4 = write-combining, to match the BSP
|
||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||
idt.loadOnThisCpu(); // the shared IDT
|
||||
|
||||
+31
-19
@@ -2,12 +2,15 @@
|
||||
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
||||
//! — just pixels.
|
||||
//!
|
||||
//! This is a **bootstrap** console — a stop-gap so early boot has something on
|
||||
//! screen. The framebuffer is a general graphics surface, *not* inherently a text
|
||||
//! terminal; once the driver machinery exists it becomes a proper graphics device
|
||||
//! driver and this text-grid crutch goes away. It is therefore kept **separate
|
||||
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file,
|
||||
//! while this only paints the handful of user-facing status lines and panics.
|
||||
//! This is a **bootstrap / fatal-fallback** console. The driver machinery now exists — the
|
||||
//! user-space **display service** ([../services/display](../services/display/display.zig),
|
||||
//! docs/display.md) owns the framebuffer in normal operation — so this no longer paints
|
||||
//! routine status. It exists for the two cases the display service can't cover: **early
|
||||
//! boot**, before the service has claimed the framebuffer, and **fatal errors** (a kernel
|
||||
//! panic or a kernel-mode fault), which force it back on (`setSuppressed`) so a dying
|
||||
//! machine's last words reach the screen even over a live display. It is kept **separate
|
||||
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file and
|
||||
//! carries all routine kernel output; this only paints those fatal cases.
|
||||
//!
|
||||
//! The module owns a single console and a `present` flag; `write` is a no-op when
|
||||
//! the firmware handed over no framebuffer (a headless machine), so the kernel
|
||||
@@ -20,6 +23,12 @@ const boot_handoff = @import("boot-handoff");
|
||||
var con: Console = undefined;
|
||||
var con_present: bool = false;
|
||||
|
||||
/// Set while a user-space display service owns the framebuffer: `write` falls silent so
|
||||
/// the kernel doesn't paint over the compositor. Driven by the display device's
|
||||
/// claim/release (system/kernel/process.zig). The terminal panic/exception paths clear
|
||||
/// it first (`setSuppressed(false)`) — a dying machine's message wins over any display.
|
||||
var suppressed: bool = false;
|
||||
|
||||
/// Set up the console over `fb`, or mark it absent if there's no usable
|
||||
/// framebuffer. Clears the screen when present.
|
||||
pub fn init(fb: boot_handoff.Framebuffer) void {
|
||||
@@ -41,13 +50,20 @@ pub fn present() bool {
|
||||
return con_present;
|
||||
}
|
||||
|
||||
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present,
|
||||
/// so it's always safe to call.
|
||||
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present, or
|
||||
/// while a display service owns the screen (`suppressed`), so it's always safe to call.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
if (!con_present) return;
|
||||
if (!con_present or suppressed) return;
|
||||
for (bytes) |c| con.putChar(c);
|
||||
}
|
||||
|
||||
/// Quiesce (or resume) the bootstrap console. Set true when a display service claims the
|
||||
/// framebuffer; set false when that claim is released, or by the panic path to force a
|
||||
/// last message onto a screen a (now-irrelevant) service was holding.
|
||||
pub fn setSuppressed(value: bool) void {
|
||||
suppressed = value;
|
||||
}
|
||||
|
||||
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
|
||||
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
|
||||
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
|
||||
@@ -122,14 +138,16 @@ pub const Console = struct {
|
||||
}
|
||||
}
|
||||
|
||||
/// Shift the visible text up one glyph row and clear the freed bottom row,
|
||||
/// leaving the cursor on that now-blank last line.
|
||||
/// The screen is full: start a fresh page at the top. NEVER scroll by
|
||||
/// copying pixel rows — that READS the framebuffer, and VRAM reads are
|
||||
/// uncached-slow on real hardware (measured: 16-core bring-up took ~90 s
|
||||
/// purely from boot lines each paying a whole-screen scroll copy). A page
|
||||
/// clear is writes only, and only once per screenful.
|
||||
fn scroll(self: *Console) void {
|
||||
const visible = self.rows * glyph_h;
|
||||
var y: u32 = 0;
|
||||
while (y + glyph_h < visible) : (y += 1) self.copyRow(y, y + glyph_h);
|
||||
while (y < visible) : (y += 1) self.fillRow(y, self.bg);
|
||||
self.row = self.rows - 1;
|
||||
self.row = 0;
|
||||
}
|
||||
|
||||
inline fn rowPtr(self: *Console, y: u32) [*]volatile u32 {
|
||||
@@ -147,10 +165,4 @@ pub const Console = struct {
|
||||
while (x < self.fb.width) : (x += 1) row[x] = color;
|
||||
}
|
||||
|
||||
fn copyRow(self: *Console, destination_y: u32, source_y: u32) void {
|
||||
const destination = self.rowPtr(destination_y);
|
||||
const source = self.rowPtr(source_y);
|
||||
var x: u32 = 0;
|
||||
while (x < self.fb.width) : (x += 1) destination[x] = source[x];
|
||||
}
|
||||
};
|
||||
|
||||
@@ -36,6 +36,11 @@ var devices: [maximum_devices]device_abi.DeviceDescriptor = undefined;
|
||||
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
||||
var count: usize = 0;
|
||||
|
||||
/// The id of the seeded framebuffer node (`seedDisplay`), or null when the machine
|
||||
/// handed over no framebuffer. Lets the process layer recognise the display claim
|
||||
/// (to quiesce the bootstrap console) without threading the id through every caller.
|
||||
var display_device: ?u64 = null;
|
||||
|
||||
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
||||
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
||||
/// which would otherwise be an entirely silent failure. Logged at boot.
|
||||
@@ -45,10 +50,55 @@ pub var dropped: usize = 0;
|
||||
pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||
count = 0;
|
||||
dropped = 0;
|
||||
display_device = null;
|
||||
for (&claimed) |*c| c.* = null;
|
||||
walk(device_tree.root, device_abi.no_parent);
|
||||
}
|
||||
|
||||
/// Publish the loader's framebuffer as a `display` device — a root-level node with one
|
||||
/// write-combining `memory` resource over the linear framebuffer and its geometry in
|
||||
/// `.display`. The framebuffer is *not* firmware-discovered (it rides the
|
||||
/// [[boot-handoff]], not the device tree), so it is seeded explicitly, after `init`.
|
||||
/// Returns the new device id, or null when there is no framebuffer (headless) or the
|
||||
/// table is full. Idempotent-ish: only ever call once per boot.
|
||||
pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32) ?u64 {
|
||||
if (base == 0 or width == 0 or height == 0) return null; // headless
|
||||
if (count >= maximum_devices) {
|
||||
dropped += 1;
|
||||
return null;
|
||||
}
|
||||
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
d.id = count;
|
||||
d.parent = device_abi.no_parent;
|
||||
d.class = @intFromEnum(device_abi.DeviceClass.display);
|
||||
d.pci_class = device_abi.no_pci_class;
|
||||
d.resource_count = 1;
|
||||
d.resources[0] = .{
|
||||
.kind = @intFromEnum(device_abi.ResourceKind.memory),
|
||||
.start = base,
|
||||
.len = @as(u64, height) * pitch,
|
||||
.flags = device_abi.resource_flag_write_combining,
|
||||
};
|
||||
d.display = .{ .width = width, .height = height, .pitch = pitch, .format = format, .refresh_hz = refresh_hz };
|
||||
devices[count] = d;
|
||||
display_device = d.id;
|
||||
count += 1;
|
||||
return d.id;
|
||||
}
|
||||
|
||||
/// The id of the seeded framebuffer device, or null when none was seeded.
|
||||
pub fn displayDevice() ?u64 {
|
||||
return display_device;
|
||||
}
|
||||
|
||||
/// Whether the framebuffer device is currently claimed by some process. The bootstrap
|
||||
/// console uses this (via the process layer) to fall silent while a display service
|
||||
/// owns the screen, and to resume if that service dies and its claim is released.
|
||||
pub fn displayClaimed() bool {
|
||||
const id = display_device orelse return false;
|
||||
return ownerOf(id) != null;
|
||||
}
|
||||
|
||||
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
||||
/// assigned it down to its children as their parent.
|
||||
fn walk(node: *platform.Device, parent_id: u64) void {
|
||||
|
||||
@@ -28,6 +28,7 @@ const architecture = @import("architecture");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
const Task = scheduler.Task;
|
||||
@@ -89,6 +90,10 @@ const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||
pub const Endpoint = struct {
|
||||
refcount: u32 = 1,
|
||||
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
|
||||
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
|
||||
owner: u32 = 0,
|
||||
dead: bool = false,
|
||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||
sender_head: ?*Task = null,
|
||||
@@ -109,10 +114,30 @@ pub const Endpoint = struct {
|
||||
|
||||
pub fn createIpcEndpoint() ?*Endpoint {
|
||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||
endpoint.* = .{};
|
||||
endpoint.* = .{ .owner = scheduler.currentId() };
|
||||
return endpoint;
|
||||
}
|
||||
|
||||
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
|
||||
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
|
||||
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
|
||||
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
|
||||
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
|
||||
pub fn killOwnedEndpointsLocked(task_id: u32) void {
|
||||
for (®istry) |*slot| {
|
||||
const endpoint = slot.* orelse continue;
|
||||
if (endpoint.owner != task_id) continue;
|
||||
endpoint.dead = true;
|
||||
while (dequeueSender(endpoint)) |sender| {
|
||||
sender.ipc_status = -EPEER;
|
||||
sender.ipc_received_cap = abi.no_cap;
|
||||
scheduler.readyLocked(sender);
|
||||
}
|
||||
slot.* = null;
|
||||
dropRef(endpoint);
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||
pub fn dropRef(endpoint: *Endpoint) void {
|
||||
@@ -123,6 +148,43 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
||||
}
|
||||
}
|
||||
|
||||
// --- capability objects: what a handle-table entry can name ------------------
|
||||
|
||||
/// The `kind` tag on a `scheduler.HandleObject` — which capability object a handle names.
|
||||
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
|
||||
pub const handle_kind_endpoint: u8 = 0;
|
||||
pub const handle_kind_shared_memory: u8 = 1;
|
||||
|
||||
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
|
||||
/// capability handles across processes and freed when the last one drops. `phys` is its
|
||||
/// contiguous physical base, `pages` its length. A sharer's address-space teardown never
|
||||
/// reclaims these frames (the mapping carries `device_grant`); this object owns them.
|
||||
pub const SharedMemoryObject = struct {
|
||||
refcount: u32 = 1,
|
||||
phys: u64,
|
||||
pages: usize,
|
||||
};
|
||||
|
||||
/// Wrap `pages` contiguous frames at `phys` (already allocated + zeroed by the caller) in a
|
||||
/// refcounted shared-memory object, or null if the heap is out of room.
|
||||
pub fn createSharedMemory(phys: u64, pages: usize) ?*SharedMemoryObject {
|
||||
const shared_memory = heap.allocator().create(SharedMemoryObject) catch return null;
|
||||
shared_memory.* = .{ .phys = phys, .pages = pages };
|
||||
return shared_memory;
|
||||
}
|
||||
|
||||
/// Drop a shared-memory reference; when the last one goes, return its frames to the
|
||||
/// allocator and free the object. (The mappings themselves are torn down with each
|
||||
/// sharer's address space; `device_grant` keeps that from freeing the frames early.)
|
||||
pub fn dropSharedMemoryReference(shared_memory: *SharedMemoryObject) void {
|
||||
if (shared_memory.refcount > 1) {
|
||||
shared_memory.refcount -= 1;
|
||||
} else {
|
||||
for (0..shared_memory.pages) |i| pmm.free(shared_memory.phys + i * page_size);
|
||||
heap.allocator().destroy(shared_memory);
|
||||
}
|
||||
}
|
||||
|
||||
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
||||
|
||||
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
||||
@@ -222,11 +284,25 @@ pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||
/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when
|
||||
/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message.
|
||||
fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
||||
const endpoint = resolveHandle(from, cap) orelse return -EBADF;
|
||||
endpoint.refcount += 1;
|
||||
const handle = installHandle(to, endpoint);
|
||||
if (cap >= from.handles.len) return -EBADF;
|
||||
const entry = from.handles[@intCast(cap)] orelse return -EBADF;
|
||||
// Bump the named object's refcount (a copy, not a move — the sender keeps its handle),
|
||||
// dispatching by kind so both endpoints and shared-memory regions can travel with a
|
||||
// message.
|
||||
switch (entry.kind) {
|
||||
handle_kind_endpoint => {
|
||||
const e: *Endpoint = @ptrCast(@alignCast(entry.ptr));
|
||||
e.refcount += 1;
|
||||
},
|
||||
handle_kind_shared_memory => {
|
||||
const s: *SharedMemoryObject = @ptrCast(@alignCast(entry.ptr));
|
||||
s.refcount += 1;
|
||||
},
|
||||
else => return -EBADF,
|
||||
}
|
||||
const handle = installEntry(to, entry);
|
||||
if (handle < 0) {
|
||||
dropRef(endpoint); // undo the bump; the receiver had no room
|
||||
dropEntry(entry); // undo the bump; the receiver had no room
|
||||
return -ENOSPC;
|
||||
}
|
||||
return handle;
|
||||
@@ -241,6 +317,7 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr:
|
||||
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (endpoint.dead) return -EPEER; // the service that owned this endpoint is gone — don't block
|
||||
|
||||
const me = scheduler.current();
|
||||
me.ipc_send_ptr = message_ptr;
|
||||
@@ -278,7 +355,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
||||
me.ipc_client = null;
|
||||
const n = @min(reply_len, client.ipc_reply_cap);
|
||||
client.ipc_received_cap = abi.no_cap;
|
||||
if (!copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
|
||||
if (!copyAcross(me.address_space, reply_ptr, client.address_space, client.ipc_reply_ptr, n)) {
|
||||
client.ipc_status = -EFAULT;
|
||||
} else if (send_cap != abi.no_cap) {
|
||||
// Transfer the reply's capability into the client. A failure fails the
|
||||
@@ -306,8 +383,8 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
||||
}
|
||||
if (popPost(endpoint)) |slot| {
|
||||
const n = @min(@as(usize, slot.length), receive_cap);
|
||||
// Copy from the kernel-resident ring slot (source aspace 0) into the receiver.
|
||||
if (!copyAcross(0, @intFromPtr(&slot.bytes), me.aspace, receive_ptr, n)) {
|
||||
// Copy from the kernel-resident ring slot (source address_space 0) into the receiver.
|
||||
if (!copyAcross(0, @intFromPtr(&slot.bytes), me.address_space, receive_ptr, n)) {
|
||||
continue; // bad receive buffer: drop this message, keep serving
|
||||
}
|
||||
out_badge.* = slot.sender_id | notify_badge_bit | notify_message_bit;
|
||||
@@ -315,7 +392,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
||||
}
|
||||
if (dequeueSender(endpoint)) |caller| {
|
||||
const n = @min(caller.ipc_send_len, receive_cap);
|
||||
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) {
|
||||
if (!copyAcross(caller.address_space, caller.ipc_send_ptr, me.address_space, receive_ptr, n)) {
|
||||
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
||||
scheduler.readyLocked(caller);
|
||||
continue;
|
||||
@@ -416,36 +493,82 @@ pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
|
||||
|
||||
// --- per-process handle table + name registry -------------------------------
|
||||
|
||||
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
|
||||
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
||||
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||
/// Install a capability object (kind + pointer) in task `t`'s handle table; returns the
|
||||
/// small-int handle or -ENOSPC. The caller has already taken/holds the reference the slot
|
||||
/// represents.
|
||||
fn installEntry(t: *Task, entry: scheduler.HandleObject) i64 {
|
||||
for (&t.handles, 0..) |*slot, i| {
|
||||
if (slot.* == null) {
|
||||
slot.* = @ptrCast(endpoint);
|
||||
slot.* = entry;
|
||||
return @intCast(i);
|
||||
}
|
||||
}
|
||||
return -ENOSPC;
|
||||
}
|
||||
|
||||
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
if (h >= t.handles.len) return null;
|
||||
const slot = t.handles[@intCast(h)] orelse return null;
|
||||
return @ptrCast(@alignCast(slot));
|
||||
/// Install an endpoint handle. The common case; keeps the endpoint callers' signature.
|
||||
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_endpoint, .ptr = @ptrCast(endpoint) });
|
||||
}
|
||||
|
||||
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
||||
/// exit path so a dead server's endpoints don't linger referenced.
|
||||
/// Install a shared-memory handle.
|
||||
pub fn installSharedMemoryHandle(t: *Task, shared_memory: *SharedMemoryObject) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_shared_memory, .ptr = @ptrCast(shared_memory) });
|
||||
}
|
||||
|
||||
/// Install an endpoint handle, reusing an existing slot that already names this
|
||||
/// endpoint (no new reference taken in that case). For callers that install per
|
||||
/// operation — fs_resolve — so a 16-slot table can't be exhausted by repeats.
|
||||
/// Any subsystem installing handles per-call should come through here.
|
||||
pub fn installHandleDeduped(t: *Task, endpoint: *Endpoint) i64 {
|
||||
for (t.handles, 0..) |slot, i| {
|
||||
const entry = slot orelse continue;
|
||||
if (entry.kind == handle_kind_endpoint and entry.ptr == @as(*anyopaque, @ptrCast(endpoint))) return @intCast(i);
|
||||
}
|
||||
const h = installHandle(t, endpoint);
|
||||
if (h >= 0) endpoint.refcount += 1; // the table entry owns a reference
|
||||
return h;
|
||||
}
|
||||
|
||||
/// Resolve a handle to its endpoint, or null if out of range, unused, or a different kind
|
||||
/// (e.g. a shared-memory handle used where an endpoint is expected).
|
||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
if (entry.kind != handle_kind_endpoint) return null;
|
||||
return @ptrCast(@alignCast(entry.ptr));
|
||||
}
|
||||
|
||||
/// Resolve a handle to its shared-memory object, or null if out of range, unused, or not
|
||||
/// a shared-memory handle.
|
||||
pub fn resolveSharedMemory(t: *Task, h: u64) ?*SharedMemoryObject {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
if (entry.kind != handle_kind_shared_memory) return null;
|
||||
return @ptrCast(@alignCast(entry.ptr));
|
||||
}
|
||||
|
||||
/// Drop every capability reference an exiting task holds, dispatching by kind so a dead
|
||||
/// task's endpoints *and* shared-memory regions are released correctly. Called from the
|
||||
/// scheduler exit path.
|
||||
pub fn closeHandles(t: *Task) void {
|
||||
for (&t.handles) |*slot| {
|
||||
if (slot.*) |p| {
|
||||
dropRef(@ptrCast(@alignCast(p)));
|
||||
if (slot.*) |entry| {
|
||||
dropEntry(entry);
|
||||
slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop the reference a handle-table entry represents, by kind.
|
||||
fn dropEntry(entry: scheduler.HandleObject) void {
|
||||
switch (entry.kind) {
|
||||
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_shared_memory => dropSharedMemoryReference(@ptrCast(@alignCast(entry.ptr))),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
||||
|
||||
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||
|
||||
+98
-44
@@ -9,6 +9,7 @@ const wall_clock = @import("wall-clock.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const process = @import("process.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
@@ -60,20 +61,34 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||
// with port-0x80 checkpoints as the only progress signal.
|
||||
architecture.serialInit();
|
||||
log.addSink(architecture.serialWrite);
|
||||
//
|
||||
// Serial is compiled in only under -Dserial (build.zig): a real machine often
|
||||
// has no live legacy COM1, and the log survives in the RAM buffer (below) and
|
||||
// is flushed to disk — so serial is now a QEMU/dev convenience the flashable
|
||||
// image leaves out. When it *is* built in, `serialInit`'s loopback probe still
|
||||
// guards against a dead port (so a -Dserial image is safe on real hardware).
|
||||
if (build_options.serial) {
|
||||
architecture.serialInit();
|
||||
log.addSink(architecture.serialWrite);
|
||||
}
|
||||
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||
// Retain the whole stream in a RAM buffer too, so a user program can later
|
||||
// read it back (klog_read) and persist the boot log to disk — the only way to
|
||||
// see it on a headless/real machine with no host capturing serial.
|
||||
log.addSink(log.ramSink);
|
||||
// (Retention is the tagged ring inside log.zig — not a sink.)
|
||||
|
||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||
//
|
||||
// The console is brought up *after* paging (below), not here: its one-time
|
||||
// full-screen clear then runs on the kernel's **write-combining** mapping of the
|
||||
// framebuffer instead of the loader's uncached one — a fast burst rather than
|
||||
// millions of uncached writes on real hardware. Until then, on-screen output is
|
||||
// absent (an early panic still lands in the serial/RAM log); the trade is worth
|
||||
// a near-instant boot. `console.write` is a safe no-op while the console is down.
|
||||
const fb = boot_information.framebuffer;
|
||||
console.init(fb);
|
||||
|
||||
log.checkpoint(cp_entry);
|
||||
|
||||
@@ -83,10 +98,10 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
architecture.init();
|
||||
|
||||
status("/system/kernel: initialising kernel...\n");
|
||||
log.write(if (console.present())
|
||||
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
if (build_options.serial) log.write(if (architecture.serialPresent())
|
||||
"/system/kernel: serial console online (COM1)\n"
|
||||
else
|
||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
"/system/kernel: no serial UART (COM1 absent) -> log kept in RAM/debugcon\n");
|
||||
log.write("/system/kernel: cpu tables online (GDT, IDT, TSS)\n");
|
||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||
@@ -142,6 +157,23 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||
|
||||
// Now on our own tables, the framebuffer window is write-combining: bring up the
|
||||
// on-screen console and clear it to a blank canvas (a fast burst here, not the loader's
|
||||
// uncached crawl). Routine boot output goes only to the log; this console now exists for
|
||||
// early-boot and fatal (`fatal`/panic) output, until the display service takes over.
|
||||
console.init(fb);
|
||||
// The console joins the log sinks: the boot transcript — kernel AND
|
||||
// userspace lines, each timestamped by the renderer — shows on screen
|
||||
// until the display service claims the framebuffer (which flips the
|
||||
// console's `suppressed` and silences this sink). On a machine with no
|
||||
// serial this is the only live view of the boot, and a slow boot becomes
|
||||
// diagnosable by eye: the timeline is right there.
|
||||
if (console.present()) log.addSink(console.write);
|
||||
log.write(if (console.present())
|
||||
"/system/kernel: framebuffer ready (early-boot + fatal fallback; the display service drives it in normal operation)\n"
|
||||
else
|
||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
|
||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||
heap.init();
|
||||
log.checkpoint(cp_heap);
|
||||
@@ -172,25 +204,24 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
log.print("/system/kernel: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
||||
}
|
||||
|
||||
// Publish the loader's framebuffer as a claimable `display` device, so a
|
||||
// user-space display service can take it over the same claim + mmio_map path as
|
||||
// any other hardware (it is not firmware-discovered; it rides the boot handoff).
|
||||
if (devices_broker.seedDisplay(fb.base, fb.width, fb.height, fb.pitch, @intFromEnum(fb.format), fb.refresh_hz)) |display_id| {
|
||||
log.print("/system/kernel: framebuffer device {d} seeded ({d}x{d}, pitch {d}, {d} Hz, write-combining)\n", .{ display_id, fb.width, fb.height, fb.pitch, fb.refresh_hz });
|
||||
}
|
||||
|
||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||
irq.init();
|
||||
|
||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
||||
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||
const pw = platform.powerInformation();
|
||||
log.write("/system/kernel: power\n");
|
||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||
if (pw.s5) |s| {
|
||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
||||
} else {
|
||||
log.write(" S5 slp_typ : (not found)\n");
|
||||
}
|
||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||
|
||||
// AML namespace parse integrity: consumed should equal total.
|
||||
const am = platform.amlStats();
|
||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
||||
|
||||
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||
@@ -305,20 +336,16 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// service supervisor and the device manager spawns the drivers it discovers.
|
||||
publishInitialRamdisk(boot_information);
|
||||
|
||||
// Hand over to user space: load /system/services/init (read off the boot volume by
|
||||
// the loader) and spawn it as a real ring-3 process, PID 1. As the supervisor it
|
||||
// brings up the system services (the VFS server, the device manager); the device
|
||||
// manager then discovers the hardware and spawns each driver. init runs on its own
|
||||
// address space, preemptively — this boot context becomes the BSP's idle loop.
|
||||
if (boot_information.init_len != 0) {
|
||||
status("/system/kernel: starting /system/services/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
process.spawnProcess(image, 4, &.{"/system/services/init"}) catch |err| {
|
||||
statusPrint("/system/kernel: /system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
} else {
|
||||
status("no /system/services/init on the boot volume.\n");
|
||||
}
|
||||
// Hand over to user space: spawn /system/services/init out of the ramdisk as a
|
||||
// real ring-3 process, PID 1 — it rides the same table as every other binary.
|
||||
// As the supervisor it brings up the system services (the VFS server, the device
|
||||
// manager); the device manager then discovers the hardware and spawns each
|
||||
// driver. init runs on its own address space, preemptively — this boot context
|
||||
// becomes the BSP's idle loop.
|
||||
status("/system/kernel: starting /system/services/init...\n");
|
||||
process.spawnBundled("/system/services/init") catch |err| {
|
||||
statusPrint("/system/kernel: /system/services/init failed to start: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
|
||||
// Become the idle task: drop below every real task and halt until an
|
||||
// interrupt. The timer keeps preempting into init and any other work.
|
||||
@@ -392,11 +419,22 @@ fn bringUpSecondaries() void {
|
||||
log.print("/system/kernel: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||
}
|
||||
|
||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
||||
/// touches the framebuffer.
|
||||
/// A user-facing status line. Now that the user-space **display service** owns the
|
||||
/// framebuffer in normal operation (docs/display.md), routine kernel output goes to the
|
||||
/// diagnostic `log` (serial/debugcon/RAM) *only* — never to the on-screen console, which
|
||||
/// the compositor is about to paint over. For a message that must reach the screen even so
|
||||
/// — a panic or a fatal fault, when the machine is going down — use `fatal`.
|
||||
fn status(message: []const u8) void {
|
||||
log.write(message);
|
||||
}
|
||||
|
||||
/// A fatal, user-facing message: to the diagnostic log *and* the on-screen console, forcing
|
||||
/// the console back on (`setSuppressed(false)`) first — a dying machine's last words outrank
|
||||
/// any display service holding the framebuffer. The console is otherwise silent in normal
|
||||
/// operation (see `status`); it exists now only for early-boot and fatal output.
|
||||
fn fatal(message: []const u8) void {
|
||||
log.appendPanic(message); // bounded lock wait: a panic never deadlocks on the log
|
||||
console.setSuppressed(false);
|
||||
console.write(message);
|
||||
}
|
||||
|
||||
@@ -405,6 +443,11 @@ fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
||||
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
fn fatalPrint(comptime fmt: []const u8, args: anytype) void {
|
||||
var buffer: [256]u8 = undefined;
|
||||
fatal(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// Frames (4 KiB pages) to whole MiB.
|
||||
fn mib(pages: u64) u64 {
|
||||
return pages * abi.page_size / (1024 * 1024);
|
||||
@@ -465,16 +508,25 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
||||
|
||||
log.checkpoint(cp_exception);
|
||||
const core = scheduler.currentCpuIndex();
|
||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
||||
// top of the diagnostic log.
|
||||
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
// The machine is going down: paint the exception on screen too — `fatalPrint` forces the
|
||||
// console back on even if a display service was holding the framebuffer — on top of the
|
||||
// diagnostic log.
|
||||
fatalPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||
// Name the culprit: which task, and whether it faulted in ring 3 (a process the
|
||||
// kernel would normally kill — landing here means it had no address space) or ring 0
|
||||
// (the trusted base itself). Without this the fatal report is anonymous.
|
||||
fatalPrint(" task : {d} ({s}), {s}\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe(), if (architecture.fromUser(state)) "ring 3 (user)" else "ring 0 (kernel)" });
|
||||
fatalPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| fatalPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
|
||||
var buffer: [128]u8 = undefined;
|
||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||
// Free the BKL if this core held it (a kernel-mode fault, or a nested fault in the
|
||||
// recovery teardown), so halting this one core doesn't deadlock every other core on
|
||||
// the lock. Only that core stops; the rest — and the supervisor — keep running.
|
||||
sync.releaseIfHeldHere();
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
@@ -486,9 +538,11 @@ pub const panic = std.debug.FullPanic(struct {
|
||||
_ = first_trace_address;
|
||||
log.checkpoint(cp_panic);
|
||||
log.recordPanic(message);
|
||||
status("\nKERNEL PANIC: ");
|
||||
status(message);
|
||||
status("\n");
|
||||
fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen
|
||||
fatal(message);
|
||||
fatal("\n");
|
||||
fatalPrint(" task : {d} ({s})\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe() });
|
||||
sync.releaseIfHeldHere(); // don't deadlock the other cores on the lock we may hold
|
||||
architecture.halt();
|
||||
}
|
||||
}.panic);
|
||||
|
||||
@@ -0,0 +1,234 @@
|
||||
//! The tagged kernel log ring — a circular byte buffer of framed records, each
|
||||
//! stamped by the writer (the kernel) with the sender's pid, task name, level,
|
||||
//! per-boot sequence number, and monotonic timestamp. Pure code over an
|
||||
//! embedded buffer — no architecture or lock imports — so it host-tests
|
||||
//! alongside the other pure kernel pieces (`zig build test`).
|
||||
//!
|
||||
//! `head` and `tail` are free-running u64 positions in a logical byte stream;
|
||||
//! the physical wrap is invisible to readers (all copies are modulo the
|
||||
//! buffer), so a record never splits logically and no padding records exist.
|
||||
//! Reclaim happens record by record: the writer parses the header at `tail`
|
||||
//! (which it wrote itself) and advances until the new record fits — `tail`
|
||||
//! always sits on a record boundary, and sequence-number gaps tell a reader
|
||||
//! exactly how many records it lost.
|
||||
//!
|
||||
//! Locking is the caller's job (log.zig holds its log lock around every call);
|
||||
//! the ring itself is single-writer, snapshot-reader.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
|
||||
pub fn Ring(comptime capacity: usize) type {
|
||||
comptime std.debug.assert(std.math.isPowerOfTwo(capacity));
|
||||
return struct {
|
||||
const Self = @This();
|
||||
|
||||
buffer: [capacity]u8 = undefined,
|
||||
head: u64 = 0,
|
||||
tail: u64 = 0,
|
||||
next_sequence: u64 = 0,
|
||||
|
||||
/// Append one record; returns its sequence number. `name` and `message`
|
||||
/// are clamped to their ABI caps (the syscall clamps earlier too — the
|
||||
/// clamp here makes the ring safe in isolation).
|
||||
pub fn append(
|
||||
self: *Self,
|
||||
pid: u32,
|
||||
name: []const u8,
|
||||
level: abi.KlogLevel,
|
||||
timestamp_ns: u64,
|
||||
message: []const u8,
|
||||
truncated: bool,
|
||||
) u64 {
|
||||
const name_len: usize = @min(name.len, abi.maximum_process_name);
|
||||
const message_len: usize = @min(message.len, abi.klog_maximum_message);
|
||||
const record_len = recordLength(name_len, message_len);
|
||||
|
||||
// Reclaim whole records until the new one fits.
|
||||
while (self.head + record_len - self.tail > capacity) self.reclaimOne();
|
||||
|
||||
const sequence = self.next_sequence;
|
||||
self.next_sequence += 1;
|
||||
|
||||
const header = abi.KlogRecordHeader{
|
||||
.magic = abi.klog_record_magic,
|
||||
.level = level,
|
||||
.name_len = @intCast(name_len),
|
||||
.pid = pid,
|
||||
.sequence = sequence,
|
||||
.timestamp_ns = timestamp_ns,
|
||||
.message_len = @intCast(message_len),
|
||||
.flags = if (truncated) abi.klog_flag_truncated else 0,
|
||||
._reserved = @splat(0),
|
||||
};
|
||||
self.put(self.head, std.mem.asBytes(&header));
|
||||
self.put(self.head + abi.klog_record_header_size, name[0..name_len]);
|
||||
self.put(self.head + abi.klog_record_header_size + name_len, message[0..message_len]);
|
||||
// The alignment pad is dead space; zero it so raw dumps stay tidy.
|
||||
var pad = abi.klog_record_header_size + name_len + message_len;
|
||||
while (pad < record_len) : (pad += 1)
|
||||
self.buffer[@intCast((self.head + pad) % capacity)] = 0;
|
||||
self.head += record_len;
|
||||
return sequence;
|
||||
}
|
||||
|
||||
/// Copy stream bytes beginning at `offset` into `out`. Returns null if
|
||||
/// `offset` fell behind `tail` (overwritten) or lies past `head` — the
|
||||
/// reader re-syncs from status(). 0 bytes means caught up.
|
||||
pub fn read(self: *const Self, offset: u64, out: []u8) ?usize {
|
||||
if (offset < self.tail or offset > self.head) return null;
|
||||
const n: usize = @intCast(@min(out.len, self.head - offset));
|
||||
self.get(offset, out[0..n]);
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Cursors for klog_status. boot_unix_seconds is the kernel wrapper's
|
||||
/// to fill — the ring knows nothing of wall clocks.
|
||||
pub fn status(self: *const Self) abi.KlogStatus {
|
||||
return .{
|
||||
.tail = self.tail,
|
||||
.head = self.head,
|
||||
.next_sequence = self.next_sequence,
|
||||
.boot_unix_seconds = 0,
|
||||
};
|
||||
}
|
||||
|
||||
fn reclaimOne(self: *Self) void {
|
||||
var header_bytes: [abi.klog_record_header_size]u8 = undefined;
|
||||
self.get(self.tail, &header_bytes);
|
||||
const header = std.mem.bytesToValue(abi.KlogRecordHeader, &header_bytes);
|
||||
// The writer wrote this header itself: the assert guards against
|
||||
// memory corruption, not bad input.
|
||||
std.debug.assert(header.magic == abi.klog_record_magic);
|
||||
self.tail += recordLength(header.name_len, header.message_len);
|
||||
}
|
||||
|
||||
fn recordLength(name_len: usize, message_len: usize) usize {
|
||||
return std.mem.alignForward(usize, abi.klog_record_header_size + name_len + message_len, abi.klog_record_alignment);
|
||||
}
|
||||
|
||||
// Byte-at-a-time modulo copies keep the wrap logic obviously correct;
|
||||
// if they ever show in a profile, split into two @memcpy spans.
|
||||
fn put(self: *Self, offset: u64, bytes: []const u8) void {
|
||||
for (bytes, 0..) |b, i| self.buffer[@intCast((offset + i) % capacity)] = b;
|
||||
}
|
||||
|
||||
fn get(self: *const Self, offset: u64, out: []u8) void {
|
||||
for (out, 0..) |*b, i| b.* = self.buffer[@intCast((offset + i) % capacity)];
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
// --- tests (host) -----------------------------------------------------------
|
||||
|
||||
const TestRing = Ring(4096);
|
||||
|
||||
/// Parse the record at `offset` out of `ring`, returning the header plus name
|
||||
/// and message copies — the same walk a userspace drainer performs.
|
||||
const Parsed = struct {
|
||||
header: abi.KlogRecordHeader,
|
||||
name: [abi.maximum_process_name]u8 = undefined,
|
||||
message: [abi.klog_maximum_message]u8 = undefined,
|
||||
|
||||
fn nameSlice(self: *const Parsed) []const u8 {
|
||||
return self.name[0..self.header.name_len];
|
||||
}
|
||||
fn messageSlice(self: *const Parsed) []const u8 {
|
||||
return self.message[0..self.header.message_len];
|
||||
}
|
||||
fn next(self: *const Parsed, offset: u64) u64 {
|
||||
return offset + std.mem.alignForward(usize, abi.klog_record_header_size + self.header.name_len + self.header.message_len, abi.klog_record_alignment);
|
||||
}
|
||||
};
|
||||
|
||||
fn parseAt(ring: *const TestRing, offset: u64) Parsed {
|
||||
var p: Parsed = undefined;
|
||||
var header_bytes: [abi.klog_record_header_size]u8 = undefined;
|
||||
std.debug.assert(ring.read(offset, &header_bytes).? == header_bytes.len);
|
||||
p.header = std.mem.bytesToValue(abi.KlogRecordHeader, &header_bytes);
|
||||
std.debug.assert(p.header.magic == abi.klog_record_magic);
|
||||
_ = ring.read(offset + abi.klog_record_header_size, p.name[0..p.header.name_len]);
|
||||
_ = ring.read(offset + abi.klog_record_header_size + p.header.name_len, p.message[0..p.header.message_len]);
|
||||
return p;
|
||||
}
|
||||
|
||||
test "header size is pinned" {
|
||||
try std.testing.expectEqual(abi.klog_record_header_size, @sizeOf(abi.KlogRecordHeader));
|
||||
}
|
||||
|
||||
test "append/read round trip" {
|
||||
var ring = std.testing.allocator.create(TestRing) catch unreachable;
|
||||
defer std.testing.allocator.destroy(ring);
|
||||
ring.* = .{};
|
||||
|
||||
_ = ring.append(7, "/system/services/fat", .info, 123, "mounted /mnt/usb", false);
|
||||
_ = ring.append(0, "kernel", .raw, 456, "wall clock online", false);
|
||||
|
||||
const first = parseAt(ring, ring.tail);
|
||||
try std.testing.expectEqual(@as(u32, 7), first.header.pid);
|
||||
try std.testing.expectEqual(abi.KlogLevel.info, first.header.level);
|
||||
try std.testing.expectEqual(@as(u64, 123), first.header.timestamp_ns);
|
||||
try std.testing.expectEqualStrings("/system/services/fat", first.nameSlice());
|
||||
try std.testing.expectEqualStrings("mounted /mnt/usb", first.messageSlice());
|
||||
|
||||
const second = parseAt(ring, first.next(ring.tail));
|
||||
try std.testing.expectEqual(@as(u32, 0), second.header.pid);
|
||||
try std.testing.expectEqualStrings("kernel", second.nameSlice());
|
||||
try std.testing.expectEqual(@as(u64, 1), second.header.sequence);
|
||||
}
|
||||
|
||||
test "wrap reclaims whole records and keeps tail on a boundary" {
|
||||
var ring = std.testing.allocator.create(TestRing) catch unreachable;
|
||||
defer std.testing.allocator.destroy(ring);
|
||||
ring.* = .{};
|
||||
|
||||
// Fill far past capacity so the ring wraps many times.
|
||||
var i: u32 = 0;
|
||||
while (i < 200) : (i += 1) {
|
||||
var message: [64]u8 = undefined;
|
||||
const m = std.fmt.bufPrint(&message, "line {d} padding padding padding", .{i}) catch unreachable;
|
||||
_ = ring.append(1, "/system/tests/writer", .info, i, m, false);
|
||||
}
|
||||
try std.testing.expect(ring.head - ring.tail <= 4096);
|
||||
|
||||
// The record at tail parses cleanly (boundary held), and walking to head
|
||||
// yields consecutive sequence numbers.
|
||||
var offset = ring.tail;
|
||||
var previous: ?u64 = null;
|
||||
while (offset < ring.head) {
|
||||
const p = parseAt(ring, offset);
|
||||
if (previous) |q| try std.testing.expectEqual(q + 1, p.header.sequence);
|
||||
previous = p.header.sequence;
|
||||
offset = p.next(offset);
|
||||
}
|
||||
try std.testing.expectEqual(ring.head, offset);
|
||||
// Records were lost (sequence at tail > 0), and the count is the gap.
|
||||
try std.testing.expect(parseAt(ring, ring.tail).header.sequence > 0);
|
||||
}
|
||||
|
||||
test "stale offset returns null; head offset reads zero bytes" {
|
||||
var ring = std.testing.allocator.create(TestRing) catch unreachable;
|
||||
defer std.testing.allocator.destroy(ring);
|
||||
ring.* = .{};
|
||||
|
||||
var i: u32 = 0;
|
||||
while (i < 300) : (i += 1)
|
||||
_ = ring.append(1, "w", .info, i, "0123456789abcdef0123456789abcdef", false);
|
||||
|
||||
var out: [16]u8 = undefined;
|
||||
try std.testing.expect(ring.read(0, &out) == null); // long overwritten
|
||||
try std.testing.expect(ring.read(ring.head + 1, &out) == null); // past the end
|
||||
try std.testing.expectEqual(@as(usize, 0), ring.read(ring.head, &out).?); // caught up
|
||||
}
|
||||
|
||||
test "truncation flag and clamping" {
|
||||
var ring = std.testing.allocator.create(TestRing) catch unreachable;
|
||||
defer std.testing.allocator.destroy(ring);
|
||||
ring.* = .{};
|
||||
|
||||
const long = "x" ** 300; // past klog_maximum_message
|
||||
_ = ring.append(2, "w", .warn, 0, long, true);
|
||||
const p = parseAt(ring, ring.tail);
|
||||
try std.testing.expectEqual(@as(u16, abi.klog_maximum_message), p.header.message_len);
|
||||
try std.testing.expect(p.header.flags & abi.klog_flag_truncated != 0);
|
||||
}
|
||||
+179
-39
@@ -4,22 +4,37 @@
|
||||
//! Output is a *diagnostic convenience, never a correctness dependency* — the
|
||||
//! kernel must boot and run correctly with zero output channels. So logging fans
|
||||
//! out to a set of registered **sinks**, each best-effort and self-guarding: the
|
||||
//! serial UART, the 0xE9 debug console, and — later — a file on a ramdisk/USB/SSD.
|
||||
//! A message reaches whatever channels exist; if none do, the kernel runs on,
|
||||
//! silent but correct.
|
||||
//! serial UART and the 0xE9 debug console. A message reaches whatever channels
|
||||
//! exist; if none do, the kernel runs on, silent but correct.
|
||||
//!
|
||||
//! Retention is the tagged RING (log-ring.zig): every emission becomes one
|
||||
//! record per line, stamped with the sender's pid, task name (its binary path),
|
||||
//! level, sequence number, and monotonic timestamp — attribution is structural,
|
||||
//! stamped by the kernel, not a naming convention a process could forge. The
|
||||
//! stamping is per LINE: an embedded '\n' ends the record, so a payload cannot
|
||||
//! imitate another sender on the line that follows. Oldest records are
|
||||
//! overwritten when the ring is full; sequence gaps make the loss countable.
|
||||
//! `klog_read`/`klog_status` expose the stream to userspace (the logger service
|
||||
//! drains it into per-process files once storage is up).
|
||||
//!
|
||||
//! Locking: a dedicated log spinlock, NOT the big kernel lock. `print` is
|
||||
//! called both inside and outside BKL sections (and from ISRs), so the log
|
||||
//! lock is taken with interrupts off and nothing inside it ever takes the BKL —
|
||||
//! lock order is strictly BKL -> log lock, never the reverse. Panic paths use a
|
||||
//! bounded try-acquire and fall back to sinks-only: a panic must never deadlock
|
||||
//! on its own diagnostics.
|
||||
//!
|
||||
//! The **framebuffer is deliberately not a sink here.** It's a separate output
|
||||
//! surface (a bootstrap text console today, a graphics device driver later), so
|
||||
//! the log never assumes the machine is text-based. `main.zig` mirrors a few
|
||||
//! user-facing status lines and panics to it explicitly; the verbose log does not.
|
||||
//!
|
||||
//! No allocation: the sink table is fixed, so the log works before the heap is up
|
||||
//! and inside a panic. Two channels don't go through the sink list because they
|
||||
//! must survive even a total-output failure: `checkpoint` (a one-byte POST code)
|
||||
//! and `recordPanic` (a breadcrumb in a fixed record).
|
||||
//! the log never assumes the machine is text-based. Two channels bypass the
|
||||
//! sink list because they must survive even a total-output failure:
|
||||
//! `checkpoint` (a one-byte POST code) and `recordPanic` (a fixed breadcrumb).
|
||||
|
||||
const std = @import("std");
|
||||
const architecture = @import("architecture");
|
||||
const abi = @import("abi");
|
||||
const log_ring = @import("log-ring.zig");
|
||||
const wall_clock = @import("wall-clock.zig");
|
||||
|
||||
pub const SinkFn = *const fn ([]const u8) void;
|
||||
|
||||
@@ -36,42 +51,149 @@ pub fn addSink(sink: SinkFn) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Fan `bytes` out to every registered sink.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
// --- the log lock ------------------------------------------------------------
|
||||
|
||||
var lock_held = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn lockAcquire() u64 {
|
||||
const flags = architecture.saveInterrupts();
|
||||
while (lock_held.cmpxchgWeak(0, 1, .acquire, .monotonic) != null) std.atomic.spinLoopHint();
|
||||
return flags;
|
||||
}
|
||||
|
||||
// --- the RAM sink: a retained copy of the whole diagnostic stream ------------
|
||||
//
|
||||
// A fixed in-image buffer that accumulates every logged byte, so a user program
|
||||
// (`log-flush`, and init at shutdown) can read it back through `klog_read` and
|
||||
// persist it to a file — the boot log survives on a headless/real machine that
|
||||
// has no host capturing serial. It is a *sink like any other*: register it with
|
||||
// `addSink(ramSink)` at boot. No allocation (works pre-heap and in a panic).
|
||||
//
|
||||
// It fills linearly and stops when full: the earliest output — the most valuable
|
||||
// for diagnosing a boot — is kept, and the tail is still on the live serial sink.
|
||||
// 256 KiB comfortably holds a full boot plus a long run (a boot is ~15 KiB).
|
||||
fn lockTryAcquire(spins: usize) ?u64 {
|
||||
const flags = architecture.saveInterrupts();
|
||||
var i: usize = 0;
|
||||
while (i < spins) : (i += 1) {
|
||||
if (lock_held.cmpxchgWeak(0, 1, .acquire, .monotonic) == null) return flags;
|
||||
std.atomic.spinLoopHint();
|
||||
}
|
||||
architecture.restoreInterrupts(flags);
|
||||
return null;
|
||||
}
|
||||
|
||||
const ram_capacity = 256 * 1024;
|
||||
var ram_buffer: [ram_capacity]u8 = undefined;
|
||||
var ram_len: usize = 0;
|
||||
fn lockRelease(flags: u64) void {
|
||||
lock_held.store(0, .release);
|
||||
architecture.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// The RAM sink. Best-effort and self-guarding like every sink: appends what fits
|
||||
/// and silently drops the rest once full. (Concurrency matches the other sinks —
|
||||
/// the dominant writer, debug_write, already holds the kernel lock; a rare torn
|
||||
/// append on a kernel-internal line is an accepted diagnostic imperfection.)
|
||||
pub fn ramSink(bytes: []const u8) void {
|
||||
const n = @min(ram_buffer.len - ram_len, bytes.len);
|
||||
if (n != 0) {
|
||||
@memcpy(ram_buffer[ram_len..][0..n], bytes[0..n]);
|
||||
ram_len += n;
|
||||
// --- the ring + renderer -----------------------------------------------------
|
||||
|
||||
/// 512 KiB: the tagged frames cost ~30% over the raw text, and the ring only
|
||||
/// needs to cover the pre-mount backlog (a boot is ~15 KiB of text) — the
|
||||
/// logger service tails it continuously once storage is up.
|
||||
const ring_capacity = 512 * 1024;
|
||||
var ring: log_ring.Ring(ring_capacity) = .{};
|
||||
|
||||
/// Renderer state: whether the sinks sit at a line start, and which pid's line
|
||||
/// is currently open — when a different sender interleaves mid-line, the
|
||||
/// renderer closes the line so serial output can't visually merge two senders.
|
||||
var at_line_start: bool = true;
|
||||
var open_line_pid: u32 = 0;
|
||||
|
||||
/// Append `bytes` as one tagged record per line and render them to the sinks.
|
||||
/// The core emission path: `debug_write` calls this with the sender's identity;
|
||||
/// kernel-internal `write`/`print` funnel here as pid 0 ("kernel", raw).
|
||||
pub fn append(pid: u32, name: []const u8, level: abi.KlogLevel, bytes: []const u8) void {
|
||||
if (bytes.len == 0) return;
|
||||
const now = architecture.nanos();
|
||||
const flags = lockAcquire();
|
||||
defer lockRelease(flags);
|
||||
appendLocked(pid, name, level, now, bytes);
|
||||
}
|
||||
|
||||
/// The panic-safe variant: bounded lock wait; on failure, sinks only — the ring
|
||||
/// entry is lost but the message still reaches serial, and the panic cannot
|
||||
/// deadlock on a core that died holding the log lock.
|
||||
pub fn appendPanic(bytes: []const u8) void {
|
||||
if (lockTryAcquire(100_000)) |flags| {
|
||||
defer lockRelease(flags);
|
||||
appendLocked(0, "kernel", .raw, architecture.nanos(), bytes);
|
||||
} else {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
}
|
||||
}
|
||||
|
||||
/// The accumulated log so far — what `klog_read` copies out.
|
||||
pub fn ramSnapshot() []const u8 {
|
||||
return ram_buffer[0..ram_len];
|
||||
fn appendLocked(pid: u32, name: []const u8, level: abi.KlogLevel, now: u64, bytes: []const u8) void {
|
||||
var rest = bytes;
|
||||
while (rest.len != 0) {
|
||||
const newline = std.mem.indexOfScalar(u8, rest, '\n');
|
||||
// The record payload excludes the newline: a record IS a line. Raw
|
||||
// emissions may leave a line open (kernel boot tables build lines from
|
||||
// pieces); a LEVELED record is a complete line by contract — std.log
|
||||
// payloads carry no trailing newline.
|
||||
const line = if (newline) |i| rest[0..i] else rest;
|
||||
const line_complete = newline != null or level != .raw;
|
||||
if (line.len != 0 or line_complete)
|
||||
_ = ring.append(pid, name, level, now, line, line.len > abi.klog_maximum_message);
|
||||
render(pid, name, level, now, line, line_complete);
|
||||
rest = if (newline) |i| rest[i + 1 ..] else rest[rest.len..];
|
||||
}
|
||||
}
|
||||
|
||||
/// Serial/debugcon rendering. Kernel output and legacy raw user output pass
|
||||
/// through byte-identical to the historical stream (services still write their
|
||||
/// own "name: " prefixes until the std.log migration). Leveled (std.log)
|
||||
/// records get a kernel-rendered "<name>: " prefix at line start — err/warn/
|
||||
/// debug also get their level spelled out.
|
||||
fn render(pid: u32, name: []const u8, level: abi.KlogLevel, now: u64, line: []const u8, line_complete: bool) void {
|
||||
if (sink_count == 0) return;
|
||||
if (line.len == 0 and !line_complete) return;
|
||||
// Compose the whole rendered piece first and emit it in ONE sink call per
|
||||
// sink: fewer, larger UART writes, and no partial-line window should any
|
||||
// path ever reach a sink without the log lock.
|
||||
var buffer: [render_buffer_size]u8 = undefined;
|
||||
var used: usize = 0;
|
||||
if (!at_line_start and open_line_pid != pid) {
|
||||
buffer[used] = '\n';
|
||||
used += 1;
|
||||
at_line_start = true;
|
||||
}
|
||||
if (at_line_start) {
|
||||
// Every line starts with its boot-relative time: the live transcript
|
||||
// (serial AND the on-screen boot console) is a readable timeline —
|
||||
// which is how a slow real-hardware boot gets diagnosed by eye.
|
||||
const seconds = now / 1_000_000_000;
|
||||
const millis = (now / 1_000_000) % 1000;
|
||||
used += (std.fmt.bufPrint(buffer[used..], "[{d:>4}.{d:0>3}] ", .{ seconds, millis }) catch buffer[used..used]).len;
|
||||
}
|
||||
if (at_line_start and level != .raw) {
|
||||
used += place(buffer[used..], name);
|
||||
used += place(buffer[used..], ": ");
|
||||
used += place(buffer[used..], switch (level) {
|
||||
.err => "error: ",
|
||||
.warn => "warning: ",
|
||||
.debug => "debug: ",
|
||||
.info, .raw => "",
|
||||
});
|
||||
}
|
||||
used += place(buffer[used..], line);
|
||||
if (line_complete and used < buffer.len) {
|
||||
buffer[used] = '\n';
|
||||
used += 1;
|
||||
}
|
||||
fanOut(buffer[0..used]);
|
||||
at_line_start = line_complete;
|
||||
open_line_pid = pid;
|
||||
}
|
||||
|
||||
/// newline + timestamp + name + ": warning: " + a full payload line + newline.
|
||||
const render_buffer_size = 1 + 16 + abi.maximum_process_name + 11 + abi.klog_maximum_message + 1;
|
||||
|
||||
fn place(destination: []u8, bytes: []const u8) usize {
|
||||
const n = @min(destination.len, bytes.len);
|
||||
@memcpy(destination[0..n], bytes[0..n]);
|
||||
return n;
|
||||
}
|
||||
|
||||
fn fanOut(bytes: []const u8) void {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
}
|
||||
|
||||
/// Kernel-internal write — a raw record from "kernel" (pid 0). The signature is
|
||||
/// unchanged so every existing kernel call site stays as it is.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
append(0, "kernel", .raw, bytes);
|
||||
}
|
||||
|
||||
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
||||
@@ -81,6 +203,24 @@ pub fn print(comptime fmt: []const u8, args: anytype) void {
|
||||
write(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// klog_read: copy ring stream bytes from `offset` into `out`. Null when the
|
||||
/// cursor was overwritten or lies past the end — the reader re-syncs via
|
||||
/// status(). Zero bytes means caught up.
|
||||
pub fn readAt(offset: u64, out: []u8) ?usize {
|
||||
const flags = lockAcquire();
|
||||
defer lockRelease(flags);
|
||||
return ring.read(offset, out);
|
||||
}
|
||||
|
||||
/// klog_status: the ring cursors plus the boot wall-clock anchor.
|
||||
pub fn status() abi.KlogStatus {
|
||||
const flags = lockAcquire();
|
||||
defer lockRelease(flags);
|
||||
var s = ring.status();
|
||||
s.boot_unix_seconds = wall_clock.bootSeconds();
|
||||
return s;
|
||||
}
|
||||
|
||||
/// Emit a one-byte checkpoint/POST code (I/O port 0x80) — the always-available
|
||||
/// progress channel for when there is no text output at all. Independent of the
|
||||
/// sink list, so it works even before any sink is registered.
|
||||
|
||||
+551
-107
File diff suppressed because it is too large
Load Diff
+333
-40
@@ -31,7 +31,10 @@ const number_priorities = 8;
|
||||
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
|
||||
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
|
||||
|
||||
const State = enum { free, ready, running, blocked };
|
||||
// `reaping` = the task has exited and is queued on its core's reap list; its slot must not
|
||||
// be reused (freeSlot skips it) until the reaper has freed its kernel stack and set it
|
||||
// `free` (docs/threading-plan.md M8/M9).
|
||||
const State = enum { free, ready, running, blocked, reaping };
|
||||
|
||||
pub const Task = struct {
|
||||
id: u32 = 0,
|
||||
@@ -75,30 +78,42 @@ pub const Task = struct {
|
||||
ipc_wait_endpoint: ?*anyopaque = null,
|
||||
// Physical root of this task's address space, or 0 for a kernel task (which
|
||||
// runs on the shared kernel page tables). A user task carries its own.
|
||||
aspace: u64 = 0,
|
||||
address_space: u64 = 0,
|
||||
user_ip: u64 = 0, // user-mode entry point (user task only)
|
||||
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
||||
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
|
||||
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
||||
// as the user heap grows; user task only.
|
||||
heap_next: u64 = 0,
|
||||
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
|
||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
||||
device_map_next: u64 = 0,
|
||||
user_arg: u64 = 0, // value delivered in the user's first argument register at first entry
|
||||
// (rdi on x86_64, via architecture.jumpToUserArg): 0 for a process (its _start ignores
|
||||
// it), the closure pointer for a thread (docs/threading.md)
|
||||
// The user address this task is blocked on in futex_wait (0 = not futex-waiting).
|
||||
// Cleared to 0 by futexWakeLocked as the "woken, not timed out" signal (docs/threading.md).
|
||||
futex_addr: u64 = 0,
|
||||
// The task id this task is blocked in `thread_join` on (0 = not joining). Woken by
|
||||
// `wakeJoinersLocked` when that task exits (docs/threading-plan.md M9).
|
||||
join_target: u32 = 0,
|
||||
// This task's user-space TLS thread pointer — 0 until set via `set_thread_pointer`.
|
||||
// Architecture-neutral: the arch layer maps it to the FS base on x86_64, `TPIDR_EL0` on
|
||||
// aarch64. Restored on every context switch to this task (docs/threading-plan.md M10).
|
||||
thread_pointer: u64 = 0,
|
||||
// The mmap / MMIO grant-arena cursors moved from Task to the per-address-space object
|
||||
// (`AddressSpaceRef`, below) so threads sharing one address space hand out disjoint grants
|
||||
// — see addressSpaceMmapNextPtr / addressSpaceDeviceMapNextPtr (docs/threading-plan.md M7).
|
||||
// --- synchronous IPC (ipc_sync.zig) ---
|
||||
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
||||
// opaque here so the scheduler and IPC modules don't import each other.
|
||||
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
|
||||
// Per-process handle table: a small-int handle names a kernel capability object.
|
||||
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
|
||||
// close/exit and cap-passing paths reclaim the right type. Kept opaque here so the
|
||||
// scheduler and IPC modules don't import each other (ipc_sync.zig owns the kinds).
|
||||
handles: [ipc_maximum_handles]?HandleObject = .{null} ** ipc_maximum_handles,
|
||||
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||
ipc_client: ?*Task = null,
|
||||
ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS)
|
||||
ipc_send_ptr: u64 = 0, // client: outgoing message (virtual_address in this task's address space)
|
||||
ipc_send_len: u64 = 0,
|
||||
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
|
||||
ipc_reply_ptr: u64 = 0, // client: reply buffer (virtual_address)
|
||||
ipc_reply_cap: u64 = 0,
|
||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
||||
shared_memory_map_next: u64 = 0, // bump pointer into this task's shared-memory arena (0 = unseeded)
|
||||
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
||||
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||
@@ -125,7 +140,135 @@ pub const maximum_task_name = abi.maximum_process_name;
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
pub const ipc_maximum_handles = 16;
|
||||
|
||||
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
|
||||
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the
|
||||
/// capability-passing path reclaim/share the right type. The `kind` values are defined by
|
||||
/// ipc_sync.zig (`handle_kind_*`); kept an opaque `u8` here so the scheduler doesn't import
|
||||
/// the IPC module.
|
||||
pub const HandleObject = struct { kind: u8, ptr: *anyopaque };
|
||||
|
||||
var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||
|
||||
/// Address-space reference counts: one live entry per address space, counting the
|
||||
/// tasks that share it. An address space is 1:1 with a process today; threads
|
||||
/// (docs/threading.md) will push a count above 1, and `destroyAddressSpace` must run
|
||||
/// only when the **last** task on an address space exits. All access is under the big
|
||||
/// kernel lock. There can be no more live address spaces than tasks, so the table is
|
||||
/// sized to the task pool and never overflows in practice.
|
||||
// The per-address-space kernel object: a reference count plus the grant-arena cursors.
|
||||
// One live entry per address space; threads sharing an address space share this entry,
|
||||
// so their mmap/mmio grants bump one cursor and never overlap (docs/threading-plan.md M7).
|
||||
// `mmap_next`/`device_map_next` are 0 until process.zig seeds them to the arena base.
|
||||
const AddressSpaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0 };
|
||||
var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks;
|
||||
var address_space_destroy_count: u64 = 0;
|
||||
|
||||
/// Total bytes of task **kernel** stacks currently allocated from the kernel heap —
|
||||
/// incremented when a task is created, decremented when the reaper frees a dead task's
|
||||
/// stack. A test-observable proof that the reaper reclaims every stack (docs/threading-
|
||||
/// plan.md M8): with no live tasks beyond the baseline, this returns to its baseline.
|
||||
var live_stack_bytes: usize = 0;
|
||||
|
||||
/// Test-observable: bytes of task kernel stacks currently live (see `live_stack_bytes`).
|
||||
pub fn liveStackBytes() usize {
|
||||
return live_stack_bytes;
|
||||
}
|
||||
|
||||
/// Free a dead task's kernel stack and drop it from `live_stack_bytes`. The task must be
|
||||
/// off that stack already (killed while not running, or reaped after it switched away).
|
||||
/// Caller holds the kernel lock.
|
||||
fn reapStackLocked(t: *Task) void {
|
||||
if (t.stack.len == 0) return; // boot/idle tasks run on a static stack — nothing to free
|
||||
live_stack_bytes -= t.stack.len;
|
||||
heap.allocator().free(t.stack);
|
||||
t.stack = &.{};
|
||||
t.kstack_top = 0;
|
||||
}
|
||||
|
||||
/// Free every `.reaping` task queued on this core's reap list and mark each `.free` (now
|
||||
/// its slot may be reused). The tasks are all off their stacks (they switched away), and
|
||||
/// the caller holds the lock, so freeing is safe (docs/threading-plan.md M8/M9).
|
||||
fn drainReapListLocked(pc: *PerCpu) void {
|
||||
var node = pc.reap_list;
|
||||
pc.reap_list = null;
|
||||
while (node) |t| {
|
||||
node = t.next; // save the link before we clear it
|
||||
t.next = null;
|
||||
reapStackLocked(t);
|
||||
t.state = .free; // reusable only now, after the stack is freed
|
||||
}
|
||||
}
|
||||
|
||||
/// Take a reference to address space `root` (0 = a kernel task, which owns none).
|
||||
/// Returns false only if the ref table is full — bounded by `maximum_tasks`, so in
|
||||
/// practice it never is. Caller holds the kernel lock.
|
||||
fn retainAddressSpace(root: u64) bool {
|
||||
if (root == 0) return true;
|
||||
var free: ?*AddressSpaceRef = null;
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) {
|
||||
entry.count += 1;
|
||||
return true;
|
||||
}
|
||||
if (entry.count == 0 and free == null) free = entry;
|
||||
}
|
||||
const slot = free orelse return false;
|
||||
slot.* = .{ .root = root, .count = 1 };
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Drop a reference to `root`; destroy the address space when the **last** one drops.
|
||||
/// A `root` with no entry — never retained, e.g. a hand-built test space — is
|
||||
/// destroyed directly, preserving the pre-refcount behaviour. Caller holds the lock.
|
||||
fn releaseAddressSpace(root: u64) void {
|
||||
if (root == 0) return;
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count == 0 or entry.root != root) continue;
|
||||
entry.count -= 1;
|
||||
if (entry.count == 0) {
|
||||
entry.root = 0;
|
||||
architecture.destroyAddressSpace(root);
|
||||
address_space_destroy_count += 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
architecture.destroyAddressSpace(root);
|
||||
address_space_destroy_count += 1;
|
||||
}
|
||||
|
||||
/// Test-observable: how many address spaces are live (entries with a nonzero count).
|
||||
pub fn liveAddressSpaceCount() u32 {
|
||||
var live: u32 = 0;
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0) live += 1;
|
||||
}
|
||||
return live;
|
||||
}
|
||||
|
||||
/// Test-observable: total address-space destructions since boot.
|
||||
pub fn addressSpaceDestroyCount() u64 {
|
||||
return address_space_destroy_count;
|
||||
}
|
||||
|
||||
/// Pointer to the mmap grant-arena cursor for address space `root`, so the mmap syscall
|
||||
/// can read-and-bump it. Per-address-space (not per-task), so sibling threads get
|
||||
/// disjoint grants. **Caller holds the kernel lock** (the entry is stable while held).
|
||||
/// Null only if `root` was never retained — which can't happen for a live user task.
|
||||
pub fn addressSpaceMmapNextPtr(root: u64) ?*u64 {
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) return &entry.mmap_next;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Pointer to the MMIO grant-arena cursor for address space `root` (see
|
||||
/// `addressSpaceMmapNextPtr`). Caller holds the kernel lock.
|
||||
pub fn addressSpaceDeviceMapNextPtr(root: u64) ?*u64 {
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) return &entry.device_map_next;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
var next_id: u32 = 1;
|
||||
|
||||
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
||||
@@ -145,11 +288,18 @@ pub const PerCpu = struct {
|
||||
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
|
||||
index: u32 = 0, // dense 0-based core index
|
||||
online: bool = false, // has this core finished bring-up?
|
||||
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
||||
loaded_address_space: u64 = 0, // the address-space root currently loaded on this core
|
||||
loaded_thread_pointer: u64 = 0, // the TLS thread pointer currently loaded on this core (docs/threading-plan.md M10)
|
||||
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||
pinned_bitmap: u8 = 0,
|
||||
// Tasks that ended while running on THIS core: they could not free the kernel stack
|
||||
// they were standing on, so each pushed itself onto this list (`.reaping` state, linked
|
||||
// via `Task.next`) and switched away. The next task to run on this core — or the timer
|
||||
// tick — frees their stacks from its own stack, safely (docs/threading-plan.md M8). A
|
||||
// *list* (not one slot) so a second death before the first is drained can't lose it.
|
||||
reap_list: ?*Task = null,
|
||||
};
|
||||
|
||||
const maximum_cpus = parameters.maximum_cpus;
|
||||
@@ -181,7 +331,7 @@ var preemption_enabled = true;
|
||||
/// boot, before interrupts are enabled — so no lock is needed here.
|
||||
pub fn init(boot_priority: Priority) void {
|
||||
const pc = &cpus[0];
|
||||
pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() };
|
||||
pc.* = .{ .index = 0, .online = true, .loaded_address_space = architecture.kernelPageTable() };
|
||||
architecture.setCpuLocal(0, @intFromPtr(pc));
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
pc.current = &tasks[0];
|
||||
@@ -220,7 +370,7 @@ pub fn secondaryMain() callconv(.c) noreturn {
|
||||
pc.current = t;
|
||||
pc.idle = t;
|
||||
pc.online = true;
|
||||
pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||
pc.loaded_address_space = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||
sync.leave(flags);
|
||||
|
||||
architecture.enableInterrupts(); // the timer now preempts this idle context into work
|
||||
@@ -306,7 +456,7 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
return ok;
|
||||
}
|
||||
|
||||
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
||||
/// Spawn a **user** task: a task with its own address space (`address_space`) that starts
|
||||
/// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]).
|
||||
/// `supervisor` is the id of the spawning process (0 = the kernel) — the kill
|
||||
/// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller
|
||||
@@ -315,19 +465,27 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
/// lands in `user_task_trampoline`.
|
||||
/// Returns the new process id, or null (creating nothing) if the table is full or
|
||||
/// out of memory.
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `address_space` holds it
|
||||
/// across the whole spawn, so the address space and the task appear atomically).
|
||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
||||
pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
||||
const t = freeSlot() orelse return null;
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch return null;
|
||||
// Take this task's reference to the address space before we commit the slot, so a
|
||||
// failure here leaves nothing to unwind (the caller still owns the raw `address_space`).
|
||||
if (!retainAddressSpace(address_space)) {
|
||||
heap.allocator().free(stack);
|
||||
return null;
|
||||
}
|
||||
live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8)
|
||||
t.* = .{
|
||||
.id = next_id,
|
||||
.state = .ready,
|
||||
.priority = priority,
|
||||
.stack = stack,
|
||||
.aspace = aspace,
|
||||
.address_space = address_space,
|
||||
.user_ip = entry,
|
||||
.user_sp = user_sp,
|
||||
.user_arg = user_arg,
|
||||
.supervisor = supervisor,
|
||||
.exit_endpoint = exit_endpoint,
|
||||
};
|
||||
@@ -353,7 +511,7 @@ fn startUserTask() void {
|
||||
// No serial chatter here: this runs on every spawn, unserialized against
|
||||
// user-space writes, and its output used to shear concurrent log lines in
|
||||
// half — the largest source of corrupted markers in the QEMU scenarios.
|
||||
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
||||
architecture.jumpToUserArg(t.user_ip, t.user_sp, t.user_arg); // noreturn (arg0 = 0 for a process)
|
||||
}
|
||||
|
||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||
@@ -362,6 +520,7 @@ fn startUserTask() void {
|
||||
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
||||
const t = freeSlot() orelse @panic("sched: task table full");
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||
live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8)
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
@@ -402,7 +561,7 @@ fn schedule() void {
|
||||
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
||||
/// user-mode interrupt lands on a good stack) and its address space (only when
|
||||
/// it differs from what's loaded — every page-table switch is a full TLB flush),
|
||||
/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used
|
||||
/// then switch registers/stacks. Kernel tasks (address_space == 0, no kstack_top used
|
||||
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
|
||||
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
|
||||
/// workload. The big kernel lock is held and interrupts are off throughout, so no
|
||||
@@ -410,12 +569,25 @@ fn schedule() void {
|
||||
/// `save_sp` receives the outgoing task's stack pointer.
|
||||
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
||||
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
|
||||
const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable();
|
||||
if (want != pc.loaded_aspace) {
|
||||
const want = if (next.address_space != 0) next.address_space else architecture.kernelPageTable();
|
||||
if (want != pc.loaded_address_space) {
|
||||
architecture.loadPageTable(want);
|
||||
pc.loaded_aspace = want;
|
||||
pc.loaded_address_space = want;
|
||||
}
|
||||
// Restore the next task's user TLS thread pointer — only on change, the same
|
||||
// conditional-load discipline as CR3 above (docs/threading-plan.md M10).
|
||||
if (next.thread_pointer != pc.loaded_thread_pointer) {
|
||||
architecture.setThreadPointer(next.thread_pointer);
|
||||
pc.loaded_thread_pointer = next.thread_pointer;
|
||||
}
|
||||
architecture.switchContext(save_sp, next.sp);
|
||||
// Resumed now (switchContext returned into our own switchTo frame). Re-fetch the core
|
||||
// via thisCpu(): the `pc` parameter is from *our* earlier switchTo call, so it names
|
||||
// the core we last ran on — stale if we migrated. switchContext only swaps stacks on
|
||||
// the current core, so thisCpu() is the core the just-dead task died on. If a task
|
||||
// died switching to us, free its kernel stack: we're on ours so it's safe, and the big
|
||||
// lock is still held so its slot can't have been reused (docs/threading-plan.md M8).
|
||||
drainReapListLocked(thisCpu());
|
||||
}
|
||||
|
||||
/// Voluntarily give up the CPU to the next ready task.
|
||||
@@ -436,6 +608,98 @@ pub fn sleep(ms: u64) void {
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
// --- futex: block/wake on a user address (docs/threading.md) ----------------
|
||||
//
|
||||
// A futex waiter is not linked into any queue — it is simply a `.blocked` task
|
||||
// tagged with the address it waits on (`futex_addr`). Waking scans the task table
|
||||
// (bounded) for matching waiters. A timed wait also sets `wake_at`, so the timer's
|
||||
// `wakeExpired` can wake it; `futex_addr` stays non-zero in that case, which is how
|
||||
// the waiter tells a timeout from a real wake.
|
||||
|
||||
pub const FutexResult = enum { woken, timed_out };
|
||||
|
||||
/// Block the current task on futex `addr` until woken, or (if `timeout_ms > 0`) the
|
||||
/// deadline. **Precondition:** the big kernel lock is held and the caller has already
|
||||
/// checked, under this same lock, that the futex word equals the expected value — so
|
||||
/// no wake can be missed. Returns with the lock still held.
|
||||
pub fn futexWaitLocked(addr: u64, timeout_ms: u64) FutexResult {
|
||||
const t = current();
|
||||
t.futex_addr = addr;
|
||||
t.wake_at = if (timeout_ms > 0) architecture.millis() + timeout_ms else 0;
|
||||
t.state = .blocked;
|
||||
schedule(); // woken by futexWakeLocked (clears futex_addr) or wakeExpired (timeout)
|
||||
const woken = t.futex_addr == 0;
|
||||
t.futex_addr = 0;
|
||||
t.wake_at = 0;
|
||||
return if (woken) .woken else .timed_out;
|
||||
}
|
||||
|
||||
/// Wake up to `count` tasks blocked in `futex_wait` on `addr` in address space
|
||||
/// `address_space`. Precondition: the big kernel lock is held. Returns how many woke.
|
||||
pub fn futexWakeLocked(address_space: u64, addr: u64, count: u32) u32 {
|
||||
var woken: u32 = 0;
|
||||
for (&tasks) |*t| {
|
||||
if (woken >= count) break;
|
||||
if (t.state == .blocked and t.address_space == address_space and t.futex_addr == addr) {
|
||||
t.futex_addr = 0; // the "woken, not timed out" signal to futexWaitLocked
|
||||
t.wake_at = 0;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
woken += 1;
|
||||
}
|
||||
}
|
||||
return woken;
|
||||
}
|
||||
|
||||
// --- thread join (docs/threading-plan.md M9) --------------------------------
|
||||
//
|
||||
// join needs no per-thread IPC endpoint: `thread_join(tid)` blocks the caller until the
|
||||
// task with id `tid` has exited, and the exit paths wake any joiner. The caller only ever
|
||||
// reclaims the joined thread's *user* stack (which the thread vacated the moment it entered
|
||||
// the kernel to exit), so waking at exit time — not reap time — is safe.
|
||||
|
||||
/// True if a task with id `tid` is still live (has not exited). Caller holds the lock.
|
||||
fn aliveTid(tid: u32) bool {
|
||||
for (&tasks) |*t| {
|
||||
if (t.id == tid and t.state != .free and t.state != .reaping) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Block the current task until the task with id `tid` exits (or return at once if it
|
||||
/// already has / never existed). **Precondition:** the big kernel lock is held; returns
|
||||
/// with it still held. Woken by `wakeJoinersLocked`.
|
||||
pub fn joinThreadLocked(tid: u32) void {
|
||||
while (aliveTid(tid)) {
|
||||
const t = current();
|
||||
t.join_target = tid;
|
||||
t.state = .blocked;
|
||||
schedule(); // woken when the joined task exits; lock handed off across the switch
|
||||
t.join_target = 0;
|
||||
}
|
||||
}
|
||||
|
||||
/// Set the calling task's user TLS thread pointer and load it now. Persisted on the Task so
|
||||
/// context switches restore it (docs/threading-plan.md M10). Caller holds the kernel lock.
|
||||
pub fn setThreadPointerLocked(addr: u64) void {
|
||||
const pc = thisCpu();
|
||||
pc.current.thread_pointer = addr;
|
||||
architecture.setThreadPointer(addr);
|
||||
pc.loaded_thread_pointer = addr;
|
||||
}
|
||||
|
||||
/// Wake every task blocked in `thread_join` on `tid` — called from the exit paths once the
|
||||
/// exiting task's state is `.free`. Caller holds the lock.
|
||||
fn wakeJoinersLocked(tid: u32) void {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .blocked and t.join_target == tid) {
|
||||
t.join_target = 0;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- event-based blocking -------------------------------------------------
|
||||
//
|
||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
||||
@@ -537,7 +801,7 @@ fn removeFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task
|
||||
/// Precondition: the big kernel lock is held.
|
||||
pub fn taskByIdLocked(id: u32) ?*Task {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state != .free and t.id == id) return t;
|
||||
if (t.state != .free and t.state != .reaping and t.id == id) return t;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -547,7 +811,7 @@ pub fn taskByIdLocked(id: u32) ?*Task {
|
||||
/// Precondition: the big kernel lock is held.
|
||||
pub fn forgetIpcClientLocked(t: *Task) void {
|
||||
for (&tasks) |*other| {
|
||||
if (other.state != .free and other.ipc_client == t) other.ipc_client = null;
|
||||
if (other.state != .free and other.state != .reaping and other.ipc_client == t) other.ipc_client = null;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -639,7 +903,11 @@ pub var reap_task_hook: ?*const fn (*Task) void = null;
|
||||
fn reapKillPendingLocked() void {
|
||||
const pc = thisCpu();
|
||||
const cur = pc.current;
|
||||
if (cur.kill_pending and cur.aspace != 0 and !cur.in_system_call) {
|
||||
// Safety net: if a dying task switched to a *fresh* task (which enters via
|
||||
// task_trampoline, not switchTo's tail), its stack is still queued here. The dying
|
||||
// task switched away before this tick, so it is off its stack — drain now (M8).
|
||||
drainReapListLocked(pc);
|
||||
if (cur.kill_pending and cur.address_space != 0 and !cur.in_system_call) {
|
||||
if (terminate_current_hook) |hook| hook(); // noreturn
|
||||
}
|
||||
if (reap_task_hook) |hook| {
|
||||
@@ -680,7 +948,11 @@ pub fn setPreemption(enabled: bool) void {
|
||||
pub fn exit() noreturn {
|
||||
_ = sync.enter();
|
||||
const pc = thisCpu();
|
||||
pc.current.state = .free;
|
||||
// Queue this task for reaping: `.reaping` keeps its slot out of freeSlot until its
|
||||
// stack is freed; `next` links it on the core's reap list (M8/M9).
|
||||
pc.current.state = .reaping;
|
||||
pc.current.next = pc.reap_list;
|
||||
pc.reap_list = pc.current;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
@@ -705,17 +977,21 @@ pub fn exitUser() noreturn {
|
||||
pub fn exitUserLocked() noreturn {
|
||||
const pc = thisCpu();
|
||||
const dying = pc.current;
|
||||
const as = dying.aspace;
|
||||
const as = dying.address_space;
|
||||
if (as != 0) {
|
||||
const kroot = architecture.kernelPageTable();
|
||||
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
||||
pc.loaded_aspace = kroot;
|
||||
architecture.destroyAddressSpace(as);
|
||||
pc.loaded_address_space = kroot;
|
||||
releaseAddressSpace(as); // destroys only when this was the last task on the space
|
||||
}
|
||||
dying.state = .free;
|
||||
dying.aspace = 0;
|
||||
dying.state = .reaping; // dead but its slot stays reserved until the stack is freed
|
||||
wakeJoinersLocked(dying.id); // let any thread_join(dying.id) return (M9)
|
||||
dying.address_space = 0;
|
||||
dying.kill_pending = false;
|
||||
dying.in_system_call = false;
|
||||
// Queue for reaping: the task we switch to (or the next tick) frees this stack (M8/M9).
|
||||
dying.next = pc.reap_list;
|
||||
pc.reap_list = dying;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
@@ -731,12 +1007,14 @@ pub fn exitUserLocked() noreturn {
|
||||
/// task isn't running). The kernel stack is leaked, as in `exitUser` (no reaper
|
||||
/// yet). Precondition: the big kernel lock is held.
|
||||
pub fn destroyTaskLocked(t: *Task) void {
|
||||
if (t.aspace != 0) architecture.destroyAddressSpace(t.aspace);
|
||||
t.aspace = 0;
|
||||
if (t.address_space != 0) releaseAddressSpace(t.address_space); // destroys only on the last reference
|
||||
reapStackLocked(t); // safe to free now: `t` is not running on any core (M8)
|
||||
t.address_space = 0;
|
||||
t.kill_pending = false;
|
||||
t.in_system_call = false;
|
||||
t.wake_at = 0;
|
||||
t.state = .free;
|
||||
wakeJoinersLocked(t.id); // a killed thread's joiners must return too (M9)
|
||||
}
|
||||
|
||||
/// Snapshot the task table into `out` (up to its length), returning the total
|
||||
@@ -750,7 +1028,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||
defer sync.leave(flags);
|
||||
var total: u64 = 0;
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .free) continue;
|
||||
if (t.state == .free or t.state == .reaping) continue; // reaping = already exited
|
||||
if (total < out.len) {
|
||||
const d = &out[total];
|
||||
d.* = .{
|
||||
@@ -760,7 +1038,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||
.ready => .ready,
|
||||
.running => .running,
|
||||
.blocked => .blocked,
|
||||
.free => unreachable,
|
||||
.free, .reaping => unreachable,
|
||||
})),
|
||||
.priority = t.priority,
|
||||
.name_length = t.name_length,
|
||||
@@ -774,7 +1052,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||
|
||||
/// Whether the running task is a user process (has its own address space).
|
||||
pub fn currentIsUserProcess() bool {
|
||||
return current().aspace != 0;
|
||||
return current().address_space != 0;
|
||||
}
|
||||
|
||||
pub fn currentId() u32 {
|
||||
@@ -791,6 +1069,21 @@ pub fn currentCpuIndex() u32 {
|
||||
return thisCpu().index;
|
||||
}
|
||||
|
||||
/// The running task's id, or 0 if this core's scheduler isn't up yet (early boot, no GS
|
||||
/// base). Safe for a fault reporter to call unconditionally — like `currentCpuIndex`,
|
||||
/// it never dereferences an unpublished per-CPU pointer and so can't fault a second time.
|
||||
pub fn currentIdSafe() u32 {
|
||||
if (architecture.cpuLocal() == 0) return 0;
|
||||
return thisCpu().current.id;
|
||||
}
|
||||
|
||||
/// The running task's name (argv[0]), or "" if this core's scheduler isn't up yet.
|
||||
/// The companion to `currentIdSafe` for naming the culprit in a fatal fault report.
|
||||
pub fn currentNameSafe() []const u8 {
|
||||
if (architecture.cpuLocal() == 0) return "";
|
||||
return thisCpu().current.name();
|
||||
}
|
||||
|
||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||
pub fn setPriority(p: Priority) void {
|
||||
current().priority = p;
|
||||
|
||||
@@ -33,6 +33,13 @@ const architecture = @import("architecture");
|
||||
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
||||
var held = std.atomic.Value(u32).init(0);
|
||||
|
||||
/// The per-CPU base pointer (`architecture.cpuLocal()`) of the core currently holding
|
||||
/// the lock, or 0 when free. Metadata only — `held` is what enforces exclusion — read
|
||||
/// solely by `releaseIfHeldHere` on the fatal-fault path. `cpuLocal()` is a unique,
|
||||
/// architecture-level token per core (0 before this core's GS base is published, which
|
||||
/// is fine: that window is single-core early boot, where no other core can deadlock).
|
||||
var owner = std.atomic.Value(usize).init(0);
|
||||
|
||||
/// Enter the kernel: disable interrupts on this core, then spin until we own the
|
||||
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
|
||||
/// Interrupts stay off for the whole critical section so this core's timer tick
|
||||
@@ -67,14 +74,29 @@ export fn releaseForFreshTask() callconv(.c) void {
|
||||
release();
|
||||
}
|
||||
|
||||
/// Release the big kernel lock **only if this core is the one holding it** — a no-op
|
||||
/// otherwise. For the fatal-fault path (a kernel-mode fault, or a nested fault inside the
|
||||
/// recovery teardown, both of which run under the lock): a core that dies holding the BKL
|
||||
/// must free it, or every other core spins forever in `acquire` and the whole machine
|
||||
/// deadlocks instead of just that core stopping. It must NOT free a lock another core
|
||||
/// owns, hence the owner check. Caveat: if we held it mid-mutation the shared state may be
|
||||
/// inconsistent — but letting the other cores (and the supervisor) run on possibly-degraded
|
||||
/// state is strictly more recoverable than a guaranteed total hang.
|
||||
pub fn releaseIfHeldHere() void {
|
||||
const me = architecture.cpuLocal();
|
||||
if (me != 0 and owner.load(.monotonic) == me) release();
|
||||
}
|
||||
|
||||
fn acquire() void {
|
||||
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
||||
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
||||
while (held.swap(1, .acquire) != 0) {
|
||||
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
||||
}
|
||||
owner.store(architecture.cpuLocal(), .monotonic);
|
||||
}
|
||||
|
||||
fn release() void {
|
||||
owner.store(0, .monotonic);
|
||||
held.store(0, .release);
|
||||
}
|
||||
|
||||
+1095
-200
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,368 @@
|
||||
//! The kernel-resident VFS root: the mount table and the kernel-backed nodes.
|
||||
//!
|
||||
//! The kernel's job here is NAMING, never data plumbing to userspace backends —
|
||||
//! the mechanism is **resolve + redirect**:
|
||||
//!
|
||||
//! - `fs_resolve(path)` walks the mount table. A path under a KERNEL-backed
|
||||
//! mount (the initrd at /system, the scratch ram nodes) resolves to a
|
||||
//! stateless node TOKEN served directly by `fs_node` (read/status/readdir
|
||||
//! with copy-out). A path under a USERSPACE mount (the fat server at
|
||||
//! /mnt/usb and /var) resolves to the backend's ENDPOINT: the kernel
|
||||
//! installs a (deduplicated) handle in the caller's table, rewrites the
|
||||
//! path mount-relative, and the caller speaks the unchanged vfs-protocol
|
||||
//! to the backend over the ordinary ipc_call rendezvous. The kernel never
|
||||
//! blocks on a userspace server.
|
||||
//!
|
||||
//! - Kernel node tokens are PERMANENT for a boot: the initrd is immutable and
|
||||
//! ram nodes are never reclaimed — no open-handle state, no close, no sweep
|
||||
//! on client death. Backend file state lives in the backend, which sweeps
|
||||
//! dead clients itself via the published exit events.
|
||||
//!
|
||||
//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the
|
||||
//! backend endpoint handle is the capability, exactly the trust of the old
|
||||
//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount
|
||||
//! into the backend's namespace ("/var" -> fat's "/var" subtree while the same
|
||||
//! backend also serves "/mnt/usb" from its root), so FHS paths stay decoupled
|
||||
//! from which volume happens to carry them.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const ipc = @import("ipc-synchronous.zig");
|
||||
|
||||
// --- node tokens -------------------------------------------------------------
|
||||
|
||||
/// Kind lives in the top byte of a token; the index below. Tokens are permanent
|
||||
/// for a boot, so userspace may cache them freely.
|
||||
pub const token_kind_shift = 56;
|
||||
pub const token_kind_initrd_file: u64 = 1;
|
||||
pub const token_kind_initrd_directory: u64 = 2;
|
||||
pub const token_kind_ram: u64 = 3;
|
||||
|
||||
fn token(kind: u64, index: u64) u64 {
|
||||
return (kind << token_kind_shift) | index;
|
||||
}
|
||||
|
||||
fn tokenKind(t: u64) u64 {
|
||||
return t >> token_kind_shift;
|
||||
}
|
||||
|
||||
fn tokenIndex(t: u64) u64 {
|
||||
return t & ((@as(u64, 1) << token_kind_shift) - 1);
|
||||
}
|
||||
|
||||
// --- the mount table ---------------------------------------------------------
|
||||
|
||||
pub const maximum_mounts = 8;
|
||||
const maximum_prefix = 64;
|
||||
const maximum_rewrite = 32;
|
||||
|
||||
const MountKind = enum(u8) { kernel_initrd, backend };
|
||||
|
||||
const Mount = struct {
|
||||
used: bool = false,
|
||||
prefix: [maximum_prefix]u8 = undefined,
|
||||
prefix_len: usize = 0,
|
||||
kind: MountKind = .backend,
|
||||
backend: ?*ipc.Endpoint = null, // referenced while mounted
|
||||
rewrite: [maximum_rewrite]u8 = undefined,
|
||||
rewrite_len: usize = 0,
|
||||
|
||||
fn prefixSlice(self: *const Mount) []const u8 {
|
||||
return self.prefix[0..self.prefix_len];
|
||||
}
|
||||
fn rewriteSlice(self: *const Mount) []const u8 {
|
||||
return self.rewrite[0..self.rewrite_len];
|
||||
}
|
||||
};
|
||||
|
||||
var mounts: [maximum_mounts]Mount = @splat(.{});
|
||||
|
||||
/// The initrd image (set once at boot) and its derived directory table.
|
||||
var ramdisk_image: ?[]const u8 = null;
|
||||
|
||||
const maximum_directories = 8;
|
||||
const Directory = struct {
|
||||
path: [maximum_prefix]u8 = undefined,
|
||||
path_len: usize = 0,
|
||||
parent: usize = 0, // index into `directories`; 0 is /system itself
|
||||
|
||||
fn slice(self: *const Directory) []const u8 {
|
||||
return self.path[0..self.path_len];
|
||||
}
|
||||
};
|
||||
var directories: [maximum_directories]Directory = @splat(.{});
|
||||
var directory_count: usize = 0;
|
||||
|
||||
// --- pure path helpers (ported from the userspace router, with its tests) ----
|
||||
|
||||
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by
|
||||
/// a path separator — return the path relative to the mount ("/" for an exact
|
||||
/// match, otherwise the tail beginning with '/'). Null when not under the
|
||||
/// mount, so "/mnt/usb" never captures "/mnt/usbextra".
|
||||
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
||||
if (path.len < mount_prefix.len) return null;
|
||||
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
||||
if (path.len == mount_prefix.len) return "/";
|
||||
if (path[mount_prefix.len] != '/') return null;
|
||||
return path[mount_prefix.len..];
|
||||
}
|
||||
|
||||
pub fn isAbsolute(path: []const u8) bool {
|
||||
return path.len > 0 and path[0] == '/';
|
||||
}
|
||||
|
||||
/// The parent directory portion of an initrd path ("/system/services/fat" ->
|
||||
/// "/system/services").
|
||||
fn parentOf(path: []const u8) []const u8 {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/') orelse return path[0..0];
|
||||
if (slash == 0) return path[0..1];
|
||||
return path[0..slash];
|
||||
}
|
||||
|
||||
// --- boot wiring -------------------------------------------------------------
|
||||
|
||||
/// Publish the initrd as the kernel-backed /system mount and derive its bounded
|
||||
/// directory table (the unique parents of the entry paths). Called once at boot.
|
||||
pub fn setInitialRamdisk(image: []const u8) void {
|
||||
ramdisk_image = image;
|
||||
installMount("/system", .kernel_initrd, null, "");
|
||||
|
||||
// Directory 0 is /system itself.
|
||||
directories[0] = .{ .parent = 0 };
|
||||
@memcpy(directories[0].path[0..7], "/system");
|
||||
directories[0].path_len = 7;
|
||||
directory_count = 1;
|
||||
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
// Register every ancestor directory strictly below /system.
|
||||
var parent = parentOf(item.name);
|
||||
while (parent.len > 7) : (parent = parentOf(parent)) {
|
||||
if (directoryIndex(parent) == null and directory_count < maximum_directories) {
|
||||
var d = &directories[directory_count];
|
||||
@memcpy(d.path[0..parent.len], parent);
|
||||
d.path_len = parent.len;
|
||||
d.parent = 0; // fixed up below once all exist
|
||||
directory_count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Parent links (a second pass so out-of-order registration doesn't matter).
|
||||
for (directories[1..directory_count]) |*d| {
|
||||
d.parent = directoryIndex(parentOf(d.slice())) orelse 0;
|
||||
}
|
||||
}
|
||||
|
||||
fn directoryIndex(path: []const u8) ?usize {
|
||||
for (directories[0..directory_count], 0..) |*d, i| {
|
||||
if (std.mem.eql(u8, d.slice(), path)) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void {
|
||||
// Remount replaces: a restarted backend re-mounts its prefix.
|
||||
var slot: ?*Mount = null;
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||
if (m.backend) |old| ipc.dropRef(old);
|
||||
slot = m;
|
||||
break;
|
||||
}
|
||||
if (slot == null and !m.used) slot = m;
|
||||
}
|
||||
const m = slot orelse return;
|
||||
m.* = .{ .used = true, .kind = kind, .backend = backend };
|
||||
@memcpy(m.prefix[0..prefix.len], prefix);
|
||||
m.prefix_len = prefix.len;
|
||||
@memcpy(m.rewrite[0..rewrite.len], rewrite);
|
||||
m.rewrite_len = rewrite.len;
|
||||
}
|
||||
|
||||
// --- resolve -----------------------------------------------------------------
|
||||
|
||||
pub const Resolved = union(enum) {
|
||||
/// Kernel-served: a permanent node token.
|
||||
kernel_node: u64,
|
||||
/// Backend-served: the endpoint plus the rewritten mount-relative path.
|
||||
backend: struct { endpoint: *ipc.Endpoint, path: [maximum_rewrite + maximum_prefix + 160]u8, path_len: usize },
|
||||
not_found: void,
|
||||
};
|
||||
|
||||
/// Longest-prefix match over the mount table, then per-kind resolution.
|
||||
/// `create`-intent on the immutable /system fails here (EROFS-style).
|
||||
pub fn resolvePath(path: []const u8, wants_create: bool) Resolved {
|
||||
if (!isAbsolute(path)) {
|
||||
return .{ .not_found = {} }; // bare names have no kernel namespace (ramfs retired)
|
||||
}
|
||||
var best: ?*Mount = null;
|
||||
var best_relative: []const u8 = undefined;
|
||||
for (&mounts) |*m| {
|
||||
if (!m.used) continue;
|
||||
const relative = underMount(path, m.prefixSlice()) orelse continue;
|
||||
if (best == null or m.prefix_len > best.?.prefix_len) {
|
||||
best = m;
|
||||
best_relative = relative;
|
||||
}
|
||||
}
|
||||
const m = best orelse return .{ .not_found = {} };
|
||||
switch (m.kind) {
|
||||
.kernel_initrd => {
|
||||
if (wants_create) return .{ .not_found = {} }; // read-only
|
||||
return resolveInitrd(path);
|
||||
},
|
||||
.backend => {
|
||||
const endpoint = m.backend orelse return .{ .not_found = {} };
|
||||
if (endpoint.dead) {
|
||||
// The backend died: treat the mount as gone (it re-mounts on
|
||||
// restart) and release our reference lazily.
|
||||
ipc.dropRef(endpoint);
|
||||
m.backend = null;
|
||||
m.used = false;
|
||||
return .{ .not_found = {} };
|
||||
}
|
||||
var out: Resolved = .{ .backend = .{ .endpoint = endpoint, .path = undefined, .path_len = 0 } };
|
||||
const rewrite = m.rewriteSlice();
|
||||
const tail = if (std.mem.eql(u8, best_relative, "/") and rewrite.len != 0) "" else best_relative;
|
||||
const total = rewrite.len + tail.len;
|
||||
if (total > out.backend.path.len or total == 0) {
|
||||
if (rewrite.len == 0 and tail.len == 0) return .{ .not_found = {} };
|
||||
if (total > out.backend.path.len) return .{ .not_found = {} };
|
||||
}
|
||||
@memcpy(out.backend.path[0..rewrite.len], rewrite);
|
||||
@memcpy(out.backend.path[rewrite.len..][0..tail.len], tail);
|
||||
out.backend.path_len = total;
|
||||
return out;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn resolveInitrd(path: []const u8) Resolved {
|
||||
if (directoryIndex(path)) |index| return .{ .kernel_node = token(token_kind_initrd_directory, index) };
|
||||
const image = ramdisk_image orelse return .{ .not_found = {} };
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return .{ .not_found = {} };
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (std.mem.eql(u8, item.name, path)) return .{ .kernel_node = token(token_kind_initrd_file, i) };
|
||||
}
|
||||
return .{ .not_found = {} };
|
||||
}
|
||||
|
||||
// --- fs_node: serving kernel-backed nodes ------------------------------------
|
||||
|
||||
/// Read `out.len` bytes of an initrd file at `offset`. Returns bytes copied
|
||||
/// (0 at EOF) or null for a bad token. Lock-free: the initrd is immutable.
|
||||
pub fn nodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
||||
if (tokenKind(node_token) != token_kind_initrd_file) return null;
|
||||
const image = ramdisk_image orelse return null;
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
||||
const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null;
|
||||
if (offset >= item.blob.len) return 0;
|
||||
const n = @min(out.len, item.blob.len - @as(usize, @intCast(offset)));
|
||||
@memcpy(out[0..n], item.blob[@intCast(offset)..][0..n]);
|
||||
return n;
|
||||
}
|
||||
|
||||
/// A node's metadata in vfs-protocol FileStatus shape (size, kind, mtime).
|
||||
pub fn nodeStatus(node_token: u64) ?abi.FileAttributes {
|
||||
switch (tokenKind(node_token)) {
|
||||
token_kind_initrd_file => {
|
||||
const image = ramdisk_image orelse return null;
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
||||
const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null;
|
||||
return .{ .size = item.blob.len, .kind = abi.file_kind_regular };
|
||||
},
|
||||
token_kind_initrd_directory => {
|
||||
if (tokenIndex(node_token) >= directory_count) return null;
|
||||
return .{ .size = 0, .kind = abi.file_kind_directory };
|
||||
},
|
||||
else => return null,
|
||||
}
|
||||
}
|
||||
|
||||
/// The `cursor`th child of an initrd directory: fills `name_out`, returns the
|
||||
/// entry header, or null past the end / bad token. Cursor enumerates
|
||||
/// subdirectories first, then files whose parent is this directory — stable,
|
||||
/// because the initrd is immutable.
|
||||
pub fn nodeReaddir(node_token: u64, cursor: u64, name_out: []u8) ?struct { header: abi.DirectoryEntryHeader, name_len: usize } {
|
||||
if (tokenKind(node_token) != token_kind_initrd_directory) return null;
|
||||
const directory_index = tokenIndex(node_token);
|
||||
if (directory_index >= directory_count) return null;
|
||||
const self_path = directories[@intCast(directory_index)].slice();
|
||||
|
||||
var index: u64 = 0;
|
||||
// Subdirectories whose parent is this directory.
|
||||
for (directories[0..directory_count], 0..) |*d, i| {
|
||||
if (i == directory_index) continue;
|
||||
if (d.parent != directory_index) continue;
|
||||
if (i == 0) continue;
|
||||
if (index == cursor) {
|
||||
const name = d.slice()[self_path.len + 1 ..];
|
||||
const n = @min(name.len, name_out.len);
|
||||
@memcpy(name_out[0..n], name[0..n]);
|
||||
return .{ .header = .{ .kind = abi.file_kind_directory, .name_len = @intCast(n), .size = 0 }, .name_len = n };
|
||||
}
|
||||
index += 1;
|
||||
}
|
||||
// Files directly inside this directory.
|
||||
const image = ramdisk_image orelse return null;
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!std.mem.eql(u8, parentOf(item.name), self_path)) continue;
|
||||
if (index == cursor) {
|
||||
const name = item.name[self_path.len + 1 ..];
|
||||
const n = @min(name.len, name_out.len);
|
||||
@memcpy(name_out[0..n], name[0..n]);
|
||||
return .{ .header = .{ .kind = abi.file_kind_regular, .name_len = @intCast(n), .size = item.blob.len }, .name_len = n };
|
||||
}
|
||||
index += 1;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- mount/unmount (syscall bodies; caller resolved the handle) --------------
|
||||
|
||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
||||
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
||||
/// shadowing or replacing /system.
|
||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
||||
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
||||
if (rewrite.len > maximum_rewrite) return false;
|
||||
if (underMount(prefix, "/system") != null) return false; // the initrd is not shadowable
|
||||
installMount(prefix, .backend, backend, rewrite);
|
||||
return true;
|
||||
}
|
||||
|
||||
pub fn unmount(prefix: []const u8) bool {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||
if (m.backend) |endpoint| ipc.dropRef(endpoint);
|
||||
m.* = .{};
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// --- tests (host) ------------------------------------------------------------
|
||||
|
||||
test "underMount matches only at path boundaries" {
|
||||
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
|
||||
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
|
||||
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null);
|
||||
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null);
|
||||
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
|
||||
try std.testing.expect(underMount("greeting", "/mnt/usb") == null);
|
||||
}
|
||||
|
||||
test "parentOf walks toward the root" {
|
||||
try std.testing.expectEqualStrings("/system/services", parentOf("/system/services/fat"));
|
||||
try std.testing.expectEqualStrings("/system", parentOf("/system/services"));
|
||||
try std.testing.expectEqualStrings("/", parentOf("/system"));
|
||||
}
|
||||
@@ -24,3 +24,9 @@ pub fn init() void {
|
||||
pub fn nowSeconds() u64 {
|
||||
return boot_unix_seconds + (architecture.nanos() -% boot_nanos) / 1_000_000_000;
|
||||
}
|
||||
|
||||
/// The wall-clock time of boot itself (the RTC anchor) — what klog_status hands
|
||||
/// the logger service to name a per-boot log directory. Zero until `init` runs.
|
||||
pub fn bootSeconds() u64 {
|
||||
return boot_unix_seconds;
|
||||
}
|
||||
|
||||
@@ -21,11 +21,6 @@ const power = runtime.power_protocol;
|
||||
/// integer decode names the opcodes instead of bare 0x0A/0x0B/… (docs/coding-standards.md).
|
||||
const opcodes = aml.opcodes;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The claimed acpi-tables node and the resource index of its broad io_port
|
||||
// window — the Hal routes every port access through this one claim.
|
||||
var node_id: u64 = 0;
|
||||
@@ -103,9 +98,11 @@ fn findTablesNode(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
// When the acpi-parse scenario spawns this directly, argv[1] is the kernel's
|
||||
// own device count to self-verify against — deterministic, no log-scraping.
|
||||
const expected: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||
// When the acpi-parse scenario spawns this directly, argv[1] is a device-count
|
||||
// *floor* to self-verify against. The kernel no longer parses AML, so there is
|
||||
// no exact count to match — proving the ring-3 parse found at least a floor of
|
||||
// devices is the check. Deterministic, no log-scraping.
|
||||
const floor: ?usize = if (init.arguments.get(1)) |a| (std.fmt.parseInt(usize, a, 10) catch null) else null;
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("/system/services/acpi: out of memory\n");
|
||||
@@ -161,12 +158,12 @@ pub fn main(init: runtime.process.Init) void {
|
||||
};
|
||||
var namespace = result.namespace;
|
||||
const devices = aml.deviceCount(&namespace);
|
||||
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||
if (expected) |want| {
|
||||
if (devices == want) {
|
||||
std.log.info("parsed {d} AML blob(s), {d} namespace devices", .{ block_count, devices });
|
||||
if (floor) |minimum| {
|
||||
if (devices >= minimum) {
|
||||
_ = runtime.system.write("acpi-parse: ok\n");
|
||||
} else {
|
||||
writeLine("acpi-parse: mismatch (ring-3 {d} vs kernel {d})\n", .{ devices, want });
|
||||
std.log.info("acpi-parse: too few (ring-3 {d} < floor {d})", .{ devices, minimum });
|
||||
}
|
||||
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
@@ -214,9 +211,9 @@ fn onInit(endpoint: runtime.ipc.Handle) bool {
|
||||
const hid = entry.hid[0..entry.hid_len];
|
||||
const desc = acpi_ids.description(hid);
|
||||
if (desc.len != 0)
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources) — {s}\n", .{ hid, entry.device_id, entry.resource_count, desc })
|
||||
std.log.info("reported {s} (device {d}, {d} resources) — {s}", .{ hid, entry.device_id, entry.resource_count, desc })
|
||||
else
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources)\n", .{ hid, entry.device_id, entry.resource_count });
|
||||
std.log.info("reported {s} (device {d}, {d} resources)", .{ hid, entry.device_id, entry.resource_count });
|
||||
if (manager) |h| {
|
||||
var report = protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
||||
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
||||
@@ -224,7 +221,7 @@ fn onInit(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
||||
}
|
||||
}
|
||||
writeLine("/system/services/acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
||||
std.log.info("reported {d} device(s) to the manager", .{registered_count});
|
||||
|
||||
armPowerButton(endpoint);
|
||||
return true;
|
||||
@@ -371,7 +368,7 @@ fn publishNotify(node: *aml.Node, code: u64) void {
|
||||
const which: power.Event = if (std.mem.eql(u8, hid[0..7], "PNP0C0A")) .battery else if (std.mem.eql(u8, hid[0..7], "ACPI0003")) .ac else if (std.mem.eql(u8, hid[0..7], "PNP0C0D")) .lid else .notify;
|
||||
var event = power.EventMessage{ .event = @intFromEnum(which), .code = @truncate(code) };
|
||||
event.hid = hid;
|
||||
writeLine("power: notify {s} code {d}\n", .{ hid[0..7], code });
|
||||
std.log.info("power: notify {s} code {d}", .{ hid[0..7], code });
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
|
||||
@@ -498,7 +495,7 @@ fn registerDevice(node: *aml.Node, hid: [8]u8, interpreter: *aml.Interpreter) vo
|
||||
applyCrs(&descriptor, node, interpreter);
|
||||
|
||||
const id = device.register(node_id, &descriptor) orelse {
|
||||
writeLine("/system/services/acpi: register refused for {s}\n", .{hid[0..@intCast(hid_len)]});
|
||||
std.log.info("register refused for {s}", .{hid[0..@intCast(hid_len)]});
|
||||
return;
|
||||
};
|
||||
registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count };
|
||||
@@ -692,8 +689,3 @@ fn rd16(bytes: []const u8, off: usize) u64 {
|
||||
fn rd32(bytes: []const u8, off: usize) u64 {
|
||||
return rd16(bytes, off) | (rd16(bytes, off + 2) << 16);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -48,8 +48,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
len += 1;
|
||||
_ = runtime.system.write(buffer[0..len]);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -16,6 +16,10 @@ pub const Operation = enum(u32) {
|
||||
read = 1,
|
||||
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
||||
write = 2,
|
||||
/// flush(): commit any device write cache to stable media (no data transfer).
|
||||
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||
flush = 3,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
|
||||
@@ -37,8 +37,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
const poison: *volatile u32 = @ptrFromInt(0xdead0000);
|
||||
poison.* = 1; // the restart machinery's fuel: a real segmentation fault
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -81,8 +81,3 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -24,14 +24,6 @@ const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const system = runtime.system;
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so output
|
||||
/// from the drivers this manager starts (which run concurrently) can never land
|
||||
/// in the middle of it.
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
||||
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
||||
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
||||
@@ -41,13 +33,23 @@ const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
||||
});
|
||||
|
||||
/// The PCI class triple of a virtio-gpu — Display Controller / Other (0x80) / 0. The class
|
||||
/// alone cannot tell it from any other display/other function, so the driver re-confirms
|
||||
/// vendor 0x1AF4 / device 0x1050 from config space once spawned; this only gets it spawned.
|
||||
const virtio_gpu_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.display),
|
||||
.subclass = 0x80, // "Other" — no named SubClass member (PCI convention)
|
||||
.prog_if = 0,
|
||||
});
|
||||
|
||||
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
||||
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
||||
/// several identical controllers — one driver instance per reported device,
|
||||
/// its registered id as argv[1].
|
||||
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
return switch (identity) {
|
||||
xhci_pci_class => "usb-xhci-bus",
|
||||
xhci_pci_class => "/system/drivers/usb-xhci-bus",
|
||||
virtio_gpu_pci_class => "/system/drivers/virtio-gpu",
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
@@ -57,8 +59,8 @@ fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
/// nodes the kernel used to build). ps2-bus is a singleton that finds both its
|
||||
/// devices by hid once spawned, so keyboard and mouse map to the same name.
|
||||
fn hidDriverFor(hid: []const u8) ?[]const u8 {
|
||||
if (std.mem.eql(u8, hid, "PNP0303")) return "ps2-bus"; // PS/2 keyboard
|
||||
if (std.mem.eql(u8, hid, "PNP0F13")) return "ps2-bus"; // PS/2 mouse
|
||||
if (std.mem.eql(u8, hid, "PNP0303")) return "/system/drivers/ps2-bus"; // PS/2 keyboard
|
||||
if (std.mem.eql(u8, hid, "PNP0F13")) return "/system/drivers/ps2-bus"; // PS/2 mouse
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -85,9 +87,9 @@ fn usbDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
@intFromEnum(usb_ids.mass_storage.Protocol.bulk_only),
|
||||
);
|
||||
return switch (identity) {
|
||||
keyboard => "usb-hid-keyboard",
|
||||
mouse => "usb-hid-mouse",
|
||||
storage => "usb-storage",
|
||||
keyboard => "/system/drivers/usb-hid-keyboard",
|
||||
mouse => "/system/drivers/usb-hid-mouse",
|
||||
storage => "/system/drivers/usb-storage",
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
@@ -122,7 +124,7 @@ const DriverState = enum {
|
||||
|
||||
const Driver = struct {
|
||||
used: bool = false,
|
||||
name_buffer: [24]u8 = undefined,
|
||||
name_buffer: [64]u8 = undefined, // fits a full binary path (abi.maximum_process_name)
|
||||
name_len: usize = 0,
|
||||
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
||||
device_id: u64 = protocol.no_device,
|
||||
@@ -149,6 +151,8 @@ var test_restart_mode = false;
|
||||
var test_usb_restart_mode = false;
|
||||
var test_usb_killed = false;
|
||||
var test_pci_restart_mode = false;
|
||||
var test_scanout_restart_mode = false;
|
||||
var test_scanout_killed = false;
|
||||
var test_kill_pid: u32 = 0;
|
||||
var test_kill_due_ns: u64 = 0;
|
||||
|
||||
@@ -211,7 +215,7 @@ fn addChild(parent: u64, bus_address: u64, identity: u64, device_id: u64, report
|
||||
fn pruneChildrenOf(reporter: u32) void {
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.reporter == reporter) {
|
||||
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
|
||||
child.used = false;
|
||||
const event = protocol.ChildRemoved{ .parent = child.parent, .bus_address = child.bus_address };
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
@@ -258,7 +262,7 @@ fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
|
||||
spawnDriver(driver);
|
||||
return;
|
||||
}
|
||||
writeLine("/system/services/device-manager: driver table full; cannot supervise {s}\n", .{name});
|
||||
std.log.info("driver table full; cannot supervise {s}", .{name});
|
||||
}
|
||||
|
||||
/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the
|
||||
@@ -273,7 +277,7 @@ fn spawnDriver(driver: *Driver) void {
|
||||
argument_count = 1;
|
||||
}
|
||||
const child = system.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
||||
writeLine("/system/services/device-manager: failed to spawn {s}\n", .{driver.name()});
|
||||
std.log.info("failed to spawn {s}", .{driver.name()});
|
||||
driver.state = .failed;
|
||||
return;
|
||||
};
|
||||
@@ -287,9 +291,9 @@ fn spawnDriver(driver: *Driver) void {
|
||||
driver.state = .running;
|
||||
}
|
||||
if (driver.device_id != protocol.no_device) {
|
||||
writeLine("/system/services/device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id });
|
||||
std.log.info("spawned {s} for device {d}", .{ driver.name(), driver.device_id });
|
||||
} else {
|
||||
writeLine("/system/services/device-manager: spawned {s}\n", .{driver.name()});
|
||||
std.log.info("spawned {s}", .{driver.name()});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -301,7 +305,7 @@ fn onDriverExit(driver: *Driver) void {
|
||||
const reason = runtime.process.exitReason(driver.process_id) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
driver.state = .stopped;
|
||||
writeLine("/system/services/device-manager: {s} exited cleanly; not restarting\n", .{driver.name()});
|
||||
std.log.info("{s} exited cleanly; not restarting", .{driver.name()});
|
||||
return;
|
||||
}
|
||||
const now = system.clock();
|
||||
@@ -309,13 +313,13 @@ fn onDriverExit(driver: *Driver) void {
|
||||
driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1;
|
||||
if (driver.restarts >= crash_loop_cap) {
|
||||
driver.state = .failed;
|
||||
writeLine("/system/services/device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()});
|
||||
std.log.info("{s} is failing repeatedly (crash loop); giving up", .{driver.name()});
|
||||
return;
|
||||
}
|
||||
const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1);
|
||||
driver.state = .restarting;
|
||||
driver.restart_due_ns = now + delay_ms * 1_000_000;
|
||||
writeLine("/system/services/device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||
std.log.info("restarting {s} in {d} ms (died: {s})", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||
_ = system.timerOnce(manager_endpoint, delay_ms + 50);
|
||||
}
|
||||
|
||||
@@ -326,7 +330,7 @@ fn onDriverExit(driver: *Driver) void {
|
||||
fn sweepDeadlines() void {
|
||||
const now = system.clock();
|
||||
if (test_kill_pid != 0 and now >= test_kill_due_ns) {
|
||||
writeLine("/system/services/device-manager: test mode: killing the reporter\n", .{});
|
||||
std.log.info("test mode: killing the reporter", .{});
|
||||
_ = system.kill(test_kill_pid);
|
||||
test_kill_pid = 0;
|
||||
}
|
||||
@@ -334,7 +338,7 @@ fn sweepDeadlines() void {
|
||||
if (!driver.used) continue;
|
||||
switch (driver.state) {
|
||||
.awaiting_hello => if (now >= driver.hello_deadline_ns) {
|
||||
writeLine("/system/services/device-manager: {s} missed its hello deadline\n", .{driver.name()});
|
||||
std.log.info("{s} missed its hello deadline", .{driver.name()});
|
||||
_ = system.kill(driver.process_id);
|
||||
// The exit notification finishes the job via onDriverExit.
|
||||
},
|
||||
@@ -411,13 +415,21 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
var status: i32 = 0;
|
||||
if (hello.version != protocol.version) {
|
||||
status = -1;
|
||||
writeLine("/system/services/device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender });
|
||||
std.log.info("refused hello (version {d}) from process {d}", .{ hello.version, sender });
|
||||
} else if (driverByProcess(sender)) |driver| {
|
||||
driver.state = .running;
|
||||
writeLine("/system/services/device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||
std.log.info("hello from {s} (device {d})", .{ driver.name(), hello.device_id });
|
||||
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
|
||||
// the normal restart policy respawns it — the compositor must survive and re-attach.
|
||||
if (test_scanout_restart_mode and !test_scanout_killed and std.mem.eql(u8, driver.name(), "/system/drivers/virtio-gpu")) {
|
||||
test_scanout_killed = true;
|
||||
test_kill_pid = sender;
|
||||
test_kill_due_ns = system.clock() + 1_500_000_000;
|
||||
_ = system.timerOnce(manager_endpoint, 1600);
|
||||
}
|
||||
} else {
|
||||
status = -1;
|
||||
writeLine("/system/services/device-manager: hello from unknown process {d}\n", .{sender});
|
||||
std.log.info("hello from unknown process {d}", .{sender});
|
||||
}
|
||||
const hello_reply = protocol.HelloReply{ .status = status };
|
||||
@memcpy(reply[0..protocol.reply_size], std.mem.asBytes(&hello_reply));
|
||||
@@ -433,7 +445,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
var status: i32 = 0;
|
||||
if (driverByProcess(sender)) |driver| {
|
||||
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||
writeLine("/system/services/device-manager: child added (device {d} port {d}, identity {d}) by {s}\n", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||
if (status == 0) publishEvent(message[0..protocol.child_added_size]);
|
||||
// Matching from reports (M19.3): a registered child whose identity
|
||||
// names a driver gets one, once — re-reports after a bus restart
|
||||
@@ -478,7 +490,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
// Only the xHCI reporter is the drill's victim — pci-bus also reports
|
||||
// now, and whichever finishes second must not trigger the kill.
|
||||
if (driverByProcess(sender)) |driver| {
|
||||
if (std.mem.eql(u8, driver.name(), "usb-xhci-bus")) {
|
||||
if (std.mem.eql(u8, driver.name(), "/system/drivers/usb-xhci-bus")) {
|
||||
// Delayed, not immediate: the device-list scenario's subscriber
|
||||
// needs a window to enumerate and subscribe before the events.
|
||||
test_usb_killed = true;
|
||||
@@ -499,7 +511,7 @@ fn onChildRemoved(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
var status: i32 = -1;
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == sender) {
|
||||
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
|
||||
child.used = false;
|
||||
status = 0;
|
||||
}
|
||||
@@ -557,6 +569,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
|
||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||
}
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.service = .device_manager,
|
||||
@@ -565,8 +578,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
//! system/services/display-demo — a hardware-free client of the display service, the
|
||||
//! `input-source` analog for the compositor. It creates a wallpaper and a rectangle it
|
||||
//! slides each frame, then drives the compositor in a present loop — proof that a
|
||||
//! *separate process* can compose a moving scene through the display service over IPC,
|
||||
//! exercising the layer client API and damage-driven present end to end
|
||||
//! (docs/display.md). It logs `display-demo: ok` once it has driven a run of frames.
|
||||
//!
|
||||
//! It draws no cursor and reads no input: the on-screen cursor is the display service's
|
||||
//! own, tracked by the service's mouse-listener thread (docs/display.md). The demo's job
|
||||
//! is only to prove client-driven animation, so its loop runs on its own frame timer and
|
||||
//! is deliberately independent of the mouse.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const display = runtime.display;
|
||||
const system = runtime.system;
|
||||
const time = runtime.time;
|
||||
|
||||
pub fn main() void {
|
||||
const mode = display.info() orelse {
|
||||
_ = system.write("display-demo: no display service\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// A full-screen wallpaper under everything.
|
||||
const wallpaper = display.createLayer(0, 0, mode.width, mode.height, 0) orelse return createFailed();
|
||||
_ = wallpaper.fill(0, 0, mode.width, mode.height, display.color(0x10, 0x18, 0x28));
|
||||
|
||||
// A rectangle that slides back and forth.
|
||||
const box_w: u32 = 140;
|
||||
const box_h: u32 = 100;
|
||||
const box_y: i32 = 200;
|
||||
const box = display.createLayer(0, box_y, box_w, box_h, 1) orelse return createFailed();
|
||||
_ = box.fill(0, 0, box_w, box_h, display.color(0xE0, 0x60, 0x40));
|
||||
|
||||
_ = display.present();
|
||||
_ = system.write("display-demo: scene up; animating\n");
|
||||
|
||||
const span: i32 = @as(i32, @intCast(mode.width)) - @as(i32, @intCast(box_w));
|
||||
var x: i32 = 0;
|
||||
var dx: i32 = 8;
|
||||
var frame: u32 = 0;
|
||||
|
||||
while (true) : (frame += 1) {
|
||||
x += dx;
|
||||
if (x <= 0) {
|
||||
x = 0;
|
||||
dx = -dx;
|
||||
} else if (x >= span) {
|
||||
x = span;
|
||||
dx = -dx;
|
||||
}
|
||||
_ = box.configure(x, box_y, 1, true); // move it; the compositor repaints old + new
|
||||
_ = display.present();
|
||||
// A run of frames drawn through the compositor is the automated proof (the visible
|
||||
// motion is a screenshot away via `zig build run-x86-64`).
|
||||
if (frame == 20) _ = system.write("display-demo: ok\n");
|
||||
time.sleep(time.Duration.fromMillis(30));
|
||||
}
|
||||
}
|
||||
|
||||
fn createFailed() void {
|
||||
_ = system.write("display-demo: create failed\n");
|
||||
}
|
||||
@@ -0,0 +1,309 @@
|
||||
//! The compositor's **scanout backend** — how a finished frame reaches the panel
|
||||
//! (docs/display-v2.md). The compositor composes its layer stack into the backend's
|
||||
//! cacheable `surface()` and calls `present(damage)`; everything device-specific lives
|
||||
//! here. Today there is one backend, `Gop` — the firmware framebuffer: a cacheable back
|
||||
//! buffer streamed write-combining to the linear framebuffer. A native virtio-gpu backend
|
||||
//! slots in beside it later (V4); the compositor never learns which is active.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const compositor = @import("compositor.zig");
|
||||
|
||||
const system = runtime.system;
|
||||
const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
const scanout_protocol = runtime.scanout_protocol;
|
||||
const Rect = compositor.Rect;
|
||||
const Surface = compositor.Surface;
|
||||
|
||||
/// The current display mode, as a backend reports it. `refresh_hz` is the panel's
|
||||
/// refresh rate from EDID (0 = unknown) — the frame clock's pacing seed; without vblank
|
||||
/// it fixes the rate, never the phase (docs/display-v2.md, "Fenced is not vsync").
|
||||
pub const Info = struct { width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32 };
|
||||
|
||||
/// Enumeration scratch — a `DeviceDescriptor` is large, and only one scan is ever needed.
|
||||
var device_table: [64]device.DeviceDescriptor = undefined;
|
||||
|
||||
/// The GOP framebuffer backend: claims the kernel-seeded `display` device, maps the linear
|
||||
/// framebuffer write-combining as the front buffer, and keeps a cacheable back buffer of
|
||||
/// the same geometry as the compose target. `present` streams the damaged rectangle from
|
||||
/// the back buffer to the LFB (sequential WC writes; the LFB is never read). No mode-set,
|
||||
/// no present fence — the portable floor (docs/display-v2.md).
|
||||
pub const Gop = struct {
|
||||
device_id: u64,
|
||||
front: [*]volatile u8, // the LFB (write-combining)
|
||||
back: [*]u8, // cacheable compose target, same geometry
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32,
|
||||
format: u32,
|
||||
refresh_hz: u32, // from the boot EDID via the display0 node (0 = unknown)
|
||||
|
||||
/// The framebuffer's id and geometry, captured together. `findDisplay` reads these out of
|
||||
/// the enumeration table and returns them by value, so the caller never re-reads the table
|
||||
/// across later syscalls (`device_enumerate` writes the whole table straight into this
|
||||
/// process's memory; reading a descriptor's tail again after other syscalls have run is a
|
||||
/// window we simply avoid by copying the few fields we need up front).
|
||||
const Found = struct { id: u64, width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32 };
|
||||
|
||||
/// The first `display`-class device with a *valid* (non-zero) geometry, or null. A zero
|
||||
/// geometry is treated as "not ready yet" so the caller retries — a real framebuffer always
|
||||
/// has a non-zero width, height, and pitch.
|
||||
fn findDisplay() ?Found {
|
||||
const total = device.enumerate(&device_table);
|
||||
const n = @min(total, device_table.len);
|
||||
for (device_table[0..n]) |*d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.display)) continue;
|
||||
if (d.display.width == 0 or d.display.height == 0 or d.display.pitch == 0) continue;
|
||||
return .{ .id = d.id, .width = d.display.width, .height = d.display.height, .pitch = d.display.pitch, .format = d.display.format, .refresh_hz = d.display.refresh_hz };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Claim the framebuffer (retrying while discovery catches up), map the LFB, and
|
||||
/// allocate the back buffer. Null if there is no framebuffer or a mapping fails.
|
||||
pub fn init() ?Gop {
|
||||
var tries: u32 = 0;
|
||||
const found = while (tries < 100) : (tries += 1) {
|
||||
if (findDisplay()) |f| break f;
|
||||
system.sleep(50);
|
||||
} else {
|
||||
_ = system.write("display: no framebuffer device (headless?)\n");
|
||||
return null;
|
||||
};
|
||||
|
||||
if (!device.claim(found.id)) {
|
||||
_ = system.write("display: could not claim the framebuffer\n");
|
||||
return null;
|
||||
}
|
||||
// Resource 0 is the framebuffer memory window; the kernel maps it write-combining
|
||||
// because the resource carries that flag (docs/display-plan.md D1).
|
||||
const front_base = device.mmioMap(found.id, 0) orelse {
|
||||
_ = system.write("display: could not map the framebuffer\n");
|
||||
return null;
|
||||
};
|
||||
const size = @as(usize, found.height) * found.pitch;
|
||||
const back_base = system.mmap(size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(back_base)) {
|
||||
_ = system.write("display: could not allocate the back buffer\n");
|
||||
return null;
|
||||
}
|
||||
return .{
|
||||
.device_id = found.id,
|
||||
.front = @ptrFromInt(front_base),
|
||||
.back = @ptrFromInt(back_base),
|
||||
.width = found.width,
|
||||
.height = found.height,
|
||||
.pitch = found.pitch,
|
||||
.format = found.format,
|
||||
.refresh_hz = found.refresh_hz,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn info(self: *const Gop) Info {
|
||||
return .{ .width = self.width, .height = self.height, .pitch = self.pitch, .format = self.format, .refresh_hz = self.refresh_hz };
|
||||
}
|
||||
|
||||
/// The cacheable compose target (the back buffer).
|
||||
pub fn surface(self: *const Gop) Surface {
|
||||
return .{
|
||||
.pixels = @ptrCast(@alignCast(self.back)),
|
||||
.stride = self.pitch / 4, // pitch is bytes; a 32-bpp row is pitch/4 pixels
|
||||
.width = self.width,
|
||||
.height = self.height,
|
||||
};
|
||||
}
|
||||
|
||||
/// Stream each damaged rectangle from the back buffer to the write-combining LFB, row
|
||||
/// by row (sequential writes — what WC memory wants; the LFB is never read). The rows
|
||||
/// are copied by `presentSpan` below, which widens the stores by hand: `volatile`
|
||||
/// keeps the compiler from eliding or reordering framebuffer writes, but it also
|
||||
/// forbids it from merging them, so a naive per-pixel loop is stuck at one 4-byte
|
||||
/// store per iteration. Keeping each copy small (the damage list) and each store wide
|
||||
/// shrinks the window in which scanout can sample a half-written frame.
|
||||
pub fn present(self: *const Gop, damage: []const Rect) void {
|
||||
const bounds = Rect{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) };
|
||||
for (damage) |rect| {
|
||||
const c = rect.intersect(bounds);
|
||||
if (c.isEmpty()) continue;
|
||||
const span: usize = @intCast(c.w);
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
const offset = @as(usize, @intCast(y)) * self.pitch + @as(usize, @intCast(c.x)) * 4;
|
||||
const source: [*]const u32 = @ptrCast(@alignCast(self.back + offset));
|
||||
const front_row: [*]volatile u32 = @ptrCast(@alignCast(self.front + offset));
|
||||
presentSpan(front_row, source, span);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// Copy `count` pixels into the write-combining front buffer with 8-byte volatile stores
|
||||
/// (plus a 4-byte head/tail where the span isn't 8-aligned — pixel spans are always
|
||||
/// 4-aligned). The loads come from the cacheable back buffer and are assembled into a
|
||||
/// `u64` in registers, so nothing here reads the front buffer.
|
||||
fn presentSpan(destination: [*]volatile u32, source: [*]const u32, count: usize) void {
|
||||
var i: usize = 0;
|
||||
if (i < count and (@intFromPtr(destination) & 7) != 0) {
|
||||
destination[0] = source[0];
|
||||
i = 1;
|
||||
}
|
||||
while (i + 2 <= count) : (i += 2) {
|
||||
const pair = @as(u64, source[i]) | (@as(u64, source[i + 1]) << 32);
|
||||
const wide: *volatile u64 = @ptrCast(@alignCast(destination + i));
|
||||
wide.* = pair;
|
||||
}
|
||||
if (i < count) destination[i] = source[i];
|
||||
}
|
||||
|
||||
/// A display mode the native backend can switch to.
|
||||
pub const Mode = scanout_protocol.Mode;
|
||||
|
||||
/// The native virtio-gpu backend: the compositor composes into a **shared** scanout surface
|
||||
/// (a shared-memory region the driver created and handed over) and `present` asks the driver to put
|
||||
/// a frame on the panel over its `.scanout` endpoint. Unlike GOP there is no local copy — the
|
||||
/// surface *is* the device's resource backing, so compositing writes land straight where the
|
||||
/// driver transfers-and-flushes from (x86 DMA is cache-coherent, so the cacheable shared pages
|
||||
/// need no explicit flush). Built by the display service when a driver announces (V4). The
|
||||
/// surface is sized to the driver's largest mode, so `stride` (its row width) is fixed while
|
||||
/// `width`/`height` — the active mode — change under `setMode` (V5).
|
||||
pub const VirtioGpu = struct {
|
||||
pixels: [*]u32, // the shared scanout surface, mapped into the compositor
|
||||
stride: u32, // the surface's row stride in pixels (the driver's max mode width) — fixed
|
||||
width: u32, // the active mode
|
||||
height: u32,
|
||||
format: u32,
|
||||
refresh_hz: u32, // from the driver's EDID read, carried in the announce (0 = unknown)
|
||||
scanout: ipc.Handle, // the driver's present + mode channel (looked up on `.scanout`)
|
||||
|
||||
pub fn info(self: *const VirtioGpu) Info {
|
||||
return .{ .width = self.width, .height = self.height, .pitch = self.stride * 4, .format = self.format, .refresh_hz = self.refresh_hz };
|
||||
}
|
||||
pub fn surface(self: *const VirtioGpu) Surface {
|
||||
return .{ .pixels = self.pixels, .stride = self.stride, .width = self.width, .height = self.height };
|
||||
}
|
||||
/// Ask the driver to present. The composited pixels are already in the shared surface, so
|
||||
/// this is a single request over `.scanout` regardless of how many damage rectangles
|
||||
/// accumulated; the driver transfers + fenced-flushes the whole frame.
|
||||
pub fn present(self: *const VirtioGpu, damage: []const Rect) void {
|
||||
_ = damage;
|
||||
var request = scanout_protocol.Request{
|
||||
.operation = @intFromEnum(scanout_protocol.Operation.present),
|
||||
.width = self.width,
|
||||
.height = self.height,
|
||||
};
|
||||
var reply: [scanout_protocol.reply_size]u8 = undefined;
|
||||
_ = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch {};
|
||||
}
|
||||
/// Fill `out` with the driver's offered modes; returns how many were written.
|
||||
pub fn modes(self: *const VirtioGpu, out: []Mode) usize {
|
||||
var request = scanout_protocol.Request{ .operation = @intFromEnum(scanout_protocol.Operation.get_modes) };
|
||||
var reply: [scanout_protocol.modes_reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (n < scanout_protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(scanout_protocol.ModesReply, reply[0..scanout_protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, scanout_protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
/// Change the scanout resolution. On success the active `width`/`height` update (the shared
|
||||
/// surface — sized to the max mode — is unchanged, so `stride` stays put).
|
||||
pub fn setMode(self: *VirtioGpu, w: u32, h: u32) bool {
|
||||
if (w == 0 or h == 0 or w > self.stride) return false;
|
||||
var request = scanout_protocol.Request{
|
||||
.operation = @intFromEnum(scanout_protocol.Operation.set_mode),
|
||||
.width = w,
|
||||
.height = h,
|
||||
};
|
||||
var reply: [scanout_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (n < scanout_protocol.reply_size) return false;
|
||||
if (std.mem.bytesToValue(scanout_protocol.Reply, reply[0..scanout_protocol.reply_size]).status != 0) return false;
|
||||
self.width = w;
|
||||
self.height = h;
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
/// The pluggable scanout backend. A tagged union so the compositor holds one value and
|
||||
/// dispatches without caring which is active; the `virtio` native backend joins `gop` at V4.
|
||||
pub const Backend = union(enum) {
|
||||
gop: Gop,
|
||||
virtio: VirtioGpu,
|
||||
|
||||
pub fn info(self: *const Backend) Info {
|
||||
return switch (self.*) {
|
||||
inline else => |*b| b.info(),
|
||||
};
|
||||
}
|
||||
pub fn surface(self: *const Backend) Surface {
|
||||
return switch (self.*) {
|
||||
inline else => |*b| b.surface(),
|
||||
};
|
||||
}
|
||||
pub fn present(self: *const Backend, damage: []const Rect) void {
|
||||
switch (self.*) {
|
||||
inline else => |*b| b.present(damage),
|
||||
}
|
||||
}
|
||||
/// The modes this backend can switch to (none for GOP); returns how many were written.
|
||||
pub fn modes(self: *const Backend, out: []Mode) usize {
|
||||
return switch (self.*) {
|
||||
.virtio => |*v| v.modes(out),
|
||||
.gop => 0,
|
||||
};
|
||||
}
|
||||
/// Change the resolution; false if this backend can't mode-set or the mode was refused.
|
||||
pub fn setMode(self: *Backend, w: u32, h: u32) bool {
|
||||
return switch (self.*) {
|
||||
.virtio => |*v| v.setMode(w, h),
|
||||
.gop => false,
|
||||
};
|
||||
}
|
||||
/// Whether this backend supports runtime mode-setting (GOP: no; virtio-gpu: yes, V5).
|
||||
pub fn canModeSet(self: *const Backend) bool {
|
||||
return switch (self.*) {
|
||||
.gop => false,
|
||||
.virtio => true,
|
||||
};
|
||||
}
|
||||
/// Whether this backend's present is **fenced** — it completes only once the device has
|
||||
/// consumed the frame (virtio-gpu: every flush carries a fence the used-ring ack waits on).
|
||||
/// A fence gives completion feedback and tear-free snapshot presents; it is *not* vblank —
|
||||
/// nothing paces presents to the display's refresh (base virtio-gpu 2D has no vblank event
|
||||
/// at all). True vsync needs a native driver's vblank interrupt. See docs/display-v2.md,
|
||||
/// "Fenced is not vsync".
|
||||
pub fn hasFencedPresent(self: *const Backend) bool {
|
||||
return switch (self.*) {
|
||||
.gop => false,
|
||||
.virtio => true,
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// Which backend to use. The pure selection *decision* is `chooseKind`; `select` below
|
||||
/// binds it to the (syscall-bound) bring-up.
|
||||
pub const Kind = enum { gop, virtio };
|
||||
|
||||
/// The selection decision, factored out of bring-up so it stays pure and host-testable:
|
||||
/// prefer a native driver when one has announced itself (docs/display-v2.md V4), else the
|
||||
/// GOP floor. Trivial today; it grows real inputs when native detection lands.
|
||||
pub fn chooseKind(native_available: bool) Kind {
|
||||
return if (native_available) .virtio else .gop;
|
||||
}
|
||||
|
||||
/// Pick and bring up the best available backend. Today the GOP framebuffer is the only one
|
||||
/// (`chooseKind(false)` → `.gop`), so this is `Gop.init()`. V4 adds the native-if-present
|
||||
/// branch, with GOP as the floor.
|
||||
pub fn select() ?Backend {
|
||||
return switch (chooseKind(false)) {
|
||||
.gop => .{ .gop = Gop.init() orelse return null },
|
||||
.virtio => unreachable, // no native detection yet (V4)
|
||||
};
|
||||
}
|
||||
|
||||
test "selection prefers native when present, else the gop floor" {
|
||||
try std.testing.expectEqual(Kind.gop, chooseKind(false));
|
||||
try std.testing.expectEqual(Kind.virtio, chooseKind(true));
|
||||
}
|
||||
@@ -0,0 +1,452 @@
|
||||
//! The compositor's pure core: rectangle math and the three blitting primitives the
|
||||
//! display service composes frames from — fill a rectangle of a surface, composite one
|
||||
//! surface onto another clipped to a damage rectangle, and copy a client-supplied pixel
|
||||
//! tile in. Deliberately free of any syscall or `runtime` dependency (it takes plain
|
||||
//! pixel pointers), so it is host-tested under `zig build test`. The service
|
||||
//! (system/services/display/display.zig) wires real mmap'd surfaces and the framebuffer
|
||||
//! to it. Pixels are opaque native 32-bit values — v1 layers don't alpha-blend, and
|
||||
//! channel order (rgbx/bgrx) is the caller's concern (see protocol.pack).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// An axis-aligned rectangle in pixels. Signed, so a surface partly off-screen (a layer
|
||||
/// dragged past an edge) clips with plain arithmetic. Half-open: covers [x, x+w) × [y, y+h).
|
||||
pub const Rect = struct {
|
||||
x: i32,
|
||||
y: i32,
|
||||
w: i32,
|
||||
h: i32,
|
||||
|
||||
pub const empty = Rect{ .x = 0, .y = 0, .w = 0, .h = 0 };
|
||||
|
||||
pub fn init(x: i32, y: i32, w: i32, h: i32) Rect {
|
||||
return .{ .x = x, .y = y, .w = w, .h = h };
|
||||
}
|
||||
|
||||
pub fn isEmpty(r: Rect) bool {
|
||||
return r.w <= 0 or r.h <= 0;
|
||||
}
|
||||
|
||||
pub fn right(r: Rect) i32 {
|
||||
return r.x + r.w;
|
||||
}
|
||||
pub fn bottom(r: Rect) i32 {
|
||||
return r.y + r.h;
|
||||
}
|
||||
|
||||
/// The overlap of two rectangles, or an empty rectangle if they don't touch.
|
||||
pub fn intersect(a: Rect, b: Rect) Rect {
|
||||
const x0 = @max(a.x, b.x);
|
||||
const y0 = @max(a.y, b.y);
|
||||
const x1 = @min(a.right(), b.right());
|
||||
const y1 = @min(a.bottom(), b.bottom());
|
||||
return .{ .x = x0, .y = y0, .w = x1 - x0, .h = y1 - y0 };
|
||||
}
|
||||
|
||||
/// The bounding box of two rectangles. An empty operand contributes nothing (returns
|
||||
/// the other), so folding damage rectangles with `unite` from `empty` yields their
|
||||
/// bounding box.
|
||||
pub fn unite(a: Rect, b: Rect) Rect {
|
||||
if (a.isEmpty()) return b;
|
||||
if (b.isEmpty()) return a;
|
||||
const x0 = @min(a.x, b.x);
|
||||
const y0 = @min(a.y, b.y);
|
||||
const x1 = @max(a.right(), b.right());
|
||||
const y1 = @max(a.bottom(), b.bottom());
|
||||
return .{ .x = x0, .y = y0, .w = x1 - x0, .h = y1 - y0 };
|
||||
}
|
||||
};
|
||||
|
||||
/// The dirty screen regions accumulated between presents. Kept as a *list* of rectangles,
|
||||
/// not one bounding box: when two small things move far apart — the cursor on one side of
|
||||
/// the screen, an animating layer on the other — a single bounding box unites them into a
|
||||
/// huge region, and presenting it streams megabytes to the framebuffer for a few thousand
|
||||
/// changed pixels. The long copy widens the window in which scanout (or QEMU's display
|
||||
/// refresh) samples a half-written frame — visible as tearing and cursor trails. Small
|
||||
/// separate rectangles keep each copy, and that window, tight.
|
||||
///
|
||||
/// A new rectangle that overlaps an existing entry is united into it (repainting a modest
|
||||
/// superset is harmless — compositing is idempotent); the grown entry is *not* re-merged
|
||||
/// against the rest, so entries may overlap, which costs only a duplicate repaint. When
|
||||
/// the table is full the newcomer folds into the last entry — degrading toward the old
|
||||
/// bounding-box behaviour instead of dropping damage.
|
||||
pub const DamageList = struct {
|
||||
pub const capacity = 16;
|
||||
|
||||
rects: [capacity]Rect = [_]Rect{Rect.empty} ** capacity,
|
||||
count: usize = 0,
|
||||
|
||||
pub fn add(self: *DamageList, r: Rect) void {
|
||||
if (r.isEmpty()) return;
|
||||
for (self.rects[0..self.count]) |*existing| {
|
||||
if (!existing.intersect(r).isEmpty()) {
|
||||
existing.* = existing.unite(r);
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (self.count < capacity) {
|
||||
self.rects[self.count] = r;
|
||||
self.count += 1;
|
||||
return;
|
||||
}
|
||||
self.rects[capacity - 1] = self.rects[capacity - 1].unite(r);
|
||||
}
|
||||
|
||||
pub fn isEmpty(self: *const DamageList) bool {
|
||||
return self.count == 0;
|
||||
}
|
||||
|
||||
pub fn slice(self: *const DamageList) []const Rect {
|
||||
return self.rects[0..self.count];
|
||||
}
|
||||
|
||||
pub fn clear(self: *DamageList) void {
|
||||
self.count = 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// The alternative damage tracker: a **fixed tile grid**, the scheme browser compositors
|
||||
/// and tile-based GPUs use. The screen is divided into `tile_size`-pixel tiles up front;
|
||||
/// `add` marks the tiles a rectangle touches (a bit per tile — merging is free and exact,
|
||||
/// no heuristics), and `collect` walks the grid turning runs of adjacent dirty tiles into
|
||||
/// repaint rectangles (horizontal runs, then equal-span rows merged vertically, so
|
||||
/// full-screen damage collapses back to a single rectangle).
|
||||
///
|
||||
/// Trade-off against `DamageList`: tracking is O(1) with a strictly bounded worst case
|
||||
/// (never more than the dirty tiles), but repaints are quantized — a 1-pixel change
|
||||
/// repaints a whole tile. Which wins depends on the workload; the display service has a
|
||||
/// compile-time switch (`damage_mode`) to compare them.
|
||||
pub const TileGrid = struct {
|
||||
pub const tile_size = 64;
|
||||
pub const maximum_columns = 128; // supports screens up to 8192 px wide…
|
||||
pub const maximum_rows = 128; // …and 8192 px tall (beyond that, edge tiles stretch)
|
||||
pub const maximum_tiles = maximum_columns * maximum_rows;
|
||||
/// The most rectangles `collect` produces; extras fold into the last (never dropped).
|
||||
pub const maximum_rects = 64;
|
||||
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
columns: u32 = 0,
|
||||
rows: u32 = 0,
|
||||
dirty_count: u32 = 0,
|
||||
dirty: [maximum_tiles]bool = [_]bool{false} ** maximum_tiles,
|
||||
|
||||
/// Size the grid for a screen. Also clears it — callers reset on a geometry change,
|
||||
/// where the mode-set paths damage the whole new screen anyway.
|
||||
pub fn reset(self: *TileGrid, width: u32, height: u32) void {
|
||||
self.width = width;
|
||||
self.height = height;
|
||||
self.columns = @min((width + tile_size - 1) / tile_size, maximum_columns);
|
||||
self.rows = @min((height + tile_size - 1) / tile_size, maximum_rows);
|
||||
self.clear();
|
||||
}
|
||||
|
||||
pub fn matches(self: *const TileGrid, width: u32, height: u32) bool {
|
||||
return self.width == width and self.height == height;
|
||||
}
|
||||
|
||||
pub fn isEmpty(self: *const TileGrid) bool {
|
||||
return self.dirty_count == 0;
|
||||
}
|
||||
|
||||
pub fn clear(self: *TileGrid) void {
|
||||
@memset(&self.dirty, false);
|
||||
self.dirty_count = 0;
|
||||
}
|
||||
|
||||
/// Mark every tile `r` touches. Clips to the screen first, so out-of-range
|
||||
/// rectangles are harmless.
|
||||
pub fn add(self: *TileGrid, r: Rect) void {
|
||||
const screen = Rect{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) };
|
||||
const c = r.intersect(screen);
|
||||
if (c.isEmpty()) return;
|
||||
const column_first: u32 = @intCast(@divTrunc(c.x, tile_size));
|
||||
const row_first: u32 = @intCast(@divTrunc(c.y, tile_size));
|
||||
const column_last: u32 = @min(@as(u32, @intCast(@divTrunc(c.right() - 1, tile_size))), self.columns - 1);
|
||||
const row_last: u32 = @min(@as(u32, @intCast(@divTrunc(c.bottom() - 1, tile_size))), self.rows - 1);
|
||||
var row = row_first;
|
||||
while (row <= row_last) : (row += 1) {
|
||||
var column = column_first;
|
||||
while (column <= column_last) : (column += 1) {
|
||||
const index = row * self.columns + column;
|
||||
if (!self.dirty[index]) {
|
||||
self.dirty[index] = true;
|
||||
self.dirty_count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The screen rectangle covered by tiles [column_first, column_end) of `row`. Edge
|
||||
/// tiles clamp to the true screen size (the last column/row may be partial — or, on a
|
||||
/// screen wider than the grid supports, stretched to cover the remainder).
|
||||
fn tileSpanRect(self: *const TileGrid, column_first: u32, column_end: u32, row: u32) Rect {
|
||||
const x: i32 = @intCast(column_first * tile_size);
|
||||
const y: i32 = @intCast(row * tile_size);
|
||||
const right: i32 = if (column_end >= self.columns) @intCast(self.width) else @intCast(column_end * tile_size);
|
||||
const bottom: i32 = if (row + 1 >= self.rows) @intCast(self.height) else @intCast((row + 1) * tile_size);
|
||||
return .{ .x = x, .y = y, .w = right - x, .h = bottom - y };
|
||||
}
|
||||
|
||||
/// Turn the dirty tiles into repaint rectangles in `out`: coalesce each row's runs of
|
||||
/// adjacent dirty tiles, then merge a run into the rectangle directly above it when
|
||||
/// the spans match — so a dirty block of tiles becomes one rectangle. Returns the
|
||||
/// filled prefix of `out`.
|
||||
pub fn collect(self: *const TileGrid, out: []Rect) []Rect {
|
||||
var count: usize = 0;
|
||||
var row: u32 = 0;
|
||||
while (row < self.rows) : (row += 1) {
|
||||
var column: u32 = 0;
|
||||
while (column < self.columns) {
|
||||
if (!self.dirty[row * self.columns + column]) {
|
||||
column += 1;
|
||||
continue;
|
||||
}
|
||||
var run_end = column + 1;
|
||||
while (run_end < self.columns and self.dirty[row * self.columns + run_end]) run_end += 1;
|
||||
const rect = self.tileSpanRect(column, run_end, row);
|
||||
column = run_end;
|
||||
|
||||
var merged = false;
|
||||
for (out[0..count]) |*existing| {
|
||||
if (existing.x == rect.x and existing.w == rect.w and existing.bottom() == rect.y) {
|
||||
existing.h += rect.h;
|
||||
merged = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (merged) continue;
|
||||
if (count < out.len) {
|
||||
out[count] = rect;
|
||||
count += 1;
|
||||
} else {
|
||||
out[count - 1] = out[count - 1].unite(rect);
|
||||
}
|
||||
}
|
||||
}
|
||||
return out[0..count];
|
||||
}
|
||||
};
|
||||
|
||||
/// A block of 32-bit pixels: `pixels` addressed row-major with `stride` pixels between
|
||||
/// row starts (≥ width — the framebuffer's stride is pitch/4, a layer's is its width).
|
||||
pub const Surface = struct {
|
||||
pixels: [*]u32,
|
||||
stride: u32, // pixels per row
|
||||
width: u32,
|
||||
height: u32,
|
||||
|
||||
pub fn bounds(s: Surface) Rect {
|
||||
return .{ .x = 0, .y = 0, .w = @intCast(s.width), .h = @intCast(s.height) };
|
||||
}
|
||||
|
||||
inline fn row(s: Surface, y: u32) [*]u32 {
|
||||
return s.pixels + @as(usize, y) * s.stride;
|
||||
}
|
||||
};
|
||||
|
||||
/// Fill `rect` of `s` with the native pixel `colour`, clipped to `s`'s bounds. Each row is
|
||||
/// one `@memset` over the clipped span, so the compiler vectorizes it and the bounds check
|
||||
/// runs once per row, not once per pixel.
|
||||
pub fn fillRect(s: Surface, rect: Rect, colour: u32) void {
|
||||
const c = rect.intersect(s.bounds());
|
||||
if (c.isEmpty()) return;
|
||||
const x0: usize = @intCast(c.x);
|
||||
const span: usize = @intCast(c.w);
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
@memset((s.row(@intCast(y)) + x0)[0..span], colour);
|
||||
}
|
||||
}
|
||||
|
||||
/// Composite the whole of `layer` onto `dst` with the layer's top-left at (`dx`, `dy`),
|
||||
/// painting only the pixels that fall inside `clip` (a `dst`-space rectangle) and inside
|
||||
/// `dst`. Opaque copy. This is the primitive `present` repeats over the visible layer
|
||||
/// stack, bottom to top, for each damaged region.
|
||||
pub fn composite(dst: Surface, dx: i32, dy: i32, layer: Surface, clip: Rect) void {
|
||||
const on_screen = Rect{ .x = dx, .y = dy, .w = @intCast(layer.width), .h = @intCast(layer.height) };
|
||||
const region = on_screen.intersect(clip).intersect(dst.bounds());
|
||||
if (region.isEmpty()) return;
|
||||
const span: usize = @intCast(region.w);
|
||||
const dst_x: usize = @intCast(region.x);
|
||||
const src_x: usize = @intCast(region.x - dx);
|
||||
var y: i32 = region.y;
|
||||
while (y < region.bottom()) : (y += 1) {
|
||||
const source_row = layer.row(@intCast(y - dy)) + src_x;
|
||||
const destination_row = dst.row(@intCast(y)) + dst_x;
|
||||
@memcpy(destination_row[0..span], source_row[0..span]);
|
||||
}
|
||||
}
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels from `src` (raw little-endian bytes, row-major,
|
||||
/// tightly packed) into `dst` at (`dx`, `dy`), clipped to `dst`'s bounds. `src` comes
|
||||
/// straight out of an IPC message buffer and carries no alignment guarantee, so each
|
||||
/// clipped row is a byte-wise `@memcpy` — which equals the old per-pixel little-endian
|
||||
/// `readInt` on every danos target (all little-endian) without the alignment concern.
|
||||
/// Returns without touching anything if `src` is short.
|
||||
pub fn blitTile(dst: Surface, dx: i32, dy: i32, src: []const u8, w: u32, h: u32) void {
|
||||
if (src.len < @as(usize, w) * h * 4) return;
|
||||
const region = Rect.init(dx, dy, @intCast(w), @intCast(h)).intersect(dst.bounds());
|
||||
if (region.isEmpty()) return;
|
||||
const span: usize = @intCast(region.w);
|
||||
const tile_x: usize = @intCast(region.x - dx);
|
||||
const dst_x: usize = @intCast(region.x);
|
||||
var y: i32 = region.y;
|
||||
while (y < region.bottom()) : (y += 1) {
|
||||
const tile_y: usize = @intCast(y - dy);
|
||||
const offset = (tile_y * w + tile_x) * 4;
|
||||
const destination_row = dst.row(@intCast(y)) + dst_x;
|
||||
@memcpy(std.mem.sliceAsBytes(destination_row[0..span]), src[offset..][0 .. span * 4]);
|
||||
}
|
||||
}
|
||||
|
||||
// --- tests ------------------------------------------------------------------
|
||||
|
||||
test "rect intersect: overlap and disjoint" {
|
||||
try std.testing.expectEqual(Rect.init(5, 5, 5, 5), Rect.init(0, 0, 10, 10).intersect(Rect.init(5, 5, 10, 10)));
|
||||
try std.testing.expect(Rect.init(0, 0, 10, 10).intersect(Rect.init(20, 20, 5, 5)).isEmpty());
|
||||
}
|
||||
|
||||
test "rect unite: bounding box, empty is identity" {
|
||||
const a = Rect.init(2, 2, 4, 4);
|
||||
try std.testing.expectEqual(Rect.init(2, 1, 10, 5), a.unite(Rect.init(10, 1, 2, 2)));
|
||||
try std.testing.expectEqual(a, a.unite(Rect.empty));
|
||||
try std.testing.expectEqual(a, Rect.empty.unite(a));
|
||||
}
|
||||
|
||||
test "fillRect clips to surface and honours stride padding" {
|
||||
// A 4×3 surface inside a 6-wide allocation (stride 6 > width 4), like pitch padding.
|
||||
var mem = [_]u32{0} ** (6 * 3);
|
||||
const s = Surface{ .pixels = &mem, .stride = 6, .width = 4, .height = 3 };
|
||||
fillRect(s, Rect.init(-1, -1, 3, 3), 0xAB); // straddles the top-left corner
|
||||
try std.testing.expectEqual(@as(u32, 0xAB), mem[0 * 6 + 0]);
|
||||
try std.testing.expectEqual(@as(u32, 0xAB), mem[1 * 6 + 1]);
|
||||
try std.testing.expectEqual(@as(u32, 0), mem[0 * 6 + 2]); // beyond the 2-wide fill
|
||||
try std.testing.expectEqual(@as(u32, 0), mem[2 * 6 + 0]); // row 2 untouched
|
||||
try std.testing.expectEqual(@as(u32, 0), mem[0 * 6 + 4]); // stride padding untouched
|
||||
}
|
||||
|
||||
test "composite: overlap shows the top layer, clipped to damage" {
|
||||
var back = [_]u32{0} ** (8 * 8);
|
||||
const dst = Surface{ .pixels = &back, .stride = 8, .width = 8, .height = 8 };
|
||||
var lo = [_]u32{0x11} ** (4 * 4);
|
||||
var hi = [_]u32{0x22} ** (4 * 4);
|
||||
const low = Surface{ .pixels = &lo, .stride = 4, .width = 4, .height = 4 };
|
||||
const high = Surface{ .pixels = &hi, .stride = 4, .width = 4, .height = 4 };
|
||||
composite(dst, 0, 0, low, dst.bounds()); // bottom at (0,0)
|
||||
composite(dst, 2, 2, high, dst.bounds()); // top overlaps at (2,2)
|
||||
try std.testing.expectEqual(@as(u32, 0x11), back[0 * 8 + 0]); // bottom-only
|
||||
try std.testing.expectEqual(@as(u32, 0x22), back[3 * 8 + 3]); // overlap → top wins
|
||||
try std.testing.expectEqual(@as(u32, 0x22), back[5 * 8 + 5]); // top-only
|
||||
try std.testing.expectEqual(@as(u32, 0), back[7 * 8 + 7]); // neither
|
||||
}
|
||||
|
||||
test "composite honours the damage rectangle" {
|
||||
var back = [_]u32{0} ** (8 * 8);
|
||||
const dst = Surface{ .pixels = &back, .stride = 8, .width = 8, .height = 8 };
|
||||
var fill = [_]u32{0x33} ** (8 * 8);
|
||||
const layer = Surface{ .pixels = &fill, .stride = 8, .width = 8, .height = 8 };
|
||||
composite(dst, 0, 0, layer, Rect.init(2, 2, 2, 2)); // only this damage region
|
||||
try std.testing.expectEqual(@as(u32, 0x33), back[2 * 8 + 2]);
|
||||
try std.testing.expectEqual(@as(u32, 0x33), back[3 * 8 + 3]);
|
||||
try std.testing.expectEqual(@as(u32, 0), back[1 * 8 + 1]); // outside damage
|
||||
try std.testing.expectEqual(@as(u32, 0), back[4 * 8 + 4]); // outside damage
|
||||
}
|
||||
|
||||
test "damage list keeps disjoint rectangles separate and merges overlap" {
|
||||
var list = DamageList{};
|
||||
list.add(Rect.init(0, 0, 10, 10));
|
||||
list.add(Rect.init(100, 100, 10, 10)); // far away: its own entry
|
||||
try std.testing.expectEqual(@as(usize, 2), list.slice().len);
|
||||
list.add(Rect.init(5, 5, 10, 10)); // overlaps the first: united into it
|
||||
try std.testing.expectEqual(@as(usize, 2), list.slice().len);
|
||||
try std.testing.expectEqual(Rect.init(0, 0, 15, 15), list.slice()[0]);
|
||||
try std.testing.expect(!list.isEmpty());
|
||||
list.clear();
|
||||
try std.testing.expect(list.isEmpty());
|
||||
}
|
||||
|
||||
test "damage list folds overflow into the last entry instead of dropping it" {
|
||||
var list = DamageList{};
|
||||
var i: i32 = 0;
|
||||
while (i < DamageList.capacity) : (i += 1) {
|
||||
list.add(Rect.init(i * 100, 0, 10, 10)); // disjoint: fills every slot
|
||||
}
|
||||
try std.testing.expectEqual(@as(usize, DamageList.capacity), list.slice().len);
|
||||
const overflow = Rect.init(0, 5000, 10, 10);
|
||||
list.add(overflow);
|
||||
try std.testing.expectEqual(@as(usize, DamageList.capacity), list.slice().len);
|
||||
const last = list.slice()[DamageList.capacity - 1];
|
||||
try std.testing.expect(!last.intersect(overflow).isEmpty()); // still covered
|
||||
}
|
||||
|
||||
test "damage list ignores empty rectangles" {
|
||||
var list = DamageList{};
|
||||
list.add(Rect.empty);
|
||||
try std.testing.expect(list.isEmpty());
|
||||
}
|
||||
|
||||
test "tile grid coalesces a run of adjacent tiles into one rectangle" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(256, 128); // 4×2 tiles of 64 px
|
||||
grid.add(Rect.init(10, 10, 100, 10)); // spans tiles (0,0) and (1,0)
|
||||
var scratch: [TileGrid.maximum_rects]Rect = undefined;
|
||||
const rects = grid.collect(&scratch);
|
||||
try std.testing.expectEqual(@as(usize, 1), rects.len);
|
||||
try std.testing.expectEqual(Rect.init(0, 0, 128, 64), rects[0]);
|
||||
}
|
||||
|
||||
test "tile grid: full-screen damage collapses back to a single rectangle" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(1280, 720); // 20×12 tiles; the bottom row is partial (720 = 11*64 + 16)
|
||||
grid.add(Rect.init(0, 0, 1280, 720));
|
||||
var scratch: [TileGrid.maximum_rects]Rect = undefined;
|
||||
const rects = grid.collect(&scratch);
|
||||
try std.testing.expectEqual(@as(usize, 1), rects.len);
|
||||
try std.testing.expectEqual(Rect.init(0, 0, 1280, 720), rects[0]);
|
||||
}
|
||||
|
||||
test "tile grid keeps far-apart damage as separate rectangles" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(1280, 720);
|
||||
grid.add(Rect.init(0, 0, 10, 10)); // top-left tile
|
||||
grid.add(Rect.init(1000, 600, 10, 10)); // a far-away tile
|
||||
var scratch: [TileGrid.maximum_rects]Rect = undefined;
|
||||
const rects = grid.collect(&scratch);
|
||||
try std.testing.expectEqual(@as(usize, 2), rects.len);
|
||||
}
|
||||
|
||||
test "tile grid clamps edge tiles to the true screen size" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(100, 100); // 2×2 tiles, both partial in each axis
|
||||
grid.add(Rect.init(0, 0, 100, 100));
|
||||
var scratch: [TileGrid.maximum_rects]Rect = undefined;
|
||||
const rects = grid.collect(&scratch);
|
||||
try std.testing.expectEqual(@as(usize, 1), rects.len);
|
||||
try std.testing.expectEqual(Rect.init(0, 0, 100, 100), rects[0]);
|
||||
}
|
||||
|
||||
test "tile grid clear empties it and reset resizes it" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(256, 256);
|
||||
grid.add(Rect.init(0, 0, 256, 256));
|
||||
try std.testing.expect(!grid.isEmpty());
|
||||
grid.clear();
|
||||
try std.testing.expect(grid.isEmpty());
|
||||
try std.testing.expect(grid.matches(256, 256));
|
||||
grid.reset(512, 512);
|
||||
try std.testing.expect(!grid.matches(256, 256));
|
||||
try std.testing.expect(grid.isEmpty());
|
||||
}
|
||||
|
||||
test "blitTile copies a packed tile, clipping and reading unaligned bytes" {
|
||||
var back = [_]u32{0} ** (4 * 4);
|
||||
const dst = Surface{ .pixels = &back, .stride = 4, .width = 4, .height = 4 };
|
||||
// A 2×2 tile in a byte buffer offset by one byte, so reads are unaligned.
|
||||
var raw = [_]u8{0} ** (1 + 2 * 2 * 4);
|
||||
const tile = raw[1..];
|
||||
for (0..4) |i| std.mem.writeInt(u32, tile[i * 4 ..][0..4], @intCast(0xA0 + i), .little);
|
||||
blitTile(dst, 3, 3, tile, 2, 2); // bottom-right corner; only (3,3) lands on-surface
|
||||
try std.testing.expectEqual(@as(u32, 0xA0), back[3 * 4 + 3]);
|
||||
try std.testing.expectEqual(@as(u32, 0), back[0]); // nothing else touched
|
||||
}
|
||||
@@ -0,0 +1,701 @@
|
||||
//! /system/services/display — the display service (docs/display.md, docs/display-v2.md).
|
||||
//! A ring-3 compositor: it composes an ordered stack of **layers** into a cacheable
|
||||
//! surface and presents finished frames. Scanout — how a frame reaches the panel — is a
|
||||
//! pluggable **backend** ([backend.zig](backend.zig)): the GOP framebuffer today, a native
|
||||
//! virtio-gpu driver later; this file never learns which is active. It owns the layer stack
|
||||
//! and damage tracking; the pixel math is the pure, host-tested
|
||||
//! [compositor.zig](compositor.zig).
|
||||
//!
|
||||
//! A layer is a server-owned surface (its own cacheable buffer) with a screen position,
|
||||
//! z-order, and visibility. Clients create layers, draw into them by command (`fill_rect`,
|
||||
//! `blit_tile`), mark `damage`, and ask for a `present`; the compositor repaints only the
|
||||
//! damaged region — clear it, paint the visible layers bottom-to-top into the backend's
|
||||
//! surface, then `backend.present(damage)`. Presents are paced by a ~60 Hz **frame clock**
|
||||
//! (see `schedulePresent`), so any number of client presents and cursor moves inside one
|
||||
//! interval coalesce into a single frame. Shared-memory client surfaces are later
|
||||
//! (docs/display-v2.md).
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const compositor = @import("compositor.zig");
|
||||
const backend_mod = @import("backend.zig");
|
||||
|
||||
const protocol = runtime.display_protocol;
|
||||
const ipc = runtime.ipc;
|
||||
const system = runtime.system;
|
||||
const input = runtime.input;
|
||||
const Thread = runtime.Thread;
|
||||
const Rect = compositor.Rect;
|
||||
const Surface = compositor.Surface;
|
||||
|
||||
/// The active scanout backend — the GOP framebuffer at boot, upgraded to a native driver
|
||||
/// (virtio-gpu) when one announces itself (V4).
|
||||
var backend: backend_mod.Backend = undefined;
|
||||
var frames: u64 = 0;
|
||||
|
||||
/// This service's endpoint, kept so `attach_scanout` can arm a one-shot timer: the very first
|
||||
/// native present must happen in a *later* loop iteration, after the reply to the driver's
|
||||
/// announce has unblocked it and it is serving its `.scanout` channel — presenting inline
|
||||
/// would deadlock (we'd call the driver while it waits on our reply).
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
|
||||
/// Set when the backend has just been upgraded to virtio-gpu: the next present repaints the
|
||||
/// whole screen into the shared surface and reads a pixel back to confirm the frame landed.
|
||||
var pending_native_verify: bool = false;
|
||||
|
||||
/// Set alongside it: after the native present is verified, run the mode-set self-check once
|
||||
/// (query the driver's modes, switch to a different one, confirm the geometry changed) — the
|
||||
/// serial proof the runtime-resolution-change + fenced-present paths work (V5).
|
||||
var pending_modeset_check: bool = false;
|
||||
|
||||
/// The wallpaper the compositor clears damaged regions to before painting layers.
|
||||
var background: u32 = 0;
|
||||
|
||||
/// The layer stack. A fixed table (a compositor has few top-level surfaces during
|
||||
/// bring-up); each used slot owns an mmap'd surface. `damage_list` accumulates the dirty
|
||||
/// screen rectangles since the last `present`, so a present touches only what changed —
|
||||
/// and keeps far-apart changes (the cursor here, an animating layer there) as *separate*
|
||||
/// small copies rather than one huge bounding box (see compositor.DamageList).
|
||||
const maximum_layers = 16;
|
||||
|
||||
const Layer = struct {
|
||||
used: bool = false,
|
||||
x: i32 = 0,
|
||||
y: i32 = 0,
|
||||
z: u32 = 0,
|
||||
visible: bool = false,
|
||||
surface: Surface = undefined,
|
||||
surface_len: usize = 0, // for munmap on destroy
|
||||
};
|
||||
|
||||
var layers: [maximum_layers]Layer = [_]Layer{.{}} ** maximum_layers;
|
||||
|
||||
/// Which damage tracker drives `present` — a compile-time A/B switch (both are in
|
||||
/// compositor.zig with the trade-off discussion):
|
||||
/// .list — free-form dirty rectangles (tight bounds, heuristic merging)
|
||||
/// .grid — a fixed 64-px tile grid (exact O(1) merging, tile-quantized repaints)
|
||||
const DamageMode = enum { list, grid };
|
||||
const damage_mode: DamageMode = .grid;
|
||||
|
||||
var damage_list: compositor.DamageList = .{};
|
||||
var damage_grid: compositor.TileGrid = .{};
|
||||
|
||||
/// The **frame clock**: client `present` requests and cursor motion don't repaint
|
||||
/// immediately — they accumulate damage and arm a one-shot timer, and the tick composites
|
||||
/// everything pending as one frame. That paces presents to ~60 Hz no matter how fast
|
||||
/// clients draw or the mouse moves (previously every mouse event became a full present).
|
||||
/// No backend has a real vblank to pace by (docs/display-v2.md, "Fenced is not vsync");
|
||||
/// this is the software stand-in, the same strategy Linux uses atop virtio-gpu. Bring-up
|
||||
/// paths that need pixels on screen *now* (initialise, the self-checks) still call
|
||||
/// `present()` directly.
|
||||
///
|
||||
/// The interval comes from the *active backend's* panel refresh rate (EDID: the loader
|
||||
/// captures it for the GOP floor while firmware still runs; the native driver reads its
|
||||
/// own and carries it in the announce). `updateFrameClock` re-derives it whenever the
|
||||
/// backend changes — the boot framebuffer's clock dies with the GOP floor at upgrade.
|
||||
/// Without a rate the clock defaults to 60 Hz, and it is clamped to [30, 120] Hz so a
|
||||
/// mis-parsed EDID can neither starve nor flood the compositor.
|
||||
var frame_interval_milliseconds: u64 = 16;
|
||||
var frame_timer_armed = false;
|
||||
|
||||
/// Derive the frame-clock interval from the active backend's refresh rate and log what
|
||||
/// the clock is now pacing to. Called at bring-up and again on every backend change.
|
||||
fn updateFrameClock() void {
|
||||
const reported = backend.info().refresh_hz;
|
||||
const rate: u64 = if (reported == 0) 60 else @min(@max(reported, 30), 120);
|
||||
frame_interval_milliseconds = @max(1000 / rate, 1);
|
||||
var line: [96]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "display: frame clock {d} Hz ({s})\n", .{
|
||||
1000 / frame_interval_milliseconds,
|
||||
if (reported == 0) "default" else "panel EDID",
|
||||
}) catch return);
|
||||
}
|
||||
|
||||
/// Arm the frame clock unless a tick is already pending: any number of requests inside
|
||||
/// one interval coalesce into that single tick's present.
|
||||
fn schedulePresent() void {
|
||||
if (frame_timer_armed) return;
|
||||
frame_timer_armed = true;
|
||||
_ = system.timerOnce(service_endpoint, frame_interval_milliseconds);
|
||||
}
|
||||
|
||||
/// A timer landing — the frame clock, or the deferred first native present armed by
|
||||
/// `attach_scanout`: present the accumulated damage, then run the one-shot mode-set
|
||||
/// self-check if the native upgrade queued it.
|
||||
fn frameTick() void {
|
||||
frame_timer_armed = false;
|
||||
present();
|
||||
if (pending_modeset_check) {
|
||||
pending_modeset_check = false;
|
||||
modesetSelfCheck();
|
||||
}
|
||||
}
|
||||
|
||||
// --- geometry helpers -------------------------------------------------------
|
||||
|
||||
fn screenRect() Rect {
|
||||
const m = backend.info();
|
||||
return .{ .x = 0, .y = 0, .w = @intCast(m.width), .h = @intCast(m.height) };
|
||||
}
|
||||
|
||||
fn layerScreenRect(l: *const Layer) Rect {
|
||||
return .{ .x = l.x, .y = l.y, .w = @intCast(l.surface.width), .h = @intCast(l.surface.height) };
|
||||
}
|
||||
|
||||
/// Add `r` (screen coordinates) to the pending damage, clipped to the screen. In grid
|
||||
/// mode the grid re-sizes itself lazily when the screen geometry changes — every
|
||||
/// geometry-changing path (`attach_scanout`, `set_mode`) damages the whole new screen
|
||||
/// right after, so damage pending from the old geometry is safely superseded.
|
||||
fn addDamage(r: Rect) void {
|
||||
const clipped = r.intersect(screenRect());
|
||||
switch (damage_mode) {
|
||||
.list => damage_list.add(clipped),
|
||||
.grid => {
|
||||
const mode = backend.info();
|
||||
if (!damage_grid.matches(mode.width, mode.height)) damage_grid.reset(mode.width, mode.height);
|
||||
damage_grid.add(clipped);
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// --- layer operations (called from onMessage and the self-check) ------------
|
||||
|
||||
fn freeLayer() ?u32 {
|
||||
for (&layers, 0..) |*l, i| {
|
||||
if (!l.used) return @intCast(i);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// A used layer by id, or null if the id is out of range or free.
|
||||
fn layerAt(id: u32) ?*Layer {
|
||||
if (id >= maximum_layers or !layers[id].used) return null;
|
||||
return &layers[id];
|
||||
}
|
||||
|
||||
fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32, visible: bool) ?u32 {
|
||||
if (w == 0 or h == 0) return null;
|
||||
const slot = freeLayer() orelse return null;
|
||||
const len = @as(usize, w) * h * 4;
|
||||
const base = system.mmap(len, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(base)) return null;
|
||||
layers[slot] = .{
|
||||
.used = true,
|
||||
.x = x,
|
||||
.y = y,
|
||||
.z = z,
|
||||
.visible = visible,
|
||||
.surface = .{ .pixels = @ptrFromInt(base), .stride = w, .width = w, .height = h },
|
||||
.surface_len = len,
|
||||
};
|
||||
return slot;
|
||||
}
|
||||
|
||||
fn fillLayer(id: u32, local: Rect, colour: u32) bool {
|
||||
const l = layerAt(id) orelse return false;
|
||||
compositor.fillRect(l.surface, local, colour);
|
||||
// Damage in screen space = the fill, translated by the layer origin, within the layer.
|
||||
const screen = Rect{ .x = l.x + local.x, .y = l.y + local.y, .w = local.w, .h = local.h };
|
||||
addDamage(screen.intersect(layerScreenRect(l)));
|
||||
return true;
|
||||
}
|
||||
|
||||
fn blitLayer(id: u32, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
const l = layerAt(id) orelse return false;
|
||||
compositor.blitTile(l.surface, x, y, pixels, w, h);
|
||||
const screen = Rect{ .x = l.x + x, .y = l.y + y, .w = @intCast(w), .h = @intCast(h) };
|
||||
addDamage(screen.intersect(layerScreenRect(l)));
|
||||
return true;
|
||||
}
|
||||
|
||||
fn configureLayer(id: u32, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
const l = layerAt(id) orelse return false;
|
||||
addDamage(layerScreenRect(l)); // the old footprint must repaint
|
||||
l.x = x;
|
||||
l.y = y;
|
||||
l.z = z;
|
||||
l.visible = visible;
|
||||
addDamage(layerScreenRect(l)); // and the new one
|
||||
return true;
|
||||
}
|
||||
|
||||
fn destroyLayer(id: u32) bool {
|
||||
const l = layerAt(id) orelse return false;
|
||||
addDamage(layerScreenRect(l));
|
||||
_ = system.munmap(@intFromPtr(l.surface.pixels), l.surface_len);
|
||||
l.* = .{};
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- compositing + present --------------------------------------------------
|
||||
|
||||
/// Repaint the damaged region `clip` of the backend's compose surface: clear it to the
|
||||
/// background, then paint every visible layer that overlaps it, bottom to top (ascending z).
|
||||
fn compositeInto(clip: Rect) void {
|
||||
const target = backend.surface();
|
||||
compositor.fillRect(target, clip, background);
|
||||
|
||||
// z-order the used, visible layers (n ≤ 16; a plain insertion sort of indices).
|
||||
var order: [maximum_layers]u32 = undefined;
|
||||
var n: usize = 0;
|
||||
for (layers, 0..) |l, i| {
|
||||
if (l.used and l.visible) {
|
||||
order[n] = @intCast(i);
|
||||
n += 1;
|
||||
}
|
||||
}
|
||||
var a: usize = 1;
|
||||
while (a < n) : (a += 1) {
|
||||
const key = order[a];
|
||||
var b: usize = a;
|
||||
while (b > 0 and layers[order[b - 1]].z > layers[key].z) : (b -= 1) order[b] = order[b - 1];
|
||||
order[b] = key;
|
||||
}
|
||||
|
||||
for (order[0..n]) |i| {
|
||||
const l = layers[i];
|
||||
compositor.composite(target, l.x, l.y, l.surface, clip);
|
||||
}
|
||||
}
|
||||
|
||||
/// Composite each accumulated damage rectangle into the backend's surface, hand the list
|
||||
/// to the backend to put on screen, then clear the damage. A no-op when nothing is dirty.
|
||||
/// The frame counter advances regardless, so callers can name frames.
|
||||
fn present() void {
|
||||
var scratch: [compositor.TileGrid.maximum_rects]Rect = undefined;
|
||||
const dirty: []const Rect = switch (damage_mode) {
|
||||
.list => damage_list.slice(),
|
||||
.grid => damage_grid.collect(&scratch),
|
||||
};
|
||||
const had_damage = dirty.len != 0;
|
||||
if (had_damage) {
|
||||
for (dirty) |region| compositeInto(region);
|
||||
backend.present(dirty);
|
||||
}
|
||||
switch (damage_mode) {
|
||||
.list => damage_list.clear(),
|
||||
.grid => damage_grid.clear(),
|
||||
}
|
||||
frames += 1;
|
||||
|
||||
// The first present after a native upgrade confirms the composited frame actually reached
|
||||
// the shared scanout surface (the automated stand-in for "it's on screen").
|
||||
if (pending_native_verify and had_damage) {
|
||||
pending_native_verify = false;
|
||||
verifyNativePresent();
|
||||
}
|
||||
}
|
||||
|
||||
/// Read a pixel straight back from the shared scanout surface after a native present. The
|
||||
/// surface starts zeroed, so a non-zero centre pixel means the compositor wrote the frame into
|
||||
/// the pages the driver transfers-and-flushes from — that, plus the driver acking the present
|
||||
/// over `.scanout`, is the serial proof the native path works.
|
||||
fn verifyNativePresent() void {
|
||||
const s = backend.surface();
|
||||
const sample = s.pixels[@as(usize, s.height / 2) * s.stride + s.width / 2];
|
||||
if (sample != 0) {
|
||||
_ = system.write("display: native present verified\n");
|
||||
} else {
|
||||
_ = system.write("display: native present FAILED (blank surface)\n");
|
||||
}
|
||||
}
|
||||
|
||||
/// A native scanout driver announced itself: map the shared surface it handed over, find its
|
||||
/// present channel, switch the backend to virtio-gpu, and queue a full-screen repaint. The
|
||||
/// present is deferred to a timer (see `service_endpoint`) so it happens after this reply
|
||||
/// unblocks the driver and it starts serving `.scanout`.
|
||||
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, refresh_hz: u32, capability: ?ipc.Handle, reply: []u8) usize {
|
||||
const cap = capability orelse return fail(reply);
|
||||
if (width == 0 or height == 0 or stride < width) return fail(reply);
|
||||
const mapped = runtime.shared_memory.map(cap) orelse return fail(reply);
|
||||
const scanout = ipc.lookup(.scanout) orelse return fail(reply);
|
||||
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
|
||||
// scanout. (The previous shared mapping leaks — there is no shared_memory_unmap syscall yet — but the
|
||||
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
|
||||
const reattach = switch (backend) {
|
||||
.virtio => true,
|
||||
else => false,
|
||||
};
|
||||
|
||||
backend = .{ .virtio = .{
|
||||
.pixels = @ptrCast(@alignCast(mapped)),
|
||||
.stride = stride,
|
||||
.width = width,
|
||||
.height = height,
|
||||
.format = format,
|
||||
.refresh_hz = refresh_hz,
|
||||
.scanout = scanout,
|
||||
} };
|
||||
background = protocol.pack(format, 0x20, 0x30, 0x48); // re-pack the wallpaper for the mode
|
||||
updateFrameClock(); // the GOP floor's clock dies here — pace by the GPU's EDID now
|
||||
addDamage(screenRect()); // the whole new surface must be painted
|
||||
pending_native_verify = true;
|
||||
if (!reattach) pending_modeset_check = true; // the mode-set self-check runs once, on first upgrade
|
||||
_ = system.timerOnce(service_endpoint, 50); // present once the driver is serving .scanout
|
||||
_ = system.write(if (reattach)
|
||||
"display: scanout re-attached\n"
|
||||
else
|
||||
"display: scanout upgraded to virtio-gpu\n");
|
||||
return ok(reply);
|
||||
}
|
||||
|
||||
/// After the native upgrade is verified, prove the runtime-resolution-change and fenced-present
|
||||
/// paths: query the driver's modes, switch to one that differs from the current, re-composite
|
||||
/// the whole screen at the new size, and confirm the backend now reports that geometry. The
|
||||
/// present goes through the driver's fenced flush, so a clean present is a *fenced* present —
|
||||
/// completion-acknowledged and tear-free, not vblank-paced (docs/display-v2.md).
|
||||
fn modesetSelfCheck() void {
|
||||
if (!backend.canModeSet()) return;
|
||||
var mode_list: [4]backend_mod.Mode = undefined;
|
||||
const count = backend.modes(&mode_list);
|
||||
if (count == 0) {
|
||||
_ = system.write("display: mode-set self-check: no modes reported\n");
|
||||
return;
|
||||
}
|
||||
const current = backend.info();
|
||||
var target: ?backend_mod.Mode = null;
|
||||
for (mode_list[0..count]) |m| {
|
||||
if (m.width != current.width or m.height != current.height) {
|
||||
target = m;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const wanted = target orelse {
|
||||
_ = system.write("display: mode-set self-check: no alternate mode offered\n");
|
||||
return;
|
||||
};
|
||||
if (!backend.setMode(wanted.width, wanted.height)) {
|
||||
_ = system.write("display: mode set FAILED\n");
|
||||
return;
|
||||
}
|
||||
addDamage(screenRect()); // repaint the whole screen at the new resolution, then present it
|
||||
present();
|
||||
|
||||
const now = backend.info();
|
||||
if (now.width == wanted.width and now.height == wanted.height) {
|
||||
var line: [80]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "display: mode set to {d}x{d}, verified\n", .{ now.width, now.height }) catch "display: mode set, verified\n");
|
||||
if (backend.hasFencedPresent()) _ = system.write("display: fenced present ok\n");
|
||||
} else {
|
||||
_ = system.write("display: mode set FAILED (geometry unchanged)\n");
|
||||
}
|
||||
}
|
||||
|
||||
// --- startup self-check -----------------------------------------------------
|
||||
|
||||
/// Prove the compositor wiring on the real backend: two overlapping opaque layers,
|
||||
/// composited, must show the top layer in the overlap and the bottom layer outside it.
|
||||
/// Exercises the whole path — mmap surfaces, the z-sort, damage, composite into the
|
||||
/// backend surface — and reads the composited result back. Cleans up after itself.
|
||||
fn selfCheck() void {
|
||||
const format = backend.info().format;
|
||||
const red = protocol.pack(format, 0xC0, 0x20, 0x20);
|
||||
const green = protocol.pack(format, 0x20, 0xC0, 0x20);
|
||||
const bottom = createLayer(100, 100, 80, 80, 0, true) orelse return fail_check("create");
|
||||
const top = createLayer(140, 140, 80, 80, 1, true) orelse return fail_check("create");
|
||||
_ = fillLayer(bottom, Rect.init(0, 0, 80, 80), red);
|
||||
_ = fillLayer(top, Rect.init(0, 0, 80, 80), green);
|
||||
present();
|
||||
|
||||
const surface = backend.surface();
|
||||
const overlap = surface.pixels[@as(usize, 150) * surface.stride + 150]; // in both → top
|
||||
const bottom_only = surface.pixels[@as(usize, 110) * surface.stride + 110]; // bottom only
|
||||
|
||||
_ = destroyLayer(top);
|
||||
_ = destroyLayer(bottom);
|
||||
present(); // repaint the self-check region back to the background
|
||||
|
||||
if (overlap == green and bottom_only == red) {
|
||||
_ = system.write("display: compositor self-check ok\n");
|
||||
} else {
|
||||
_ = system.write("display: compositor self-check FAILED\n");
|
||||
}
|
||||
}
|
||||
|
||||
fn fail_check(_: []const u8) void {
|
||||
_ = system.write("display: compositor self-check FAILED (setup)\n");
|
||||
}
|
||||
|
||||
// --- cursor + mouse-input thread --------------------------------------------
|
||||
//
|
||||
// The compositor is the single owner of the framebuffer: only the main service
|
||||
// loop touches `backend` and the layer stack. A dedicated listener thread (spawned
|
||||
// in `initialise`) blocks on the input service's mouse stream, accumulates relative
|
||||
// motion into an absolute cursor position, and hands that position to the main loop
|
||||
// through `cursor_channel` — a single-slot latest-value cell (the renderer wants
|
||||
// where the cursor *is*, not a replay of every delta). The listener never touches
|
||||
// the compositor; it only writes the channel and pokes the main loop awake with a
|
||||
// self-directed `ipc.send`, which arrives as a message-notification in the service
|
||||
// loop (docs/threading.md, docs/display.md). Shared fate: a fault in the listener
|
||||
// takes the whole display down and the supervisor restarts it (docs/resilience.md).
|
||||
|
||||
const cursor_size = 10; // a small square sprite — enough to prove tracking
|
||||
const cursor_z = 0xFFFF_FFFF; // always above client layers
|
||||
const cursor_report_threshold = 5; // px of travel before the tracking marker latches
|
||||
|
||||
var cursor_layer: ?u32 = null;
|
||||
var cursor_origin_x: i32 = 0;
|
||||
var cursor_origin_y: i32 = 0;
|
||||
/// Latched once the cursor has demonstrably tracked a run of motion end to end
|
||||
/// (source -> input service -> listener -> channel -> render): the `display-cursor`
|
||||
/// test's success marker.
|
||||
var cursor_tracking_reported: bool = false;
|
||||
|
||||
const poke_byte = [_]u8{0}; // the poke carries no payload; the value lives in the channel
|
||||
|
||||
/// Shared between the listener thread (producer) and the main loop (consumer).
|
||||
/// Latest-value semantics with a coalesced wake: at most one poke is queued while
|
||||
/// the main loop has not drained the last one, so a fast mouse cannot flood the
|
||||
/// service endpoint.
|
||||
const CursorChannel = struct {
|
||||
lock: Thread.Mutex = .{},
|
||||
poke_endpoint: ipc.Handle = 0,
|
||||
x: i32 = 0,
|
||||
y: i32 = 0,
|
||||
buttons: u32 = 0,
|
||||
dirty: bool = false,
|
||||
poke_pending: bool = false,
|
||||
|
||||
const Snapshot = struct { x: i32, y: i32, buttons: u32 };
|
||||
|
||||
/// Producer (listener thread): record the newest position and, unless a wake is
|
||||
/// already queued, poke the main loop awake.
|
||||
fn publish(self: *CursorChannel, x: i32, y: i32, buttons: u32) void {
|
||||
self.lock.lock();
|
||||
self.x = x;
|
||||
self.y = y;
|
||||
self.buttons = buttons;
|
||||
self.dirty = true;
|
||||
const need_poke = !self.poke_pending;
|
||||
if (need_poke) self.poke_pending = true;
|
||||
self.lock.unlock();
|
||||
if (need_poke) _ = ipc.send(self.poke_endpoint, &poke_byte);
|
||||
}
|
||||
|
||||
/// Consumer (main loop): take the latest position, or null if nothing changed
|
||||
/// since the last take. Clears the wake latch so the next publish pokes again.
|
||||
fn take(self: *CursorChannel) ?Snapshot {
|
||||
self.lock.lock();
|
||||
defer self.lock.unlock();
|
||||
self.poke_pending = false;
|
||||
if (!self.dirty) return null;
|
||||
self.dirty = false;
|
||||
return .{ .x = self.x, .y = self.y, .buttons = self.buttons };
|
||||
}
|
||||
};
|
||||
|
||||
var cursor_channel: CursorChannel = .{};
|
||||
|
||||
fn clampAxis(value: i32, max: i32) i32 {
|
||||
if (value < 0) return 0;
|
||||
if (value > max) return max;
|
||||
return value;
|
||||
}
|
||||
|
||||
/// The mouse-listener thread. Blocks on the input service's mouse stream, accumulates
|
||||
/// relative motion into an absolute position clamped to the screen, and publishes each
|
||||
/// update. Runs for the life of the process; a parked `next()` leaves the core free to
|
||||
/// halt (docs/halting.md). It reads only its own state and the channel — never the
|
||||
/// compositor — so no lock guards the framebuffer.
|
||||
fn mouseListener(width: u32, height: u32) void {
|
||||
var mouse = input.subscribeMouse() orelse {
|
||||
_ = system.write("display: mouse subscribe failed\n");
|
||||
return;
|
||||
};
|
||||
// Our own handle to the compositor's endpoint. IPC handles are per-thread, so we
|
||||
// cannot reuse the main thread's service handle — we look the service up to install a
|
||||
// handle in this thread's table. A poke posted here wakes the compositor loop parked
|
||||
// in replyWait (docs/threading.md: handles do not cross threads).
|
||||
cursor_channel.poke_endpoint = ipc.lookup(.display) orelse {
|
||||
_ = system.write("display: mouse listener could not reach the compositor endpoint\n");
|
||||
return;
|
||||
};
|
||||
const max_x: i32 = @as(i32, @intCast(width)) - 1;
|
||||
const max_y: i32 = @as(i32, @intCast(height)) - 1;
|
||||
var x: i32 = @divTrunc(max_x, 2);
|
||||
var y: i32 = @divTrunc(max_y, 2);
|
||||
var buttons: u32 = 0;
|
||||
while (true) {
|
||||
const event = mouse.next() orelse continue;
|
||||
// Switch on the raw kind (not @enumFromInt, which would panic on a scroll or
|
||||
// future kind): motion moves the cursor, anything else just updates buttons.
|
||||
if (event.kind == @intFromEnum(input.MouseEventKind.motion)) {
|
||||
x = clampAxis(x + event.dx, max_x);
|
||||
y = clampAxis(y + event.dy, max_y);
|
||||
} else {
|
||||
buttons = event.buttons;
|
||||
}
|
||||
cursor_channel.publish(x, y, buttons);
|
||||
}
|
||||
}
|
||||
|
||||
/// Consume the latest cursor position from the channel and move the cursor layer to it.
|
||||
/// Runs on the main loop (the compositor owner) in response to a listener poke.
|
||||
/// `configureLayer` damages both the old and new footprints; the frame clock presents
|
||||
/// them at the next tick, so a fast mouse coalesces to at most ~60 repaints a second.
|
||||
fn renderCursor() void {
|
||||
const snapshot = cursor_channel.take() orelse return;
|
||||
const id = cursor_layer orelse return;
|
||||
_ = configureLayer(id, snapshot.x, snapshot.y, cursor_z, true);
|
||||
schedulePresent();
|
||||
if (!cursor_tracking_reported and
|
||||
@abs(snapshot.x - cursor_origin_x) >= cursor_report_threshold and
|
||||
@abs(snapshot.y - cursor_origin_y) >= cursor_report_threshold)
|
||||
{
|
||||
cursor_tracking_reported = true;
|
||||
_ = system.write("display: cursor tracking mouse ok\n");
|
||||
}
|
||||
}
|
||||
|
||||
/// Create the cursor sprite (a top-z square) at screen centre and spawn the listener
|
||||
/// thread. Called from `initialise` once the backend is up. If either step fails the
|
||||
/// display still serves drawing clients — it just has no cursor.
|
||||
fn startCursorTracking() void {
|
||||
const mode = backend.info();
|
||||
cursor_origin_x = @divTrunc(@as(i32, @intCast(mode.width)), 2);
|
||||
cursor_origin_y = @divTrunc(@as(i32, @intCast(mode.height)), 2);
|
||||
const id = createLayer(cursor_origin_x, cursor_origin_y, cursor_size, cursor_size, cursor_z, true) orelse {
|
||||
_ = system.write("display: could not create cursor layer\n");
|
||||
return;
|
||||
};
|
||||
cursor_layer = id;
|
||||
_ = fillLayer(id, Rect.init(0, 0, cursor_size, cursor_size), protocol.pack(mode.format, 0xF0, 0xF0, 0xF0));
|
||||
present(); // show the cursor at its start position
|
||||
|
||||
_ = Thread.spawn(.{}, mouseListener, .{ mode.width, mode.height }) catch {
|
||||
_ = system.write("display: could not spawn mouse listener\n");
|
||||
};
|
||||
}
|
||||
|
||||
// --- service ----------------------------------------------------------------
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
|
||||
// Pick the scanout backend (GOP today). It logs the reason on failure.
|
||||
backend = backend_mod.select() orelse return false;
|
||||
const mode = backend.info();
|
||||
background = protocol.pack(mode.format, 0x20, 0x30, 0x48); // a dark slate wallpaper
|
||||
|
||||
// Clear the whole screen through the compose surface → present path (double buffering:
|
||||
// no direct-to-scanout drawing).
|
||||
addDamage(screenRect());
|
||||
present();
|
||||
|
||||
var line: [96]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "display: online {d}x{d} pitch {d} format {d}\n", .{
|
||||
mode.width, mode.height, mode.pitch, mode.format,
|
||||
}) catch "display: online\n");
|
||||
updateFrameClock();
|
||||
_ = system.write("display: presented frame 0\n");
|
||||
|
||||
selfCheck();
|
||||
|
||||
// Bring up the cursor and the mouse-listener thread now that the backend is live.
|
||||
startCursorTracking();
|
||||
return true;
|
||||
}
|
||||
|
||||
fn writeReply(reply: []u8, value: protocol.Reply) usize {
|
||||
const bytes = std.mem.asBytes(&value);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
return bytes.len;
|
||||
}
|
||||
|
||||
fn ok(reply: []u8) usize {
|
||||
return writeReply(reply, .{ .status = 0 });
|
||||
}
|
||||
|
||||
fn fail(reply: []u8) usize {
|
||||
return writeReply(reply, .{ .status = -1 });
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
if (message.len < protocol.request_size) return fail(reply);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
// Switch on the raw operation value — an out-of-range one must fail cleanly, not
|
||||
// panic an `@enumFromInt`.
|
||||
switch (request.operation) {
|
||||
@intFromEnum(protocol.Operation.info) => {
|
||||
const m = backend.info();
|
||||
return writeReply(reply, .{ .status = 0, .width = m.width, .height = m.height, .pitch = m.pitch, .format = m.format });
|
||||
},
|
||||
@intFromEnum(protocol.Operation.create_layer) => {
|
||||
// x/y are signed coordinates carried in the u32 wire fields — reinterpret the
|
||||
// bits (@bitCast), don't range-check (@intCast) which a negative would fail.
|
||||
const slot = createLayer(@bitCast(request.x), @bitCast(request.y), request.width, request.height, request.z, request.visible != 0) orelse return fail(reply);
|
||||
return writeReply(reply, .{ .status = 0, .layer = slot });
|
||||
},
|
||||
@intFromEnum(protocol.Operation.configure_layer) => {
|
||||
return if (configureLayer(request.layer, @bitCast(request.x), @bitCast(request.y), request.z, request.visible != 0)) ok(reply) else fail(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.destroy_layer) => {
|
||||
return if (destroyLayer(request.layer)) ok(reply) else fail(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.fill_rect) => {
|
||||
const local = Rect.init(@bitCast(request.x), @bitCast(request.y), @intCast(request.width), @intCast(request.height));
|
||||
return if (fillLayer(request.layer, local, request.colour)) ok(reply) else fail(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.blit_tile) => {
|
||||
return if (blitLayer(request.layer, @bitCast(request.x), @bitCast(request.y), request.width, request.height, payload)) ok(reply) else fail(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.damage) => {
|
||||
const l = layerAt(request.layer) orelse return fail(reply);
|
||||
const screen = Rect{ .x = l.x + @as(i32, @bitCast(request.x)), .y = l.y + @as(i32, @bitCast(request.y)), .w = @intCast(request.width), .h = @intCast(request.height) };
|
||||
addDamage(screen.intersect(layerScreenRect(l)));
|
||||
return ok(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.present) => {
|
||||
// Scheduled, not immediate: the frame clock composites the accumulated damage
|
||||
// at the next tick, so back-to-back client presents coalesce into one frame.
|
||||
schedulePresent();
|
||||
return ok(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.attach_scanout) => {
|
||||
return attachScanout(request.x, request.width, request.height, request.colour, request.y, capability, reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.set_mode) => {
|
||||
if (!backend.setMode(request.width, request.height)) return fail(reply);
|
||||
addDamage(screenRect()); // repaint the whole screen at the new resolution
|
||||
present();
|
||||
return ok(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.get_modes) => {
|
||||
var list: [4]backend_mod.Mode = undefined;
|
||||
const count = backend.modes(&list);
|
||||
var response = protocol.ModesReply{ .status = 0, .count = @intCast(count), .modes = undefined };
|
||||
for (0..protocol.max_modes) |i| {
|
||||
response.modes[i] = if (i < count)
|
||||
.{ .width = list[i].width, .height = list[i].height }
|
||||
else
|
||||
.{ .width = 0, .height = 0 };
|
||||
}
|
||||
const bytes = std.mem.asBytes(&response);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
return bytes.len;
|
||||
},
|
||||
else => return fail(reply),
|
||||
}
|
||||
}
|
||||
|
||||
/// Two notification sources reach the compositor, and one coalesced badge can carry
|
||||
/// both, so each bit is handled independently. A **message-notification** is a poke from
|
||||
/// the mouse-listener thread (a buffered self-`ipc.send`, `notify_message_bit`): fold the
|
||||
/// newest cursor position into the scene. A **timer** (`notify_timer_bit`) is the frame
|
||||
/// clock — or the deferred first native present after `attach_scanout` — either way,
|
||||
/// present the accumulated damage.
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & ipc.notify_message_bit != 0) renderCursor();
|
||||
if (badge & ipc.notify_timer_bit != 0) frameTick();
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.service = .display,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,115 @@
|
||||
//! The display wire protocol — what a client says to the display service over its
|
||||
//! well-known `.display` endpoint. extern-struct messages with an `Operation` tag, the
|
||||
//! same shape as block/vfs/input protocols. The compositor owns the framebuffer and an
|
||||
//! ordered stack of **layers**; a client creates layers, draws into them with these
|
||||
//! operations, marks damage, and asks for a `present`. v1 surfaces are server-owned (a
|
||||
//! client draws by command); shared-memory surfaces are a later milestone (docs/display.md).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
/// info() -> { width, height, pitch, format }: the display's current mode.
|
||||
info = 0,
|
||||
/// create_layer(x, y, width, height, z) -> { layer }: a new server-owned surface.
|
||||
create_layer = 1,
|
||||
/// configure_layer(layer, x, y, z, visible): move, restack, show, or hide a layer.
|
||||
configure_layer = 2,
|
||||
/// destroy_layer(layer): release a layer.
|
||||
destroy_layer = 3,
|
||||
/// fill_rect(layer, x, y, width, height, colour): fill a rectangle of a layer.
|
||||
fill_rect = 4,
|
||||
/// blit_tile(layer, x, y, width, height, <inline pixels>): copy a small pixel tile in.
|
||||
blit_tile = 5,
|
||||
/// damage(layer, x, y, width, height): mark a region dirty for the next present.
|
||||
damage = 6,
|
||||
/// present(): composite the dirty layers and flush to the screen.
|
||||
present = 7,
|
||||
/// attach_scanout(x=stride, y=refresh_hz, width, height, colour=format) + <surface
|
||||
/// capability>: a native scanout driver announces itself, handing over the shared scanout
|
||||
/// surface as an `ipc_call` send_cap. The compositor maps it, looks up the driver's
|
||||
/// `.scanout` present channel, and upgrades off the GOP floor (docs/display-v2.md V4).
|
||||
/// `x` is the surface's row stride in pixels, `y` the panel refresh rate from the
|
||||
/// driver's EDID read (0 = unknown; paces the compositor's frame clock), `colour` the
|
||||
/// DisplayFormat.
|
||||
attach_scanout = 8,
|
||||
/// set_mode(width, height): change the display resolution — only a native backend that
|
||||
/// reports `canModeSet` honours it; on the GOP floor it fails (docs/display-v2.md V5).
|
||||
set_mode = 9,
|
||||
/// get_modes() -> ModesReply: the resolutions the display can switch to (empty on GOP).
|
||||
get_modes = 10,
|
||||
};
|
||||
|
||||
/// The fixed request header. A `blit_tile`'s pixel payload (width*height 32-bit pixels)
|
||||
/// follows this header inline in the same message, up to `maximum_payload`.
|
||||
pub const Request = extern struct {
|
||||
operation: u32,
|
||||
layer: u32 = 0, // create/configure/destroy/fill/blit/damage: the target layer
|
||||
x: u32 = 0,
|
||||
y: u32 = 0,
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
z: u32 = 0, // create_layer / configure_layer: stacking order (higher = in front)
|
||||
colour: u32 = 0, // fill_rect: the fill colour (native pixel value)
|
||||
visible: u32 = 1, // configure_layer: 0 hides the layer
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Reply = extern struct {
|
||||
status: i32, // 0 on success, negative on failure
|
||||
reserved: u32 = 0,
|
||||
// info():
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
pitch: u32 = 0,
|
||||
format: u32 = 0, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||
// create_layer():
|
||||
layer: u32 = 0,
|
||||
reserved2: u32 = 0,
|
||||
};
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = extern struct { width: u32, height: u32 };
|
||||
pub const max_modes = 4;
|
||||
|
||||
/// The reply to `get_modes`: a small fixed list of resolutions the display can switch to.
|
||||
pub const ModesReply = extern struct {
|
||||
status: i32,
|
||||
count: u32,
|
||||
modes: [max_modes]Mode,
|
||||
};
|
||||
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||
|
||||
/// The IPC message size — the kernel caps every message at `MESSAGE_MAXIMUM` (256 bytes,
|
||||
/// system/kernel/ipc-synchronous.zig), so this matches it (a larger receive/reply buffer
|
||||
/// is rejected with -E2BIG). A `blit_tile` therefore carries only a *small* tile inline —
|
||||
/// `maximum_payload` bytes = up to 54 pixels, enough for a cursor or small sprite; larger
|
||||
/// bitmaps are the deferred shared-memory surface path (docs/display.md).
|
||||
pub const message_maximum: usize = 256;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
pub const maximum_payload: usize = message_maximum - request_size;
|
||||
|
||||
/// Pack an 8-bit-per-channel colour into the display's native 32-bit pixel for `format`
|
||||
/// (a device-abi `DisplayFormat`: 0 = rgbx, 1 = bgrx). Shared so a `colour` in a
|
||||
/// `fill_rect` request means the same thing to the client that sends it and the
|
||||
/// compositor that paints it. Little-endian memory, reserved byte 0: rgbx puts red in
|
||||
/// the low byte, bgrx puts blue there.
|
||||
pub fn pack(format: u32, r: u8, g: u8, b: u8) u32 {
|
||||
const rr: u32 = r;
|
||||
const gg: u32 = g;
|
||||
const bb: u32 = b;
|
||||
return switch (format) {
|
||||
1 => bb | (gg << 8) | (rr << 16), // bgrx
|
||||
else => rr | (gg << 8) | (bb << 16), // rgbx
|
||||
};
|
||||
}
|
||||
|
||||
test "pack encodes native byte order for rgbx and bgrx" {
|
||||
// rgbx: red in the low byte, blue in byte 2.
|
||||
try std.testing.expectEqual(@as(u32, 0x0000_00AA), pack(0, 0xAA, 0, 0));
|
||||
try std.testing.expectEqual(@as(u32, 0x00AA_0000), pack(0, 0, 0, 0xAA));
|
||||
// bgrx: blue in the low byte, red in byte 2.
|
||||
try std.testing.expectEqual(@as(u32, 0x0000_00AA), pack(1, 0, 0, 0xAA));
|
||||
try std.testing.expectEqual(@as(u32, 0x00AA_0000), pack(1, 0xAA, 0, 0));
|
||||
try std.testing.expectEqual(@as(u32, 0x0000_3020), pack(0, 0x20, 0x30, 0)); // green in byte 1
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
//! The scanout wire protocol — what the compositor says to a native scanout driver (e.g.
|
||||
//! virtio-gpu) over its well-known `.scanout` endpoint to put a composited frame on screen.
|
||||
//! The driver owns the panel and the shared scanout surface it handed the compositor (via the
|
||||
//! display service's `attach_scanout`); the compositor composites into that surface, then asks
|
||||
//! the driver to present a damaged rectangle. Tiny by design — one present request. Separate
|
||||
//! from the display protocol because the directions differ: clients call the compositor over
|
||||
//! `.display`; the compositor calls the driver over `.scanout`. See docs/display-v2.md.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
/// present(x, y, width, height): put the given rectangle of the shared scanout surface on
|
||||
/// the panel (on virtio-gpu: transfer-to-host of the region, then a fenced resource flush).
|
||||
present = 0,
|
||||
/// get_modes() -> ModesReply: the display modes this scanout can switch to (V5).
|
||||
get_modes = 1,
|
||||
/// set_mode(width, height): change the scanout resolution — the shared surface is sized to
|
||||
/// the largest mode, so this just re-points the scanout rectangle; the surface is unchanged.
|
||||
set_mode = 2,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
operation: u32,
|
||||
x: u32 = 0,
|
||||
y: u32 = 0,
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Reply = extern struct {
|
||||
status: i32, // 0 on success, negative on failure
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// One offered display mode.
|
||||
pub const Mode = extern struct { width: u32, height: u32 };
|
||||
pub const max_modes = 4;
|
||||
|
||||
/// The reply to `get_modes`: a small fixed list of modes.
|
||||
pub const ModesReply = extern struct {
|
||||
status: i32,
|
||||
count: u32,
|
||||
modes: [max_modes]Mode,
|
||||
};
|
||||
|
||||
pub const message_maximum: usize = 64;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||
+591
-37
@@ -17,21 +17,37 @@
|
||||
const std = @import("std");
|
||||
const on_disk = @import("on-disk.zig");
|
||||
|
||||
/// The most sectors one multi-sector transfer moves. Runs of contiguous full
|
||||
/// sectors within a cluster are coalesced into a single device command up to this
|
||||
/// cap; it bounds the DMA bounce buffer the fat server must provide (see
|
||||
/// `IpcBlock` in fat.zig — its bounce is `max_transfer_sectors * 512` bytes).
|
||||
pub const max_transfer_sectors = 8;
|
||||
|
||||
/// A block device the engine reads and writes in fixed-size blocks. The two
|
||||
/// function pointers let the same engine run over a real `.block` driver or a
|
||||
/// RAM buffer (the tests).
|
||||
/// function pointers move a *run* of `count` contiguous sectors in one call, so a
|
||||
/// bulk file read/write costs one device round-trip per run rather than one per
|
||||
/// sector; the same engine runs over a real `.block` driver (which turns a run
|
||||
/// into one multi-sector SCSI command) or a RAM buffer (the tests). Metadata
|
||||
/// accesses (FAT sectors, directory sectors) use the single-block convenience
|
||||
/// wrappers below.
|
||||
pub const BlockDevice = struct {
|
||||
context: *anyopaque,
|
||||
block_size: u32,
|
||||
block_count: u64,
|
||||
readBlockFn: *const fn (context: *anyopaque, lba: u64, buffer: []u8) bool,
|
||||
writeBlockFn: *const fn (context: *anyopaque, lba: u64, buffer: []const u8) bool,
|
||||
readBlocksFn: *const fn (context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool,
|
||||
writeBlocksFn: *const fn (context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool,
|
||||
|
||||
pub fn readBlocks(self: BlockDevice, lba: u64, count: u32, buffer: []u8) bool {
|
||||
return self.readBlocksFn(self.context, lba, count, buffer);
|
||||
}
|
||||
pub fn writeBlocks(self: BlockDevice, lba: u64, count: u32, buffer: []const u8) bool {
|
||||
return self.writeBlocksFn(self.context, lba, count, buffer);
|
||||
}
|
||||
pub fn readBlock(self: BlockDevice, lba: u64, buffer: []u8) bool {
|
||||
return self.readBlockFn(self.context, lba, buffer);
|
||||
return self.readBlocksFn(self.context, lba, 1, buffer);
|
||||
}
|
||||
pub fn writeBlock(self: BlockDevice, lba: u64, buffer: []const u8) bool {
|
||||
return self.writeBlockFn(self.context, lba, buffer);
|
||||
return self.writeBlocksFn(self.context, lba, 1, buffer);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -54,6 +70,20 @@ pub const Node = struct {
|
||||
const sector_size = 512;
|
||||
const entries_per_sector = sector_size / @sizeOf(on_disk.DirectoryEntry); // 16
|
||||
|
||||
// A small write-through cache of single-sector (metadata) accesses: FAT sectors,
|
||||
// directory sectors, and directory-entry writebacks. Its payoff is repeated scans
|
||||
// — resolving many paths under the same directory (a logging burst opening dozens
|
||||
// of files under /var/log/<stamp>/) re-reads the same directory and FAT sectors,
|
||||
// which now come from RAM instead of a USB round trip each. Bulk file data (the
|
||||
// multi-sector run path) bypasses the cache — it is large and not re-read — and a
|
||||
// run write invalidates any overlapping cached sector to stay coherent.
|
||||
const block_cache_lines = 16;
|
||||
const CacheLine = struct {
|
||||
lba: u64 = 0,
|
||||
valid: bool = false,
|
||||
data: [sector_size]u8 = undefined,
|
||||
};
|
||||
|
||||
pub const FileSystem = struct {
|
||||
device: BlockDevice,
|
||||
geometry: on_disk.Geometry,
|
||||
@@ -70,13 +100,84 @@ pub const FileSystem = struct {
|
||||
// server before a mutating op. 0 leaves the on-disk timestamps untouched (host
|
||||
// tests that don't care about time, and reads).
|
||||
current_time_epoch: u64 = 0,
|
||||
// Where the next allocateCluster scan starts — clusters below this were seen
|
||||
// in use, so a fresh scan needn't re-read them (frees rewind it). Without
|
||||
// this the scan re-read the FAT from cluster 2 per allocation: measured at
|
||||
// ~1 s/cluster on a part-full volume (a 37 s shutdown log flush).
|
||||
next_free_hint: u32 = 2,
|
||||
// Which absolute LBA `fat_sector` currently holds (0 = none). Lets a FAT
|
||||
// scan serve consecutive entries from one device read; every write through
|
||||
// the sector keeps the cache coherent (writeFatBytes updates it in place).
|
||||
fat_sector_lba: u64 = 0,
|
||||
// Write-through single-sector cache (see CacheLine). Keyed by filesystem-
|
||||
// relative LBA; round-robin eviction.
|
||||
cache: [block_cache_lines]CacheLine = [_]CacheLine{.{}} ** block_cache_lines,
|
||||
cache_cursor: u32 = 0,
|
||||
|
||||
// Every filesystem-relative sector access adds the partition base.
|
||||
// --- single-sector cache ------------------------------------------------
|
||||
|
||||
fn cacheFind(self: *FileSystem, lba: u64) ?*CacheLine {
|
||||
for (&self.cache) |*line| {
|
||||
if (line.valid and line.lba == lba) return line;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// Install `data` (one sector) for `lba`: refresh an existing line or evict the
|
||||
// next round-robin slot. Called on every device read (populate) and write
|
||||
// (write-through), so a hit always mirrors the device.
|
||||
fn cacheInstall(self: *FileSystem, lba: u64, data: []const u8) void {
|
||||
const line = self.cacheFind(lba) orelse blk: {
|
||||
const slot = &self.cache[self.cache_cursor];
|
||||
self.cache_cursor = (self.cache_cursor + 1) % block_cache_lines;
|
||||
slot.valid = true;
|
||||
slot.lba = lba;
|
||||
break :blk slot;
|
||||
};
|
||||
@memcpy(&line.data, data[0..sector_size]);
|
||||
}
|
||||
fn cacheInvalidateRange(self: *FileSystem, lba: u64, count: u32) void {
|
||||
for (&self.cache) |*line| {
|
||||
if (line.valid and line.lba >= lba and line.lba < lba + count) line.valid = false;
|
||||
}
|
||||
}
|
||||
|
||||
// Every filesystem-relative sector access adds the partition base. Single-sector
|
||||
// reads/writes go through the cache; the write path is write-through.
|
||||
fn blockRead(self: *FileSystem, lba: u64, buffer: []u8) bool {
|
||||
return self.device.readBlock(self.base_lba + lba, buffer);
|
||||
if (self.cacheFind(lba)) |line| {
|
||||
@memcpy(buffer[0..sector_size], &line.data);
|
||||
return true;
|
||||
}
|
||||
if (!self.device.readBlock(self.base_lba + lba, buffer)) return false;
|
||||
self.cacheInstall(lba, buffer);
|
||||
return true;
|
||||
}
|
||||
fn blockWrite(self: *FileSystem, lba: u64, buffer: []const u8) bool {
|
||||
return self.device.writeBlock(self.base_lba + lba, buffer);
|
||||
if (!self.device.writeBlock(self.base_lba + lba, buffer)) return false;
|
||||
self.cacheInstall(lba, buffer);
|
||||
return true;
|
||||
}
|
||||
// Write one sector without caching it: for bulk cluster-zeroing, whose sectors
|
||||
// are write-once and would only evict useful metadata. Still invalidates any
|
||||
// stale cached copy so a later read sees the zeros.
|
||||
fn blockWriteUncached(self: *FileSystem, lba: u64, buffer: []const u8) bool {
|
||||
if (!self.device.writeBlock(self.base_lba + lba, buffer)) return false;
|
||||
self.cacheInvalidateRange(lba, 1);
|
||||
return true;
|
||||
}
|
||||
// Move `count` contiguous full sectors in one device command. `buffer` must be
|
||||
// exactly count*sector_size and sector-aligned in meaning (whole sectors only);
|
||||
// callers use these for the aligned middle of a file transfer, falling back to
|
||||
// the single-sector read-modify-write path for partial head/tail sectors. Bulk
|
||||
// data bypasses the cache; a run write invalidates any overlapping cached
|
||||
// sector so metadata that happens to share the range never goes stale.
|
||||
fn blockReadRun(self: *FileSystem, lba: u64, count: u32, buffer: []u8) bool {
|
||||
return self.device.readBlocks(self.base_lba + lba, count, buffer);
|
||||
}
|
||||
fn blockWriteRun(self: *FileSystem, lba: u64, count: u32, buffer: []const u8) bool {
|
||||
if (!self.device.writeBlocks(self.base_lba + lba, count, buffer)) return false;
|
||||
self.cacheInvalidateRange(lba, count);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Mount the filesystem on `device`: either a bare FAT with its boot sector at
|
||||
@@ -139,7 +240,10 @@ pub const FileSystem = struct {
|
||||
while (done < out.len) {
|
||||
const lba = position / sector_size;
|
||||
const within: usize = @intCast(position % sector_size);
|
||||
if (!self.blockRead(lba, &self.fat_sector)) return false;
|
||||
if (lba != self.fat_sector_lba) {
|
||||
if (!self.blockRead(lba, &self.fat_sector)) return false;
|
||||
self.fat_sector_lba = lba;
|
||||
}
|
||||
const n = @min(out.len - done, sector_size - within);
|
||||
@memcpy(out[done .. done + n], self.fat_sector[within .. within + n]);
|
||||
done += n;
|
||||
@@ -159,7 +263,10 @@ pub const FileSystem = struct {
|
||||
while (done < in.len) {
|
||||
const lba = position / sector_size;
|
||||
const within: usize = @intCast(position % sector_size);
|
||||
if (!self.blockRead(lba, &self.fat_sector)) return false;
|
||||
if (lba != self.fat_sector_lba) {
|
||||
if (!self.blockRead(lba, &self.fat_sector)) return false;
|
||||
self.fat_sector_lba = lba;
|
||||
}
|
||||
const n = @min(in.len - done, sector_size - within);
|
||||
@memcpy(self.fat_sector[within .. within + n], in[done .. done + n]);
|
||||
if (!self.blockWrite(lba, &self.fat_sector)) return false;
|
||||
@@ -237,11 +344,19 @@ pub const FileSystem = struct {
|
||||
|
||||
// Find and claim a free cluster, marking it end-of-chain. Returns its number.
|
||||
fn allocateCluster(self: *FileSystem) ?u32 {
|
||||
var cluster: u32 = 2;
|
||||
while (cluster < self.geometry.cluster_count + 2) : (cluster += 1) {
|
||||
if (self.readFatEntry(cluster) == on_disk.free_cluster) {
|
||||
if (!self.writeFatEntry(cluster, self.endOfChainValue())) return null;
|
||||
return cluster;
|
||||
const limit = self.geometry.cluster_count + 2;
|
||||
// Two passes: hint..end, then 2..hint (the hint only skips known-used
|
||||
// ground, it never hides a freed cluster — freeChain rewinds it).
|
||||
var pass: u2 = 0;
|
||||
while (pass < 2) : (pass += 1) {
|
||||
var cluster: u32 = if (pass == 0) self.next_free_hint else 2;
|
||||
const end: u32 = if (pass == 0) limit else self.next_free_hint;
|
||||
while (cluster < end) : (cluster += 1) {
|
||||
if (self.readFatEntry(cluster) == on_disk.free_cluster) {
|
||||
if (!self.writeFatEntry(cluster, self.endOfChainValue())) return null;
|
||||
self.next_free_hint = cluster + 1;
|
||||
return cluster;
|
||||
}
|
||||
}
|
||||
}
|
||||
return null;
|
||||
@@ -257,6 +372,7 @@ pub const FileSystem = struct {
|
||||
while (cluster >= 2 and cluster < limit and guard < limit) : (guard += 1) {
|
||||
const next = self.readFatEntry(cluster);
|
||||
_ = self.writeFatEntry(cluster, on_disk.free_cluster);
|
||||
if (cluster < self.next_free_hint) self.next_free_hint = cluster;
|
||||
if (self.isEndOfChain(next) or next < 2) break;
|
||||
cluster = next;
|
||||
}
|
||||
@@ -294,7 +410,9 @@ pub const FileSystem = struct {
|
||||
var zero = [_]u8{0} ** sector_size;
|
||||
var s: u32 = 0;
|
||||
while (s < self.geometry.sectors_per_cluster) : (s += 1) {
|
||||
_ = self.blockWrite(self.clusterSector(cluster, s), &zero);
|
||||
// Uncached: a freshly-zeroed cluster is write-once bulk; caching its
|
||||
// sectors would only evict live metadata.
|
||||
_ = self.blockWriteUncached(self.clusterSector(cluster, s), &zero);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -520,6 +638,7 @@ pub const FileSystem = struct {
|
||||
const want = @min(buffer.len, available);
|
||||
const cluster_bytes = self.geometry.sectors_per_cluster * sector_size;
|
||||
|
||||
const spc = self.geometry.sectors_per_cluster;
|
||||
var produced: usize = 0;
|
||||
var position = offset;
|
||||
while (produced < want) {
|
||||
@@ -527,7 +646,23 @@ pub const FileSystem = struct {
|
||||
const in_cluster = position % cluster_bytes;
|
||||
const sector_in_cluster = in_cluster / sector_size;
|
||||
const in_sector = in_cluster % sector_size;
|
||||
if (!self.blockRead(self.clusterSector(cluster, sector_in_cluster), &self.sector)) break;
|
||||
const lba = self.clusterSector(cluster, sector_in_cluster);
|
||||
|
||||
// Aligned middle: read a run of whole sectors straight into the caller's
|
||||
// buffer in one device command (no per-sector round-trip, no staging
|
||||
// copy). Bounded by the sectors left in this cluster (the next cluster
|
||||
// may not be physically contiguous), the transfer cap, and the request.
|
||||
if (in_sector == 0 and want - produced >= sector_size) {
|
||||
const full = @min((want - produced) / sector_size, spc - sector_in_cluster);
|
||||
const run: u32 = @intCast(@min(full, max_transfer_sectors));
|
||||
if (!self.blockReadRun(lba, run, buffer[produced .. produced + run * sector_size])) break;
|
||||
produced += run * sector_size;
|
||||
position += run * sector_size;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Partial head or tail sector: read one sector, copy the slice needed.
|
||||
if (!self.blockRead(lba, &self.sector)) break;
|
||||
const n = @min(want - produced, sector_size - in_sector);
|
||||
@memcpy(buffer[produced .. produced + n], self.sector[in_sector .. in_sector + n]);
|
||||
produced += n;
|
||||
@@ -549,6 +684,7 @@ pub const FileSystem = struct {
|
||||
node.first_cluster = fresh;
|
||||
}
|
||||
|
||||
const spc = self.geometry.sectors_per_cluster;
|
||||
var consumed: usize = 0;
|
||||
var position = offset;
|
||||
while (consumed < data.len) {
|
||||
@@ -557,7 +693,22 @@ pub const FileSystem = struct {
|
||||
const sector_in_cluster = in_cluster / sector_size;
|
||||
const in_sector = in_cluster % sector_size;
|
||||
const lba = self.clusterSector(cluster, sector_in_cluster);
|
||||
// Read-modify-write the sector for a partial write.
|
||||
|
||||
// Aligned middle: write a run of whole sectors straight from the caller's
|
||||
// data in one device command. These sectors are fully overwritten, so no
|
||||
// read-modify-write is needed — a double saving over the per-sector RMW
|
||||
// path. Bounded by the sectors left in this (already-resolved) cluster,
|
||||
// the transfer cap, and the data remaining.
|
||||
if (in_sector == 0 and data.len - consumed >= sector_size) {
|
||||
const full = @min((data.len - consumed) / sector_size, spc - sector_in_cluster);
|
||||
const run: u32 = @intCast(@min(full, max_transfer_sectors));
|
||||
if (!self.blockWriteRun(lba, run, data[consumed .. consumed + run * sector_size])) break;
|
||||
consumed += run * sector_size;
|
||||
position += run * sector_size;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Partial head or tail sector: read-modify-write one sector.
|
||||
if (!self.blockRead(lba, &self.sector)) break;
|
||||
const n = @min(data.len - consumed, sector_size - in_sector);
|
||||
@memcpy(self.sector[in_sector .. in_sector + n], data[consumed .. consumed + n]);
|
||||
@@ -603,6 +754,192 @@ pub const FileSystem = struct {
|
||||
_ = self.blockWrite(node.entry_sector, &self.dir_sector);
|
||||
}
|
||||
|
||||
// --- long-name creation --------------------------------------------------
|
||||
|
||||
// The standard 8.3 short-name checksum carried by every long-name entry.
|
||||
fn shortChecksum(raw: [11]u8) u8 {
|
||||
var sum: u8 = 0;
|
||||
for (raw) |c| sum = ((sum & 1) << 7) +% (sum >> 1) +% c;
|
||||
return sum;
|
||||
}
|
||||
|
||||
fn valid83Char(c: u8) bool {
|
||||
return (c >= 'A' and c <= 'Z') or (c >= '0' and c <= '9') or c == '-' or c == '_';
|
||||
}
|
||||
|
||||
// Whether an 8.3 entry with exactly this raw name exists in `dir`.
|
||||
const RawContext = struct { raw: [11]u8, found: *bool };
|
||||
fn rawVisit(context: *const RawContext, entry: on_disk.DirectoryEntry, name: []const u8, entry_sector: u64, entry_offset: u32) bool {
|
||||
_ = name;
|
||||
_ = entry_sector;
|
||||
_ = entry_offset;
|
||||
if (std.mem.eql(u8, &entry.name, &context.raw)) {
|
||||
context.found.* = true;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
fn shortNameExists(self: *FileSystem, dir: Node, raw: [11]u8) bool {
|
||||
var found = false;
|
||||
var context = RawContext{ .raw = raw, .found = &found };
|
||||
self.scanDirectory(dir, &context, rawVisit);
|
||||
return found;
|
||||
}
|
||||
|
||||
// A mangled STEM~N.EXT short name that collides with nothing in `dir` — the
|
||||
// alias behind a long-name chain.
|
||||
fn shortNameFor(self: *FileSystem, dir: Node, name: []const u8) ?[11]u8 {
|
||||
const dot = std.mem.lastIndexOfScalar(u8, name, '.');
|
||||
const base = if (dot) |d| name[0..d] else name;
|
||||
const ext = if (dot) |d| name[d + 1 ..] else name[0..0];
|
||||
|
||||
var stem: [6]u8 = undefined;
|
||||
var stem_len: usize = 0;
|
||||
for (base) |c| {
|
||||
if (stem_len == stem.len) break;
|
||||
const upper = std.ascii.toUpper(c);
|
||||
if (valid83Char(upper)) {
|
||||
stem[stem_len] = upper;
|
||||
stem_len += 1;
|
||||
}
|
||||
}
|
||||
if (stem_len == 0) {
|
||||
stem[0] = 'X';
|
||||
stem_len = 1;
|
||||
}
|
||||
|
||||
var raw = [_]u8{' '} ** 11;
|
||||
var ext_len: usize = 0;
|
||||
for (ext) |c| {
|
||||
if (ext_len == 3) break;
|
||||
const upper = std.ascii.toUpper(c);
|
||||
if (valid83Char(upper)) {
|
||||
raw[8 + ext_len] = upper;
|
||||
ext_len += 1;
|
||||
}
|
||||
}
|
||||
|
||||
var index: u32 = 1;
|
||||
while (index <= 999_999) : (index += 1) {
|
||||
var tail_buffer: [8]u8 = undefined;
|
||||
const tail = std.fmt.bufPrint(&tail_buffer, "~{d}", .{index}) catch return null;
|
||||
const keep = @min(stem_len, 8 - tail.len);
|
||||
@memset(raw[0..8], ' ');
|
||||
@memcpy(raw[0..keep], stem[0..keep]);
|
||||
@memcpy(raw[keep .. keep + tail.len], tail);
|
||||
if (!self.shortNameExists(dir, raw)) return raw;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Fill one long-name entry's 13 UTF-16 slots from `name` starting at
|
||||
// `offset`: the name's bytes widened, then a 0x0000 terminator, then 0xFFFF.
|
||||
fn fillLongNamePiece(lfn: *on_disk.LongNameEntry, name: []const u8, offset: usize) void {
|
||||
var units: [13]u16 = undefined;
|
||||
var i: usize = 0;
|
||||
while (i < 13) : (i += 1) {
|
||||
const at = offset + i;
|
||||
units[i] = if (at < name.len) name[at] else if (at == name.len) 0x0000 else 0xFFFF;
|
||||
}
|
||||
lfn.name1 = units[0..5].*;
|
||||
lfn.name2 = units[5..11].*;
|
||||
lfn.name3 = units[11..13].*;
|
||||
}
|
||||
|
||||
// The first entry index of a run of `count` free slots in `dir`, growing the
|
||||
// directory as needed. Fresh clusters are zeroed, so growth always yields
|
||||
// free slots; only the fixed FAT12/16 root can genuinely run out.
|
||||
fn findFreeRun(self: *FileSystem, dir: Node, count: usize) ?u32 {
|
||||
var run_start: u32 = 0;
|
||||
var run_len: usize = 0;
|
||||
var sector_index: u32 = 0;
|
||||
while (self.dirSectorLba(dir, sector_index, true)) |lba| : (sector_index += 1) {
|
||||
if (!self.blockRead(lba, &self.dir_sector)) return null;
|
||||
var i: u32 = 0;
|
||||
while (i < entries_per_sector) : (i += 1) {
|
||||
const offset = i * @sizeOf(on_disk.DirectoryEntry);
|
||||
const entry = std.mem.bytesToValue(on_disk.DirectoryEntry, self.dir_sector[offset .. offset + @sizeOf(on_disk.DirectoryEntry)]);
|
||||
if (entry.isFree()) {
|
||||
if (run_len == 0) run_start = sector_index * entries_per_sector + i;
|
||||
run_len += 1;
|
||||
if (run_len == count) return run_start;
|
||||
} else {
|
||||
run_len = 0;
|
||||
}
|
||||
}
|
||||
if (sector_index > 4096) return null; // runaway guard
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Write one 32-byte directory entry at a global entry index (read-modify-
|
||||
// write of its sector). Returns the entry's (sector, offset) or null.
|
||||
fn writeEntryAt(self: *FileSystem, dir: Node, index: u32, bytes: *const [32]u8) ?EntryLoc {
|
||||
const lba = self.dirSectorLba(dir, index / entries_per_sector, true) orelse return null;
|
||||
if (!self.blockRead(lba, &self.dir_sector)) return null;
|
||||
const offset = (index % entries_per_sector) * @sizeOf(on_disk.DirectoryEntry);
|
||||
@memcpy(self.dir_sector[offset .. offset + 32], bytes);
|
||||
if (!self.blockWrite(lba, &self.dir_sector)) return null;
|
||||
return .{ .sector = lba, .offset = offset };
|
||||
}
|
||||
|
||||
// Add a named directory entry, creating a long-name chain when the name is
|
||||
// not its own 8.3 form. Write order is LFN pieces first, 8.3 entry last: an
|
||||
// interrupted create leaves orphaned long-name entries, which every FAT
|
||||
// reader (this engine's scanner included) skips as unattached — never a
|
||||
// mismatched chain.
|
||||
fn addEntryNamed(self: *FileSystem, dir: Node, name: []const u8, attributes: u8, first_cluster: u32, size: u32) ?Node {
|
||||
if (to83(name)) |raw| {
|
||||
var display: [12]u8 = undefined;
|
||||
// Only a name that IS its 8.3 form (already uppercase) skips the
|
||||
// chain — a lowercase name gets one so its exact case survives,
|
||||
// matching tools/make-fat-image.py.
|
||||
if (std.mem.eql(u8, format83(raw, &display), name))
|
||||
return self.addEntry(dir, raw, attributes, first_cluster, size);
|
||||
}
|
||||
if (name.len == 0 or name.len > 255) return null;
|
||||
|
||||
const raw = self.shortNameFor(dir, name) orelse return null;
|
||||
const checksum = shortChecksum(raw);
|
||||
const piece_count: u32 = @intCast((name.len + 12) / 13);
|
||||
if (piece_count > 20) return null;
|
||||
const start = self.findFreeRun(dir, piece_count + 1) orelse return null;
|
||||
|
||||
var k: u32 = 0;
|
||||
while (k < piece_count) : (k += 1) {
|
||||
const piece = piece_count - k; // stored last-logical-first
|
||||
var lfn = std.mem.zeroes(on_disk.LongNameEntry);
|
||||
lfn.order = @intCast(piece | (if (k == 0) @as(u8, 0x40) else 0));
|
||||
lfn.attributes = on_disk.attribute_long_name;
|
||||
lfn.checksum = checksum;
|
||||
fillLongNamePiece(&lfn, name, (piece - 1) * 13);
|
||||
_ = self.writeEntryAt(dir, start + k, std.mem.asBytes(&lfn)[0..32]) orelse return null;
|
||||
}
|
||||
|
||||
var entry = std.mem.zeroes(on_disk.DirectoryEntry);
|
||||
entry.name = raw;
|
||||
entry.attributes = attributes;
|
||||
entry.file_size = size;
|
||||
entry.setFirstCluster(first_cluster);
|
||||
const stamp = on_disk.epochToFatDateTime(self.current_time_epoch);
|
||||
entry.creation_date = stamp.date;
|
||||
entry.creation_time = stamp.time;
|
||||
entry.write_date = stamp.date;
|
||||
entry.write_time = stamp.time;
|
||||
entry.last_access_date = stamp.date;
|
||||
const location = self.writeEntryAt(dir, start + piece_count, std.mem.asBytes(&entry)[0..32]) orelse return null;
|
||||
return .{
|
||||
.first_cluster = first_cluster,
|
||||
.size = size,
|
||||
.is_directory = attributes & on_disk.attribute_directory != 0,
|
||||
.mtime = self.current_time_epoch,
|
||||
.entry_sector = location.sector,
|
||||
.entry_offset = location.offset,
|
||||
.has_entry = true,
|
||||
};
|
||||
}
|
||||
|
||||
// Add an 8.3 directory entry to `dir` with the given attributes, first cluster,
|
||||
// and size, reusing a free (0x00 or 0xE5) slot and growing the directory chain
|
||||
// if needed. Returns the new node (with its entry location) or null if full.
|
||||
@@ -645,19 +982,20 @@ pub const FileSystem = struct {
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Create an 8.3-named file in directory `dir`. Returns the new (empty) node,
|
||||
/// or null if the name is not 8.3-representable or no directory slot is free.
|
||||
/// Create a file in directory `dir`. Uppercase 8.3 names get a bare short
|
||||
/// entry; anything else gets a long-name chain over a mangled ~N alias.
|
||||
/// Returns the new (empty) node, or null (bad name / directory full /
|
||||
/// duplicate — the caller checks existence first if it must distinguish).
|
||||
pub fn createFile(self: *FileSystem, dir: Node, name: []const u8) ?Node {
|
||||
const raw = to83(name) orelse return null;
|
||||
return self.addEntry(dir, raw, on_disk.attribute_archive, 0, 0);
|
||||
return self.addEntryNamed(dir, name, on_disk.attribute_archive, 0, 0);
|
||||
}
|
||||
|
||||
/// Create an 8.3-named subdirectory in `dir`: allocate and initialise its first
|
||||
/// Create a subdirectory in `dir`: allocate and initialise its first
|
||||
/// cluster with "." (itself) and ".." (the parent) entries, then add its
|
||||
/// directory entry to `dir`. Returns the new directory node, or null (bad name,
|
||||
/// no free cluster, or the directory is full). Long names are not created.
|
||||
/// directory entry to `dir` (long-name chain when the name needs one).
|
||||
/// Returns the new directory node, or null (bad name, no free cluster, or
|
||||
/// the directory is full).
|
||||
pub fn createDirectory(self: *FileSystem, dir: Node, name: []const u8) ?Node {
|
||||
const raw = to83(name) orelse return null;
|
||||
const cluster = self.allocateCluster() orelse return null;
|
||||
self.zeroCluster(cluster);
|
||||
|
||||
@@ -685,7 +1023,7 @@ pub const FileSystem = struct {
|
||||
self.freeChain(cluster);
|
||||
return null;
|
||||
}
|
||||
return self.addEntry(dir, raw, on_disk.attribute_directory, cluster, 0) orelse {
|
||||
return self.addEntryNamed(dir, name, on_disk.attribute_directory, cluster, 0) orelse {
|
||||
self.freeChain(cluster);
|
||||
return null;
|
||||
};
|
||||
@@ -838,18 +1176,29 @@ pub const FileSystem = struct {
|
||||
|
||||
const RamDisk = struct {
|
||||
bytes: []u8,
|
||||
fn readBlock(context: *anyopaque, lba: u64, buffer: []u8) bool {
|
||||
// Instrumentation for the multi-sector tests: how many device commands were
|
||||
// issued, and the largest run (in sectors) any single command carried.
|
||||
reads: usize = 0,
|
||||
writes: usize = 0,
|
||||
max_run: u32 = 0,
|
||||
fn readBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool {
|
||||
const self: *RamDisk = @ptrCast(@alignCast(context));
|
||||
const len = @as(usize, count) * sector_size;
|
||||
const start = lba * sector_size;
|
||||
if (start + sector_size > self.bytes.len) return false;
|
||||
@memcpy(buffer[0..sector_size], self.bytes[start .. start + sector_size]);
|
||||
if (start + len > self.bytes.len or buffer.len < len) return false;
|
||||
@memcpy(buffer[0..len], self.bytes[start .. start + len]);
|
||||
self.reads += 1;
|
||||
if (count > self.max_run) self.max_run = count;
|
||||
return true;
|
||||
}
|
||||
fn writeBlock(context: *anyopaque, lba: u64, buffer: []const u8) bool {
|
||||
fn writeBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool {
|
||||
const self: *RamDisk = @ptrCast(@alignCast(context));
|
||||
const len = @as(usize, count) * sector_size;
|
||||
const start = lba * sector_size;
|
||||
if (start + sector_size > self.bytes.len) return false;
|
||||
@memcpy(self.bytes[start .. start + sector_size], buffer[0..sector_size]);
|
||||
if (start + len > self.bytes.len or buffer.len < len) return false;
|
||||
@memcpy(self.bytes[start .. start + len], buffer[0..len]);
|
||||
self.writes += 1;
|
||||
if (count > self.max_run) self.max_run = count;
|
||||
return true;
|
||||
}
|
||||
fn device(self: *RamDisk) BlockDevice {
|
||||
@@ -857,8 +1206,8 @@ const RamDisk = struct {
|
||||
.context = self,
|
||||
.block_size = sector_size,
|
||||
.block_count = self.bytes.len / sector_size,
|
||||
.readBlockFn = readBlock,
|
||||
.writeBlockFn = writeBlock,
|
||||
.readBlocksFn = readBlocks,
|
||||
.writeBlocksFn = writeBlocks,
|
||||
};
|
||||
}
|
||||
};
|
||||
@@ -867,13 +1216,17 @@ const RamDisk = struct {
|
||||
// two reserved entries, an empty root directory. Enough for the engine to mount
|
||||
// and operate on.
|
||||
fn formatFat16(bytes: []u8) void {
|
||||
formatFat16Spc(bytes, 1);
|
||||
}
|
||||
|
||||
fn formatFat16Spc(bytes: []u8, sectors_per_cluster: u8) void {
|
||||
@memset(bytes, 0);
|
||||
const total_sectors: u16 = @intCast(bytes.len / sector_size);
|
||||
var bpb = std.mem.zeroes(on_disk.BiosParameterBlock);
|
||||
bpb.jump = .{ 0xEB, 0x3C, 0x90 };
|
||||
bpb.oem_name = "MSWIN4.1".*;
|
||||
bpb.bytes_per_sector = sector_size;
|
||||
bpb.sectors_per_cluster = 1;
|
||||
bpb.sectors_per_cluster = sectors_per_cluster;
|
||||
bpb.reserved_sector_count = 1;
|
||||
bpb.fat_count = 2;
|
||||
bpb.root_entry_count = 512;
|
||||
@@ -948,6 +1301,97 @@ test "create, write, read back a file through the engine" {
|
||||
try std.testing.expect(fs.listEntry(fs.rootNode(), 1) == null);
|
||||
}
|
||||
|
||||
test "the block cache serves repeated metadata reads and stays write-through coherent" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
var node = fs.createFile(fs.rootNode(), "CACHE.TXT").?;
|
||||
var small = [_]u8{0x5A} ** 64;
|
||||
_ = fs.writeFile(&node, 0, &small);
|
||||
|
||||
// First resolve warms the cache; the second re-reads the same directory (and
|
||||
// FAT) sectors, which must now come entirely from RAM — zero device reads.
|
||||
_ = fs.resolve("/CACHE.TXT").?;
|
||||
disk.reads = 0;
|
||||
const again = fs.resolve("/CACHE.TXT").?;
|
||||
try std.testing.expectEqual(@as(usize, 0), disk.reads);
|
||||
try std.testing.expectEqual(@as(u32, small.len), again.size);
|
||||
|
||||
// Write-through coherence: a mid-file overwrite reaches the device (a fresh
|
||||
// mount, cold cache, reads the new bytes back) and the cache reflects it too.
|
||||
var patch = [_]u8{0xC3} ** 16;
|
||||
_ = fs.writeFile(&node, 16, &patch);
|
||||
var cached_back: [16]u8 = undefined;
|
||||
_ = fs.readFile(fs.resolve("/CACHE.TXT").?, 16, &cached_back);
|
||||
try std.testing.expectEqualSlices(u8, &patch, &cached_back);
|
||||
|
||||
var cold = FileSystem.mount(disk.device()).?; // cold cache, reads from device
|
||||
var device_back: [16]u8 = undefined;
|
||||
_ = cold.readFile(cold.resolve("/CACHE.TXT").?, 16, &device_back);
|
||||
try std.testing.expectEqualSlices(u8, &patch, &device_back);
|
||||
}
|
||||
|
||||
test "multi-sector transfers coalesce and round-trip identically" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 8000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
// 8 sectors per cluster: aligned runs can hit the max_transfer_sectors cap.
|
||||
formatFat16Spc(bytes, 8);
|
||||
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
// A payload that is many whole sectors plus a partial tail, all sector-aligned
|
||||
// at the start (offset 0): the aligned-run path should carry the bulk.
|
||||
const size = 20 * sector_size + 100;
|
||||
const payload = try allocator.alloc(u8, size);
|
||||
defer allocator.free(payload);
|
||||
for (payload, 0..) |*b, i| b.* = @truncate(i * 7 + 3);
|
||||
|
||||
var node = fs.createFile(fs.rootNode(), "BIG.BIN").?;
|
||||
const written = fs.writeFile(&node, 0, payload);
|
||||
try std.testing.expectEqual(size, written);
|
||||
|
||||
// Overwrite the same region (no allocation, no cluster-zeroing, no FAT writes),
|
||||
// so the command count reflects only data movement. The 21 sectors it spans
|
||||
// (three clusters of 8) coalesce into a handful of runs — far fewer than the
|
||||
// >=21 writes a per-sector engine would issue, and each full sector skips the
|
||||
// read-modify-write entirely.
|
||||
disk.writes = 0;
|
||||
disk.max_run = 0;
|
||||
_ = fs.writeFile(&node, 0, payload);
|
||||
// A per-sector engine would issue >=21 writes (one RMW per sector); coalescing
|
||||
// plus skip-RMW-on-full-sector brings it to the three aligned runs + one partial
|
||||
// tail + the directory-entry writeback.
|
||||
try std.testing.expect(disk.max_run > 1);
|
||||
try std.testing.expect(disk.writes <= 5);
|
||||
|
||||
const resolved = fs.resolve("/BIG.BIN").?;
|
||||
try std.testing.expectEqual(@as(u32, size), resolved.size);
|
||||
|
||||
// Read back the whole file and compare byte-for-byte.
|
||||
const readback = try allocator.alloc(u8, size);
|
||||
defer allocator.free(readback);
|
||||
disk.reads = 0;
|
||||
disk.max_run = 0;
|
||||
const got = fs.readFile(resolved, 0, readback);
|
||||
try std.testing.expectEqual(size, got);
|
||||
try std.testing.expectEqualSlices(u8, payload, readback);
|
||||
try std.testing.expect(disk.max_run > 1);
|
||||
try std.testing.expect(disk.reads < 21);
|
||||
|
||||
// Equivalence at an unaligned offset spanning a cluster boundary: same bytes as
|
||||
// a naive per-sector engine would produce (the payload we wrote).
|
||||
var window: [1000]u8 = undefined;
|
||||
const n = fs.readFile(resolved, 300, &window);
|
||||
try std.testing.expectEqual(@as(usize, 1000), n);
|
||||
try std.testing.expectEqualSlices(u8, payload[300 .. 300 + 1000], window[0..1000]);
|
||||
}
|
||||
|
||||
test "truncate frees the chain and zeroes the file" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
@@ -1103,3 +1547,113 @@ test "a create stamps the modification time" {
|
||||
try std.testing.expectEqual(@as(u64, 1_700_000_000), fs.resolve("/STAMP.TXT").?.mtime);
|
||||
try std.testing.expectEqual(@as(u64, 1_700_000_000), fs.listEntry(fs.rootNode(), 0).?.mtime);
|
||||
}
|
||||
|
||||
test "long-name create: directory + file round-trip by long name" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
// The per-boot log directory shape: an 18-char stamp, nested paths, .log names.
|
||||
const stamp_dir = fs.createDirectory(fs.rootNode(), "2026-07-21T101530Z").?;
|
||||
try std.testing.expect(stamp_dir.is_directory);
|
||||
const file = fs.createFile(stamp_dir, "device-manager.log").?;
|
||||
_ = file;
|
||||
|
||||
// Resolve by exact long name, and case-insensitively (FAT semantics).
|
||||
try std.testing.expect(fs.resolve("/2026-07-21T101530Z/device-manager.log") != null);
|
||||
try std.testing.expect(fs.resolve("/2026-07-21t101530z/DEVICE-MANAGER.LOG") != null);
|
||||
|
||||
// The listing shows the long names, not the ~N aliases.
|
||||
var listing = fs.listEntry(fs.rootNode(), 0).?;
|
||||
try std.testing.expectEqualStrings("2026-07-21T101530Z", listing.name_buffer[0..listing.name_len]);
|
||||
var inner = fs.listEntry(stamp_dir, 2).?; // after "." and ".."
|
||||
try std.testing.expectEqualStrings("device-manager.log", inner.name_buffer[0..inner.name_len]);
|
||||
|
||||
// Write through the created file and read it back by long-name resolve.
|
||||
var node = fs.resolve("/2026-07-21T101530Z/device-manager.log").?;
|
||||
try std.testing.expectEqual(@as(usize, 10), fs.writeFile(&node, 0, "hello logs"));
|
||||
var buffer: [16]u8 = undefined;
|
||||
try std.testing.expectEqual(@as(usize, 10), fs.readFile(node, 0, buffer[0..10]));
|
||||
try std.testing.expectEqualStrings("hello logs", buffer[0..10]);
|
||||
}
|
||||
|
||||
test "long-name create: ~N alias collision suffixes stay distinct" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
_ = fs.createFile(fs.rootNode(), "logger-alpha.log").?;
|
||||
_ = fs.createFile(fs.rootNode(), "logger-beta.log").?;
|
||||
// Same 6-char mangle stem (LOGGER) — the second must take ~2.
|
||||
var raw_one = false;
|
||||
var raw_two = false;
|
||||
var cursor: u32 = 0;
|
||||
while (fs.listEntry(fs.rootNode(), cursor)) |entry| : (cursor += 1) {
|
||||
if (std.mem.eql(u8, entry.name_buffer[0..entry.name_len], "logger-alpha.log")) raw_one = true;
|
||||
if (std.mem.eql(u8, entry.name_buffer[0..entry.name_len], "logger-beta.log")) raw_two = true;
|
||||
}
|
||||
try std.testing.expect(raw_one and raw_two);
|
||||
try std.testing.expect(fs.resolve("/logger-alpha.log") != null);
|
||||
try std.testing.expect(fs.resolve("/logger-beta.log") != null);
|
||||
// Their short aliases took distinct ~N tails. (Alias LOOKUP is not a
|
||||
// feature — findChild matches display names — but the on-disk aliases
|
||||
// must not collide for other FAT readers.)
|
||||
try std.testing.expect(fs.shortNameExists(fs.rootNode(), "LOGGER~1LOG".*));
|
||||
try std.testing.expect(fs.shortNameExists(fs.rootNode(), "LOGGER~2LOG".*));
|
||||
}
|
||||
|
||||
test "long-name create: unlink removes the chain; slots are reused cleanly" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
_ = fs.createFile(fs.rootNode(), "a-rather-long-file-name.txt").?;
|
||||
try std.testing.expect(fs.removeFile(fs.rootNode(), "a-rather-long-file-name.txt"));
|
||||
try std.testing.expect(fs.resolve("/a-rather-long-file-name.txt") == null);
|
||||
|
||||
// A new long name reuses the freed run without inheriting the old chain.
|
||||
_ = fs.createFile(fs.rootNode(), "an-entirely-different-name.md").?;
|
||||
try std.testing.expect(fs.resolve("/an-entirely-different-name.md") != null);
|
||||
try std.testing.expect(fs.resolve("/a-rather-long-file-name.txt") == null);
|
||||
var listing = fs.listEntry(fs.rootNode(), 0).?;
|
||||
try std.testing.expectEqualStrings("an-entirely-different-name.md", listing.name_buffer[0..listing.name_len]);
|
||||
}
|
||||
|
||||
test "8.3 fast path: an uppercase-compliant name gets one bare entry" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
_ = fs.createFile(fs.rootNode(), "DANOS.LOG").?;
|
||||
// Exactly one directory entry: entry 0 is the file, entry 1 is the end.
|
||||
var listing = fs.listEntry(fs.rootNode(), 0).?;
|
||||
try std.testing.expectEqualStrings("DANOS.LOG", listing.name_buffer[0..listing.name_len]);
|
||||
try std.testing.expect(fs.listEntry(fs.rootNode(), 1) == null);
|
||||
// A lowercase 8.3-shaped name is case-preserved via a chain instead.
|
||||
_ = fs.createFile(fs.rootNode(), "fat.log").?;
|
||||
var second = fs.listEntry(fs.rootNode(), 1).?;
|
||||
try std.testing.expectEqualStrings("fat.log", second.name_buffer[0..second.name_len]);
|
||||
}
|
||||
|
||||
test "short-name checksum matches the reference vector" {
|
||||
// "README TXT" is a widely published example: checksum 0x15... compute a
|
||||
// fixed pair to pin the rotate-add against regressions.
|
||||
const a = FileSystem.shortChecksum("README TXT".*);
|
||||
const b = FileSystem.shortChecksum("LOGGER~1LOG".*);
|
||||
try std.testing.expect(a != b);
|
||||
// The algorithm is order-sensitive: swapped bytes change the sum.
|
||||
const c = FileSystem.shortChecksum("REDAME TXT".*);
|
||||
try std.testing.expect(a != c);
|
||||
}
|
||||
|
||||
@@ -101,8 +101,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
_ = runtime.system.write("fat-test: root listing was empty\n");
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
+103
-43
@@ -16,35 +16,39 @@ const on_disk = @import("on-disk.zig");
|
||||
const protocol = runtime.vfs_protocol;
|
||||
const dma = runtime.dma;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
const mount_point = "/mnt/usb";
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
// buffer the driver reads/writes by physical address.
|
||||
const IpcBlock = struct {
|
||||
device: runtime.block.Device,
|
||||
bounce: dma.Region,
|
||||
bounce: dma.Region, // engine.max_transfer_sectors * 512 bytes
|
||||
|
||||
fn readBlock(context: *anyopaque, lba: u64, buffer: []u8) bool {
|
||||
fn readBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (!self.device.read(lba, 1, self.bounce.physical)) return false;
|
||||
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||
const len = count * 512;
|
||||
if (!self.device.read(lba, count, self.bounce.physical)) return false;
|
||||
const source: [*]const u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(buffer[0..512], source[0..512]);
|
||||
@memcpy(buffer[0..len], source[0..len]);
|
||||
return true;
|
||||
}
|
||||
fn writeBlock(context: *anyopaque, lba: u64, buffer: []const u8) bool {
|
||||
fn writeBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||
const len = count * 512;
|
||||
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(destination[0..512], buffer[0..512]);
|
||||
return self.device.write(lba, 1, self.bounce.physical);
|
||||
@memcpy(destination[0..len], buffer[0..len]);
|
||||
if (!self.device.write(lba, count, self.bounce.physical)) return false;
|
||||
device_dirty = true; // a block reached the device; a close will flush it
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
var ipc_block: IpcBlock = undefined;
|
||||
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
|
||||
// Open handles the VFS holds against this backend: each maps a node id to a
|
||||
@@ -76,43 +80,93 @@ fn fail(out: []u8) usize {
|
||||
return writeReply(out, .{ .status = -1 }, &.{});
|
||||
}
|
||||
|
||||
/// How often to look for a block device while none is mounted. Storage arriving
|
||||
/// is EVENT-shaped (the usb chain registering, possibly after a driver restart),
|
||||
/// but the registry has no subscription — a slow poll from our own harness loop
|
||||
/// keeps the service responsive (ping, terminate) while it waits, and keeps it
|
||||
/// alive to catch storage that appears LATE (a restarted usb-storage after a
|
||||
/// transient failure — the resilience half of docs/logging.md's storage story).
|
||||
const mount_retry_ms = 500;
|
||||
|
||||
var mounted = false;
|
||||
var service_endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = runtime.system.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
const device = runtime.block.open() orelse {
|
||||
_ = runtime.system.write("/system/services/fat: no block device (no storage attached)\n");
|
||||
return false; // clean exit: nothing to serve
|
||||
};
|
||||
// With the router in the kernel, clients hold OUR node ids directly; sweep
|
||||
// a dead client's open handles via the published exit events (the pattern
|
||||
// the old userspace router used for its own table).
|
||||
_ = runtime.process.subscribeExits(endpoint);
|
||||
tryBringUp();
|
||||
if (!mounted) _ = runtime.system.timerOnce(endpoint, mount_retry_ms);
|
||||
return true; // serve regardless: requests fail politely until storage mounts
|
||||
}
|
||||
|
||||
/// One storage bring-up attempt: block device -> FAT mount -> VFS mounts. Sets
|
||||
/// `mounted` on success; a failure leaves everything untouched for the next tick.
|
||||
fn tryBringUp() void {
|
||||
if (mounted) return;
|
||||
const device = runtime.block.tryOpen() orelse return;
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = runtime.system.write("/system/services/fat: block geometry unavailable\n");
|
||||
return false;
|
||||
return;
|
||||
};
|
||||
ipc_block = .{ .device = device, .bounce = dma.alloc(4096, dma.coherent) orelse return false };
|
||||
const bounce = dma.alloc(engine.max_transfer_sectors * 512, dma.coherent) orelse return;
|
||||
ipc_block = .{ .device = device, .bounce = bounce };
|
||||
|
||||
const block_device = engine.BlockDevice{
|
||||
.context = &ipc_block,
|
||||
.block_size = geometry.block_size,
|
||||
.block_count = geometry.block_count,
|
||||
.readBlockFn = IpcBlock.readBlock,
|
||||
.writeBlockFn = IpcBlock.writeBlock,
|
||||
.readBlocksFn = IpcBlock.readBlocks,
|
||||
.writeBlocksFn = IpcBlock.writeBlocks,
|
||||
};
|
||||
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||
_ = runtime.system.write("/system/services/fat: not a FAT filesystem\n");
|
||||
return false;
|
||||
return;
|
||||
};
|
||||
writeLine("/system/services/fat: mounted FAT ({s}, {d} clusters, partition lba {d})\n", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
|
||||
// Mount ourselves into the VFS namespace at /mnt/usb (retry while the VFS
|
||||
// comes up). From here the VFS routes /mnt/usb/... to this server.
|
||||
var tries: u32 = 0;
|
||||
while (tries < 100) : (tries += 1) {
|
||||
if (runtime.fs.mount(mount_point, endpoint)) {
|
||||
writeLine("/system/services/fat: mounted {s}\n", .{mount_point});
|
||||
return true;
|
||||
}
|
||||
runtime.system.sleep(50);
|
||||
// Mount ourselves into the kernel VFS at /mnt/usb — and serve /var from the
|
||||
// volume's /var subtree, so FHS paths (the logger's /var/log) stay decoupled
|
||||
// from which volume carries them.
|
||||
if (runtime.fs.mount(mount_point, endpointForMount())) {
|
||||
std.log.info("mounted {s}", .{mount_point});
|
||||
} else {
|
||||
_ = runtime.system.write("/system/services/fat: could not mount /mnt/usb\n");
|
||||
}
|
||||
_ = runtime.system.write("/system/services/fat: could not mount into the VFS\n");
|
||||
return true; // still serve directly, even if the namespace mount didn't take
|
||||
if (runtime.fs.mountRewritten("/var", endpointForMount(), "/var")) {
|
||||
std.log.info("mounted /var", .{});
|
||||
} else {
|
||||
_ = runtime.system.write("/system/services/fat: could not mount /var\n");
|
||||
}
|
||||
mounted = true;
|
||||
}
|
||||
|
||||
fn endpointForMount() runtime.ipc.Handle {
|
||||
return service_endpoint;
|
||||
}
|
||||
|
||||
/// A subscribed process-exit event: release every open handle the dead client
|
||||
/// held, so a crashed reader can't pin table slots (or, later, locks).
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = runtime.ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
tryBringUp();
|
||||
if (!mounted) _ = runtime.system.timerOnce(service_endpoint, mount_retry_ms);
|
||||
return;
|
||||
}
|
||||
if (!got.isChildExit()) return;
|
||||
const dead = got.childProcessId();
|
||||
var released: u32 = 0;
|
||||
for (&open_nodes) |*o| {
|
||||
if (o.used and o.owner == dead) {
|
||||
o.* = .{};
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
@@ -127,7 +181,7 @@ fn splitParent(path: []const u8) ParentLeaf {
|
||||
};
|
||||
}
|
||||
|
||||
fn handleOpen(out: []u8, path: []const u8, flags: u32) usize {
|
||||
fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize {
|
||||
var node = filesystem.resolve(path);
|
||||
if (node == null and flags & protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
@@ -141,13 +195,13 @@ fn handleOpen(out: []u8, path: []const u8, flags: u32) usize {
|
||||
filesystem.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return fail(out);
|
||||
open_nodes[index] = .{ .used = true, .node = resolved };
|
||||
open_nodes[index] = .{ .used = true, .node = resolved, .owner = sender };
|
||||
return writeReply(out, .{ .status = 0, .node = index }, &.{});
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = capability;
|
||||
_ = sender;
|
||||
if (!mounted) return fail(out); // storage not up (yet): fail politely, clients retry
|
||||
if (message.len < protocol.request_size) return fail(out);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
@@ -157,7 +211,7 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i
|
||||
filesystem.current_time_epoch = runtime.system.wallClock();
|
||||
|
||||
switch (request.operation) {
|
||||
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags),
|
||||
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags, sender),
|
||||
.read => {
|
||||
const o = openAt(request.node) orelse return fail(out);
|
||||
var buffer: [protocol.maximum_payload]u8 = undefined;
|
||||
@@ -192,10 +246,20 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i
|
||||
},
|
||||
.close => {
|
||||
if (openAt(request.node)) |o| o.used = false;
|
||||
// Durable-on-close: if any block reached the device since the last
|
||||
// flush, commit its cache to stable media now (best-effort). This is
|
||||
// what makes init's shutdown log flush survive a real power-off, and is
|
||||
// the right default for removable media the user may unplug.
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
.mkdir => {
|
||||
const split = splitParent(payload[0..@min(payload.len, request.len)]);
|
||||
const path = payload[0..@min(payload.len, request.len)];
|
||||
if (filesystem.resolve(path) != null) return fail(out); // already exists — no duplicate entries
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return fail(out);
|
||||
if (filesystem.createDirectory(parent, split.leaf) == null) return fail(out);
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
@@ -227,10 +291,6 @@ pub fn main() void {
|
||||
.service = .fat,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -28,8 +28,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
// supervisor reads a clean exit as "meant to stop" — correct for a
|
||||
// placeholder.
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -20,22 +20,47 @@
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const power = runtime.power_protocol;
|
||||
const build_options = @import("build_options");
|
||||
|
||||
/// Where the kernel boot log is persisted on the USB FAT volume — an 8.3 name at
|
||||
/// the mount root (see system/services/log-flush). init writes it at shutdown;
|
||||
/// the log-flush one-shot writes it once at boot.
|
||||
const log_path = "/mnt/usb/DANOS.LOG";
|
||||
/// The system services init brings up at boot, in order, by binary path. This is
|
||||
/// init's policy — the microkernel keeps such choices in user space, not the
|
||||
/// kernel. Drivers are absent on purpose: the device manager owns those. (A
|
||||
/// future init reads this from a manifest under /system/services instead of a
|
||||
/// hardcoded list.)
|
||||
const boot_services = if (build_options.diagnose) [_][]const u8{
|
||||
// The diagnose boot: no display service, so the kernel's on-screen boot
|
||||
// transcript is never suppressed — the timestamped timeline (USB bring-up,
|
||||
// storage, logger) stays readable on real hardware with no serial.
|
||||
"/system/services/input",
|
||||
"/system/services/device-manager",
|
||||
"/system/services/fat",
|
||||
"/system/services/logger",
|
||||
} else [_][]const u8{
|
||||
"/system/services/input",
|
||||
"/system/services/device-manager",
|
||||
"/system/services/fat",
|
||||
"/system/services/display",
|
||||
"/system/services/display-demo",
|
||||
// Last: at shutdown children stop in reverse order, so the logger goes down
|
||||
// FIRST — its final drain still has the fat server (and the whole storage
|
||||
// chain) alive underneath it.
|
||||
"/system/services/logger",
|
||||
};
|
||||
|
||||
/// The system services init brings up at boot, in order. This is init's policy — the
|
||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||
/// on purpose: the device manager owns those. (A future init reads this from a
|
||||
/// manifest under /system/services instead of a hardcoded list.)
|
||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat" };
|
||||
|
||||
var children: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var child_count: usize = 0;
|
||||
/// The live process id of each boot service (0 = not running), indexed by its position
|
||||
/// in `boot_services`, plus how many times init has restarted it. init supervises these:
|
||||
/// it spawns them against `supervision_endpoint` and, on a child's death, restarts it (up
|
||||
/// to `maximum_restarts`) — the reincarnation half of resilience (docs/resilience.md), the
|
||||
/// service-level counterpart to the device manager's driver restarts.
|
||||
var child_ids: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var restart_counts: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var shutting_down = false;
|
||||
var supervision_endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
|
||||
/// that faults immediately on every spawn doesn't respawn forever.
|
||||
const maximum_restarts = 3;
|
||||
|
||||
pub fn main() void {
|
||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||
// mmaps pages from the kernel and carves them with the free list), write into
|
||||
@@ -63,30 +88,22 @@ pub fn main() void {
|
||||
// Bring up the boot services, supervised so init can stop them cleanly.
|
||||
// Best-effort and silent: each service announces its own readiness, and in
|
||||
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
||||
for (boot_services) |service| {
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| {
|
||||
children[child_count] = id;
|
||||
child_count += 1;
|
||||
}
|
||||
for (boot_services, 0..) |service, i| {
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| child_ids[i] = id;
|
||||
}
|
||||
|
||||
// Once the storage stack is up, a one-shot copies the boot log to the USB
|
||||
// volume (/mnt/usb/DANOS.LOG) so it can be read on another machine — the only
|
||||
// way to see it on a headless/real board with no host capturing serial. Fire
|
||||
// and forget: it polls for the mount itself, and is deliberately NOT one of
|
||||
// init's supervised children (a transient one-shot must not be stopped-and-
|
||||
// waited-for during shutdown).
|
||||
_ = runtime.system.spawn("log-flush");
|
||||
|
||||
// Subscribe to power events (retry: the power service registers well after
|
||||
// init starts). Best-effort — without it, a `terminate` signal still
|
||||
// triggers the same shutdown path.
|
||||
subscribePower();
|
||||
|
||||
// A re-arming timer drives the liveness heartbeat: proof PID 1 is alive
|
||||
// (the init test's marker) while the loop stays free to receive signals,
|
||||
// power events, and children's exit notifications.
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
// A re-arming timer drives the liveness heartbeat — proof PID 1 is alive (the
|
||||
// init test's marker) and a -Dserial diagnostic. It is a serial/test-build-only
|
||||
// concern: a flashable (serial-off) image runs a purely event-driven PID 1 that
|
||||
// wakes only for real work (signals, power events, children's exits), never for a
|
||||
// periodic beat. `build_options.serial` is comptime, so the heartbeat — its timer
|
||||
// and the handler below — folds away entirely when serial is off.
|
||||
if (build_options.serial) _ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
|
||||
var receive: [power.message_maximum]u8 = undefined;
|
||||
while (true) {
|
||||
@@ -95,7 +112,7 @@ pub fn main() void {
|
||||
if (signals.has(.terminate)) shutDown();
|
||||
continue;
|
||||
}
|
||||
if (got.isTimer()) {
|
||||
if (build_options.serial and got.isTimer()) {
|
||||
_ = runtime.system.write("/system/services/init: heartbeat\n");
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
continue;
|
||||
@@ -105,11 +122,42 @@ pub fn main() void {
|
||||
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
||||
continue;
|
||||
}
|
||||
// Child-exit notifications and anything else: keep waiting.
|
||||
if (got.isChildExit()) {
|
||||
restartChild(got.childProcessId());
|
||||
continue;
|
||||
}
|
||||
// Anything else: keep waiting.
|
||||
if (got.isNotification()) continue;
|
||||
}
|
||||
}
|
||||
|
||||
/// A supervised boot service died. Find which one and restart it — unless it exited
|
||||
/// cleanly (it chose to stop, e.g. a driver with no hardware) or has hit the crash-loop
|
||||
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
|
||||
/// iron rule 1); init only decides whether to bring it back.
|
||||
fn restartChild(id: u32) void {
|
||||
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
|
||||
for (boot_services, 0..) |service, i| {
|
||||
if (child_ids[i] != id) continue;
|
||||
child_ids[i] = 0;
|
||||
// An unknown reason (the record aged out) is treated as a crash worth restarting.
|
||||
const reason = runtime.process.exitReason(id) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
std.log.info("{s} exited cleanly; not restarting", .{service});
|
||||
return;
|
||||
}
|
||||
restart_counts[i] += 1;
|
||||
if (restart_counts[i] > maximum_restarts) {
|
||||
std.log.info("{s} keeps crashing; giving up after {d} restarts", .{ service, maximum_restarts });
|
||||
return;
|
||||
}
|
||||
std.log.info("{s} died ({s}); restarting ({d}/{d})", .{ service, @tagName(reason), restart_counts[i], maximum_restarts });
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |new_id| child_ids[i] = new_id;
|
||||
return;
|
||||
}
|
||||
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
||||
}
|
||||
|
||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||
/// call's capability) so events arrive as buffered messages here.
|
||||
fn subscribePower() void {
|
||||
@@ -128,40 +176,20 @@ fn subscribePower() void {
|
||||
_ = runtime.ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||
}
|
||||
|
||||
/// Copy the whole kernel log to /mnt/usb/DANOS.LOG (the same file log-flush
|
||||
/// writes at boot), so a poweroff captures the fullest log. Best-effort: if the
|
||||
/// USB volume is not mounted, the open fails and it does nothing. Must run while
|
||||
/// the storage services are still alive (see shutDown).
|
||||
fn flushKernelLog() void {
|
||||
// Truncate on open so this fuller flush replaces the boot-time one cleanly.
|
||||
var file = runtime.fs.open(log_path, .{ .create = true, .truncate = true }) orelse return; // no USB volume
|
||||
defer file.close();
|
||||
var chunk: [4096]u8 = undefined;
|
||||
var offset: usize = 0;
|
||||
while (true) {
|
||||
const got = runtime.system.klogRead(offset, &chunk);
|
||||
if (got == 0) break; // reached the end of the accumulated log
|
||||
if (file.writeAll(chunk[0..got]) == null) break; // storage went away
|
||||
offset += got;
|
||||
}
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, "/system/services/init: flushed log to {s} ({d} bytes)\n", .{ log_path, offset }) catch "");
|
||||
}
|
||||
|
||||
/// The stop sequence: persist the log while storage is still up, then terminate
|
||||
/// each child in reverse spawn order (vfs last — other services may flush through
|
||||
/// it), waiting up to a deadline for each to exit before killing it, then ask the
|
||||
/// power service to enter S5.
|
||||
fn shutDown() void {
|
||||
shutting_down = true; // the stop loop below kills children — those deaths aren't crashes
|
||||
_ = runtime.system.write("/system/services/init: shutting down\n");
|
||||
// Persist the fullest log to the USB volume BEFORE tearing anything down: the
|
||||
// reverse-order stop loop below kills the fat server (children[3]) first, so
|
||||
// /mnt/usb must be written while it is still mounted.
|
||||
flushKernelLog();
|
||||
var i = child_count;
|
||||
// Log persistence is the logger service's job: it is the LAST boot service,
|
||||
// so the reverse-order stop below terminates it first and its final drain
|
||||
// runs while the whole storage chain is still alive.
|
||||
var i = boot_services.len;
|
||||
while (i > 0) {
|
||||
i -= 1;
|
||||
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint);
|
||||
if (child_ids[i] != 0) runtime.process.stop(child_ids[i], 2000, supervision_endpoint);
|
||||
}
|
||||
if (runtime.ipc.lookup(.power)) |h| {
|
||||
const request = power.Shutdown{};
|
||||
@@ -171,8 +199,3 @@ fn shutDown() void {
|
||||
// If S5 did not take, init has nothing left to do but idle.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -10,17 +10,38 @@
|
||||
//! keyboard and mouse drivers publish their own synthetic streams today; swapping in
|
||||
//! decoded hardware is a follow-up (see docs/input.md).
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const input = runtime.input;
|
||||
const system = runtime.system;
|
||||
|
||||
pub fn main() void {
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
var source = input.connectSource() orelse {
|
||||
_ = system.write("input-source: input service unavailable\n");
|
||||
return;
|
||||
};
|
||||
_ = system.write("input-source: publishing synthetic input events\n");
|
||||
|
||||
// "mouse" mode publishes a steady stream of pure motion (dx=dy=+1), for driving a
|
||||
// cursor (the `display-cursor` test). The default "rotate" mode cycles all device
|
||||
// classes to exercise the service's per-device routing (the `input` test).
|
||||
const mode = init.arguments.get(1) orelse "rotate";
|
||||
if (std.mem.eql(u8, mode, "mouse")) {
|
||||
_ = system.write("input-source: publishing synthetic mouse motion\n");
|
||||
while (true) {
|
||||
_ = source.publishMouseEvent(.{
|
||||
.kind = @intFromEnum(input.MouseEventKind.motion),
|
||||
.button = 0,
|
||||
.dx = 1,
|
||||
.dy = 1,
|
||||
.scroll_x = 0,
|
||||
.scroll_y = 0,
|
||||
.buttons = 0,
|
||||
});
|
||||
system.sleep(20); // ~50 events/sec: moves the cursor briskly
|
||||
}
|
||||
}
|
||||
|
||||
_ = system.write("input-source: publishing synthetic input events\n");
|
||||
var step: usize = 0;
|
||||
while (true) : (step +%= 1) {
|
||||
// Rotate across the device classes so every publish path (and the service's
|
||||
@@ -33,8 +54,3 @@ pub fn main() void {
|
||||
system.sleep(200);
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -49,8 +49,3 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -138,8 +138,3 @@ pub fn main() void {
|
||||
reply_len = handle(receive[0..got.len], got, &reply_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -1,67 +0,0 @@
|
||||
//! system/services/log-flush — a one-shot that copies the kernel's in-memory
|
||||
//! diagnostic log to a file on the mounted USB FAT volume, so the boot log
|
||||
//! survives to be read on another machine. On a headless or real board there is
|
||||
//! no host capturing serial, so without this the log is lost at power-off; this
|
||||
//! is the on-disk equivalent of QEMU's `-serial file:`.
|
||||
//!
|
||||
//! It reads the whole kernel log back through `klog_read` (the RAM sink in
|
||||
//! system/kernel/log.zig) and writes it to /mnt/usb/DANOS.LOG. The name is 8.3
|
||||
//! (FAT short-name rule: base <= 8, extension <= 3) and lives at the mount root
|
||||
//! (there is no mkdir on the FAT path yet). init spawns this once the boot
|
||||
//! services are up; init itself repeats the flush at shutdown for a fuller log.
|
||||
//!
|
||||
//! If no USB volume is mounted — no stick, or the initial-ramdisk sweep that
|
||||
//! spawns every bundled binary bare with no VFS — it waits briefly, then exits
|
||||
//! silently, deranging no other test's output.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const fs = runtime.fs;
|
||||
|
||||
const log_path = "/mnt/usb/DANOS.LOG";
|
||||
|
||||
/// Copy the whole kernel log to the open file, looping klog_read -> write until
|
||||
/// the log is exhausted. Returns the number of bytes written.
|
||||
fn drainKernelLog(file: *fs.File) usize {
|
||||
var chunk: [4096]u8 = undefined;
|
||||
var offset: usize = 0;
|
||||
while (true) {
|
||||
const got = runtime.system.klogRead(offset, &chunk);
|
||||
if (got == 0) break; // reached the end of the accumulated log
|
||||
if (file.writeAll(chunk[0..got]) == null) break; // storage went away
|
||||
offset += got;
|
||||
}
|
||||
return offset;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// Wait for the fat server to mount /mnt/usb (it must bring up the whole USB
|
||||
// storage chain first, so it races us at boot). Bounded: if the mount never
|
||||
// appears — no volume, or the no-VFS ramdisk sweep — give up silently.
|
||||
var ready = false;
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1400) : (tries += 1) {
|
||||
if (fs.openDirectory("/mnt/usb")) |directory| {
|
||||
var dir = directory;
|
||||
dir.close();
|
||||
ready = true;
|
||||
break;
|
||||
}
|
||||
runtime.system.sleep(50);
|
||||
}
|
||||
if (!ready) return; // /mnt/usb never became available — nothing to persist to
|
||||
|
||||
// Truncate on open: each flush replaces the file, so a shorter log on a later
|
||||
// boot of the same stick leaves no stale tail from a previous, longer one.
|
||||
var file = fs.open(log_path, .{ .create = true, .truncate = true }) orelse return;
|
||||
const written = drainKernelLog(&file);
|
||||
file.close();
|
||||
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, "log-flush: wrote {d} bytes to {s}\n", .{ written, log_path }) catch return);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -0,0 +1,304 @@
|
||||
//! The logger service — the per-process log persister.
|
||||
//!
|
||||
//! Drains the tagged kernel log ring (`klog_read`/`klog_status`) and
|
||||
//! demultiplexes it into **one file per process** on the flash volume:
|
||||
//!
|
||||
//! <base>/<boot-stamp>/<binary-path>.log
|
||||
//! e.g. /mnt/usb/var/log/2026-07-21T101530Z/system/services/fat.log
|
||||
//!
|
||||
//! The boot stamp is the wall-clock time of boot (from klog_status), so one
|
||||
//! boot session is one self-contained directory; the kernel's own records go to
|
||||
//! kernel.log. Records carry the sender's pid and binary path, stamped by the
|
||||
//! kernel — the logger trusts the ring, never the payload.
|
||||
//!
|
||||
//! Storage is best-effort and late: until the FAT volume mounts, the ring
|
||||
//! simply buffers (it holds a full boot many times over), and the first drain
|
||||
//! writes the whole backlog. The storage stack's own records are captured the
|
||||
//! same way — services never write their own log files (the fat service
|
||||
//! logging through itself would rendezvous-deadlock; the ring sidesteps that
|
||||
//! by design).
|
||||
//!
|
||||
//! The logger announces itself ONCE (a periodic status line would feed the
|
||||
//! very stream it drains — self-sustaining churn). Lost records surface as an
|
||||
//! explicit "-- N records lost --" line derived from sequence-number gaps.
|
||||
//!
|
||||
//! Durability: files are opened create-once and kept open across a burst, then
|
||||
//! all closed after a quiet period (~2 s) — each close is the fat server's
|
||||
//! SCSI SYNCHRONIZE CACHE, so data-at-risk is bounded by the last busy burst
|
||||
//! without thrashing the device on every record. `on_terminate` does a final
|
||||
//! drain and closes everything, so an orderly shutdown loses nothing (init
|
||||
//! stops the logger FIRST — reverse boot order — while fat is still up).
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
|
||||
const system = runtime.system;
|
||||
const fs = runtime.fs;
|
||||
|
||||
/// Where log trees live: the FHS path. The kernel VFS routes /var to whatever
|
||||
/// volume the fat server mounted there (today: the /var subtree of the USB
|
||||
/// flash volume) — swapping the persistent medium later touches fat's two
|
||||
/// mount calls, never this constant.
|
||||
const base = "/var/log";
|
||||
|
||||
/// Drain cadence and the quiet period after which files are closed (flushed).
|
||||
const tick_ms = 250;
|
||||
const quiet_close_ticks = 8; // 8 * 250 ms = 2 s
|
||||
|
||||
/// One cached open file per source process path. Sized above the practical
|
||||
/// process count; the fat server's global open-node table (32) is the real
|
||||
/// ceiling, so stay comfortably below it.
|
||||
const maximum_files = 24;
|
||||
|
||||
const CachedFile = struct {
|
||||
used: bool = false,
|
||||
name: [system.maximum_process_name]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
file: fs.File = undefined,
|
||||
};
|
||||
|
||||
var files: [maximum_files]CachedFile = @splat(.{});
|
||||
var endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
/// The drain cursor into the ring's byte stream, and loss accounting.
|
||||
var cursor: u64 = 0;
|
||||
var next_expected_sequence: u64 = 0;
|
||||
|
||||
/// Carry buffer: a record can straddle two klog_read chunks.
|
||||
var carry: [carry_capacity]u8 = undefined;
|
||||
var carry_len: usize = 0;
|
||||
const carry_capacity = 64 + 256 + 64; // header + payload + name, padded generously
|
||||
|
||||
/// The per-boot directory, formatted once storage appears.
|
||||
var boot_directory: [base.len + 1 + 19]u8 = undefined;
|
||||
var boot_directory_len: usize = 0;
|
||||
var storage_ready = false;
|
||||
var announced = false;
|
||||
var ticks_since_record: u32 = 0;
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(64, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
.on_terminate = onTerminate,
|
||||
});
|
||||
}
|
||||
|
||||
fn initialise(harness_endpoint: runtime.ipc.Handle) bool {
|
||||
endpoint = harness_endpoint;
|
||||
const status = system.klogStatus() orelse return false;
|
||||
cursor = status.tail;
|
||||
// Sequence expectations start at the tail record's sequence — discovered on
|
||||
// the first drain; 0 is right for a fresh boot either way.
|
||||
formatBootDirectory(status.boot_unix_seconds);
|
||||
_ = system.timerOnce(endpoint, tick_ms);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The logger serves no protocol; the ping is answered by the harness.
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & runtime.ipc.notify_timer_bit == 0) return;
|
||||
tick();
|
||||
_ = system.timerOnce(endpoint, tick_ms);
|
||||
}
|
||||
|
||||
fn onTerminate() void {
|
||||
// The completeness receipt FIRST: this record enters the ring before the
|
||||
// final drain, so the drain carries it into logger.log — a directory whose
|
||||
// logger.log ends with this marker is complete through shutdown; one that
|
||||
// doesn't was cut early and may be missing tails.
|
||||
_ = system.write("logger: shutting down; final flush\n");
|
||||
drain();
|
||||
closeAll();
|
||||
// Serial-only epilogue (after the drain, so it reaches no file — by design).
|
||||
var line: [96]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "logger: flushed through sequence {d}\n", .{next_expected_sequence}) catch return);
|
||||
}
|
||||
|
||||
fn tick() void {
|
||||
if (!storage_ready) {
|
||||
// makePath doubles as the readiness probe: while /var is unmounted the
|
||||
// resolve fails fast (no storage round trip) and the ring buffers; the
|
||||
// first success creates the whole per-boot tree.
|
||||
if (!fs.makePath(boot_directory[0..boot_directory_len])) return;
|
||||
storage_ready = true;
|
||||
if (!announced) {
|
||||
announced = true; // once — a periodic line would feed the stream we drain
|
||||
var line: [128]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "logger: logging to {s}\n", .{boot_directory[0..boot_directory_len]}) catch "");
|
||||
}
|
||||
}
|
||||
drain();
|
||||
// Quiet-period close: one device cache flush per burst.
|
||||
ticks_since_record += 1;
|
||||
if (ticks_since_record == quiet_close_ticks) closeAll();
|
||||
}
|
||||
|
||||
fn drain() void {
|
||||
if (!storage_ready) return;
|
||||
var chunk: [4096]u8 = undefined;
|
||||
while (true) {
|
||||
@memcpy(chunk[0..carry_len], carry[0..carry_len]);
|
||||
const got = system.klogRead(cursor, chunk[carry_len..]) orelse {
|
||||
// Cursor overwritten: re-sync to the ring tail; the sequence gap is
|
||||
// reported by the next record's header.
|
||||
const status = system.klogStatus() orelse return;
|
||||
cursor = status.tail;
|
||||
carry_len = 0;
|
||||
continue;
|
||||
};
|
||||
if (got == 0) return; // caught up (any partial record stays carried)
|
||||
cursor += got;
|
||||
consume(chunk[0 .. carry_len + got]);
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse whole records out of `bytes`; keep any trailing partial in `carry`.
|
||||
fn consume(bytes: []u8) void {
|
||||
const header_size = system.klog_record_header_size;
|
||||
var offset: usize = 0;
|
||||
while (bytes.len - offset >= header_size) {
|
||||
const header = std.mem.bytesToValue(system.KlogRecordHeader, bytes[offset..][0..32]);
|
||||
if (header.magic != system.klog_record_magic) {
|
||||
// Corrupt frame — should not happen; drop the carry and re-sync.
|
||||
carry_len = 0;
|
||||
const status = system.klogStatus() orelse return;
|
||||
cursor = status.head;
|
||||
return;
|
||||
}
|
||||
const record_len = recordLength(header);
|
||||
if (bytes.len - offset < record_len) break; // partial — carry it
|
||||
const name = bytes[offset + header_size ..][0..header.name_len];
|
||||
const message = bytes[offset + header_size + header.name_len ..][0..header.message_len];
|
||||
deliver(header, name, message);
|
||||
offset += record_len;
|
||||
}
|
||||
const rest = bytes.len - offset;
|
||||
if (rest > carry_capacity) {
|
||||
carry_len = 0; // cannot happen with sane frames; drop rather than overflow
|
||||
return;
|
||||
}
|
||||
@memcpy(carry[0..rest], bytes[offset..]);
|
||||
carry_len = rest;
|
||||
}
|
||||
|
||||
fn deliver(header: system.KlogRecordHeader, name: []const u8, message: []const u8) void {
|
||||
ticks_since_record = 0;
|
||||
const file = fileFor(if (header.pid == 0 or name.len == 0) "kernel" else name) orelse return;
|
||||
|
||||
if (header.sequence != next_expected_sequence and next_expected_sequence != 0) {
|
||||
var gap_line: [64]u8 = undefined;
|
||||
const lost = header.sequence - next_expected_sequence;
|
||||
if (std.fmt.bufPrint(&gap_line, "-- {d} records lost --\n", .{lost})) |line| {
|
||||
_ = file.writeAll(line);
|
||||
} else |_| {}
|
||||
}
|
||||
next_expected_sequence = header.sequence + 1;
|
||||
|
||||
// [+ssssss.mmm] level: payload
|
||||
var stamp: [48]u8 = undefined;
|
||||
const seconds = header.timestamp_ns / 1_000_000_000;
|
||||
const millis = (header.timestamp_ns / 1_000_000) % 1000;
|
||||
const level: []const u8 = switch (header.level) {
|
||||
.err => "error: ",
|
||||
.warn => "warning: ",
|
||||
.debug => "debug: ",
|
||||
.info, .raw => "",
|
||||
};
|
||||
if (std.fmt.bufPrint(&stamp, "[{d:>6}.{d:0>3}] {s}", .{ seconds, millis, level })) |prefix| {
|
||||
_ = file.writeAll(prefix);
|
||||
} else |_| {}
|
||||
_ = file.writeAll(message);
|
||||
if (header.flags & system.klog_flag_truncated != 0) _ = file.writeAll("~");
|
||||
_ = file.writeAll("\n");
|
||||
}
|
||||
|
||||
/// The cached (or freshly opened) file for a source name. The file path is the
|
||||
/// binary path with its leading '/' stripped, ".log" appended, under the
|
||||
/// per-boot directory; parents are created on first use.
|
||||
fn fileFor(name: []const u8) ?*fs.File {
|
||||
for (&files) |*cached| {
|
||||
if (cached.used and std.mem.eql(u8, cached.name[0..cached.name_len], name)) return &cached.file;
|
||||
}
|
||||
var slot: ?*CachedFile = null;
|
||||
for (&files) |*cached| {
|
||||
if (!cached.used) {
|
||||
slot = cached;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const cached = slot orelse evictOne() orelse return null;
|
||||
|
||||
var path: [base.len + 1 + 19 + 1 + system.maximum_process_name + 4]u8 = undefined;
|
||||
const relative = if (name.len != 0 and name[0] == '/') name[1..] else name;
|
||||
const full = std.fmt.bufPrint(&path, "{s}/{s}.log", .{ boot_directory[0..boot_directory_len], relative }) catch return null;
|
||||
|
||||
// Parent directories: everything up to the final slash.
|
||||
if (std.mem.lastIndexOfScalar(u8, full, '/')) |last| {
|
||||
if (!fs.makePath(full[0..last])) return null;
|
||||
}
|
||||
var file = fs.open(full, .{ .create = true }) orelse return null;
|
||||
// Append: land after whatever an earlier open of this boot wrote.
|
||||
if (file.attributes()) |attributes| file.seekTo(attributes.size);
|
||||
|
||||
cached.* = .{ .used = true, .file = file };
|
||||
@memcpy(cached.name[0..name.len], name);
|
||||
cached.name_len = name.len;
|
||||
return &cached.file;
|
||||
}
|
||||
|
||||
fn evictOne() ?*CachedFile {
|
||||
// All slots busy: close the first (oldest-created) and reuse it. Simple and
|
||||
// rare — the process count sits well under the cache size.
|
||||
for (&files) |*cached| {
|
||||
if (cached.used) {
|
||||
cached.file.close();
|
||||
cached.used = false;
|
||||
return cached;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn closeAll() void {
|
||||
for (&files) |*cached| {
|
||||
if (cached.used) {
|
||||
cached.file.close();
|
||||
cached.used = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn recordLength(header: system.KlogRecordHeader) usize {
|
||||
return std.mem.alignForward(usize, system.klog_record_header_size + header.name_len + header.message_len, system.klog_record_alignment);
|
||||
}
|
||||
|
||||
/// Format the per-boot directory "<base>/YYYY-MM-DDTHHMMSSZ" from the boot
|
||||
/// wall-clock anchor. No colons — FAT names cannot carry them. A dead RTC
|
||||
/// (anchor 0) yields the 1970 epoch directory, which is still a valid,
|
||||
/// distinct-per-boot-rarely name and better than refusing to log.
|
||||
fn formatBootDirectory(boot_unix_seconds: u64) void {
|
||||
const epoch_seconds = std.time.epoch.EpochSeconds{ .secs = boot_unix_seconds };
|
||||
const year_day = epoch_seconds.getEpochDay().calculateYearDay();
|
||||
const month_day = year_day.calculateMonthDay();
|
||||
const day_seconds = epoch_seconds.getDaySeconds();
|
||||
const written = std.fmt.bufPrint(&boot_directory, "{s}/{d:0>4}-{d:0>2}-{d:0>2}T{d:0>2}{d:0>2}{d:0>2}Z", .{
|
||||
base,
|
||||
year_day.year,
|
||||
month_day.month.numeric(),
|
||||
@as(u32, month_day.day_index) + 1,
|
||||
day_seconds.getHoursIntoDay(),
|
||||
day_seconds.getMinutesIntoHour(),
|
||||
day_seconds.getSecondsIntoMinute(),
|
||||
}) catch return;
|
||||
boot_directory_len = written.len;
|
||||
}
|
||||
@@ -147,8 +147,8 @@ pub fn main(init: runtime.process.Init) void {
|
||||
const spinner = runtime.system.spawnSupervised("process-test", &.{"spinner"}, endpoint) orelse fail("spawn spinner");
|
||||
|
||||
runtime.system.sleep(100); // let the sleeper block and the spinner get a core
|
||||
if (!listed(sleeper, "process-test")) fail("sleeper not in process_enumerate");
|
||||
if (!listed(spinner, "process-test")) fail("spinner not in process_enumerate");
|
||||
if (!listed(sleeper, "/system/tests/process-test")) fail("sleeper not in process_enumerate");
|
||||
if (!listed(spinner, "/system/tests/process-test")) fail("spinner not in process_enumerate");
|
||||
|
||||
// Kills that must be refused: a kernel task (id 0), and an id that was never
|
||||
// issued — both -ESRCH. (-EPERM needs a second supervisor; the kernel-level
|
||||
@@ -167,8 +167,8 @@ pub fn main(init: runtime.process.Init) void {
|
||||
if (!runtime.system.kill(spinner)) fail("kill spinner");
|
||||
if (awaitChildExit(endpoint) != spinner) fail("spinner exit notification");
|
||||
|
||||
if (listed(sleeper, "process-test")) fail("sleeper still listed after kill");
|
||||
if (listed(spinner, "process-test")) fail("spinner still listed after kill");
|
||||
if (listed(sleeper, "/system/tests/process-test")) fail("sleeper still listed after kill");
|
||||
if (listed(spinner, "/system/tests/process-test")) fail("spinner still listed after kill");
|
||||
|
||||
// M17.2: both children were killed by us, and the reason says so — the whole
|
||||
// restart-policy input, read through the runtime like a real supervisor would.
|
||||
@@ -178,8 +178,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
|
||||
_ = runtime.system.write("process-test: ok\n");
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
//! system/services/shared-memory-client — the creating half of the shared-memory test (docs/display-v2.md V2).
|
||||
//! It `shared_memory_create`s a shared region, writes a known pattern into it, and hands the region's
|
||||
//! capability to `shared-memory-server` as an `ipc_call` send_cap. The server maps that capability and
|
||||
//! confirms the pattern is visible — proving cross-process shared memory over the extended
|
||||
//! capability-passing path.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const system = runtime.system;
|
||||
const shared_memory = runtime.shared_memory;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const pattern_len = 4096;
|
||||
|
||||
/// The pattern the server checks — must match shared-memory-server.zig.
|
||||
fn expected(i: usize) u8 {
|
||||
return @truncate(i *% 7 +% 3);
|
||||
}
|
||||
|
||||
fn lookupServer() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.shared_memory_test)) |h| return h;
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const region = shared_memory.create(pattern_len) orelse {
|
||||
_ = system.write("shared-memory: create failed\n");
|
||||
return;
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < pattern_len) : (i += 1) region.ptr[i] = expected(i);
|
||||
|
||||
const server = lookupServer() orelse {
|
||||
_ = system.write("shared-memory: no server\n");
|
||||
return;
|
||||
};
|
||||
// A non-empty message (so it reaches on_message, not the ping path), carrying the shared-memory
|
||||
// region's capability. The reply is empty; we just need the round trip.
|
||||
var reply: [64]u8 = undefined;
|
||||
_ = ipc.callCap(server, "shared-memory", &reply, region.handle) catch {
|
||||
_ = system.write("shared-memory: call failed\n");
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
//! system/services/shared-memory-server — the receiving half of the shared-memory test (docs/display-v2.md V2).
|
||||
//! It registers under `ServiceId.shared_memory_test`; when `shared-memory-client` calls it carrying a
|
||||
//! shared-memory capability, it `shared_memory_map`s that capability and checks the client's pattern
|
||||
//! is visible through the mapping — proving the two processes share the same physical pages
|
||||
//! (not a copy). On success it prints `shared-memory: shared 4096 bytes ok`, the test's marker.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const system = runtime.system;
|
||||
const shared_memory = runtime.shared_memory;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const pattern_len = 4096;
|
||||
|
||||
/// The pattern the client writes — must match shared-memory-client.zig.
|
||||
fn expected(i: usize) u8 {
|
||||
return @truncate(i *% 7 +% 3);
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
const cap = capability orelse {
|
||||
_ = system.write("shared-memory: shared FAILED (no capability)\n");
|
||||
return 0;
|
||||
};
|
||||
const ptr = shared_memory.map(cap) orelse {
|
||||
_ = system.write("shared-memory: shared FAILED (map)\n");
|
||||
return 0;
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < pattern_len) : (i += 1) {
|
||||
if (ptr[i] != expected(i)) {
|
||||
_ = system.write("shared-memory: shared FAILED (mismatch)\n");
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
_ = system.write("shared-memory: shared 4096 bytes ok\n");
|
||||
return 0; // empty reply — the client only needs the round trip to unblock
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(64, .{ .service = .shared_memory_test, .on_message = onMessage });
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user