Compare commits
96
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
203528c8a7 | ||
|
|
bf0c3fd3e0 | ||
|
|
3af0110483 | ||
|
|
757c6f14c3 | ||
|
|
52d6e372fd | ||
|
|
f023f1cfd6 | ||
|
|
b9cec7d1be | ||
|
|
6271278d4d | ||
|
|
37326c7664 | ||
|
|
23bcd77c58 | ||
|
|
dded46726b | ||
|
|
60f32ee9ff | ||
|
|
3e69712b97 | ||
|
|
11f567ee20 | ||
|
|
3a155cdc7d | ||
|
|
5b874fc756 | ||
|
|
7d540c4b2f | ||
|
|
ac2d102878 | ||
|
|
3c9475e33a | ||
|
|
44122bd44d | ||
|
|
9ef22d6e55 | ||
|
|
07c901c18c | ||
|
|
25cc4d610e | ||
|
|
f90dc6c121 | ||
|
|
1d1234c963 | ||
|
|
a7f0c1a450 | ||
|
|
f86f2987d5 | ||
|
|
794a8b5782 | ||
|
|
ea470afe84 | ||
|
|
d27377ab1e | ||
|
|
cacbacd76b | ||
|
|
e63d6ef5ef | ||
|
|
c8191570e1 | ||
|
|
2bc2a0d70d | ||
|
|
1882161cb4 | ||
|
|
08e139ebba | ||
|
|
b09a62bc36 | ||
|
|
daca0d9216 | ||
|
|
e78810195d | ||
|
|
e854f65623 | ||
|
|
52df2ba6f6 | ||
|
|
a081f69def | ||
|
|
f9dea7790a | ||
|
|
7e4071a065 | ||
|
|
6e261ccda1 | ||
|
|
ded555961c | ||
|
|
802d51ba74 | ||
|
|
570f6f545c | ||
|
|
da5f404041 | ||
|
|
ab732dc455 | ||
|
|
4ea4a040d2 | ||
|
|
0b10a637ae | ||
|
|
10956c6660 | ||
|
|
18408b0666 | ||
|
|
27f87cb5ba | ||
|
|
7ee6033fa4 | ||
|
|
e52ae242fc | ||
|
|
9da1a9899c | ||
|
|
92b03b8d07 | ||
|
|
983b4ed05a | ||
|
|
0ac07dadc9 | ||
|
|
4091ea6912 | ||
|
|
d8cf533b73 | ||
|
|
1638845a4b | ||
|
|
7082699f5f | ||
|
|
f0611ef8ac | ||
|
|
eb6e8edafe | ||
|
|
59ba95a315 | ||
|
|
f587e7e05e | ||
|
|
b541921218 | ||
|
|
a91365b3d9 | ||
|
|
ffa45edc8b | ||
|
|
d446ddd2ed | ||
|
|
0a4388c3bc | ||
|
|
0045fd87ba | ||
|
|
ad0fd52cb8 | ||
|
|
65a44a5568 | ||
|
|
e186858315 | ||
|
|
7c5645fe48 | ||
|
|
127ea2dad9 | ||
|
|
30d6ea622a | ||
|
|
d0c1b3e45b | ||
|
|
f480c5d790 | ||
|
|
9f18d8340e | ||
|
|
0a84e52bf8 | ||
|
|
1cf9985da6 | ||
|
|
2d0858caf6 | ||
|
|
ec6e888076 | ||
|
|
4f02f75602 | ||
|
|
16618d2cdc | ||
|
|
15107f54be | ||
|
|
23f915c593 | ||
|
|
981a4af7e0 | ||
|
|
5ab7263c9c | ||
|
|
7f415e724f | ||
|
|
cf140eb772 |
+2
-1
@@ -6,4 +6,5 @@ zig-out/
|
||||
.idea/
|
||||
|
||||
.claude/
|
||||
.github/
|
||||
.github/
|
||||
/var/log/
|
||||
|
||||
@@ -52,7 +52,8 @@ zig build
|
||||
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||
`zig-out/system/drivers/`, the test fixtures under `zig-out/test/system/services/`,
|
||||
and the initial-ramdisk at `zig-out/boot/`.
|
||||
|
||||
## Release media
|
||||
|
||||
@@ -62,7 +63,7 @@ zig build release-x86-64
|
||||
|
||||
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
|
||||
USB stick (balenaEtcher, dd) or burned to optical media — see
|
||||
[docs/release-iso.md](docs/release-iso.md). `zig build check-iso-image`
|
||||
[docs/release-iso.md](docs/os-development/release-iso.md). `zig build check-iso-image`
|
||||
validates it without booting.
|
||||
|
||||
## Run
|
||||
|
||||
+260
-49
@@ -2,6 +2,7 @@ const std = @import("std");
|
||||
const uefi = std.os.uefi;
|
||||
const elf = std.elf;
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||
@@ -15,11 +16,12 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
/// The kernel image: /system/kernel.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||
|
||||
/// The init program: /system/services/init.
|
||||
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||
|
||||
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||
/// The user binaries: everything under /system except the kernel itself, plus
|
||||
/// the test fixtures under /test. The loader walks both trees and packs them
|
||||
/// into the in-RAM initial_ramdisk image — the volume's file structure is the
|
||||
/// single source of truth (no packed image artifact on disk).
|
||||
const system_directory_name = std.unicode.utf8ToUtf16LeStringLiteral("system");
|
||||
const test_directory_name = std.unicode.utf8ToUtf16LeStringLiteral("test");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
@@ -64,20 +66,14 @@ fn boot() !noreturn {
|
||||
|
||||
const entry = try loadKernel(bs, &boot_information);
|
||||
|
||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_information) catch |err| {
|
||||
log("EFI: no /system/services/init (");
|
||||
// Best effort: a volume without a /system tree of user binaries still boots
|
||||
// (kernel-only). The tree — init included — becomes the initial_ramdisk.
|
||||
loadSystemTree(bs, &boot_information) catch |err| {
|
||||
log("EFI: no /system binaries (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||
log("EFI: no initial_ramdisk (");
|
||||
logBytes(@errorName(err));
|
||||
log(")\r\n");
|
||||
};
|
||||
|
||||
// Build the page tables the kernel starts life on: identity + a physmap of
|
||||
// low RAM, plus the higher-half kernel image once it links high. Allocated
|
||||
// now, while boot services (and the memory map) are still stable — nothing
|
||||
@@ -97,7 +93,7 @@ fn boot() !noreturn {
|
||||
}
|
||||
|
||||
/// A display resolution in pixels.
|
||||
const Resolution = struct { width: u32, height: u32 };
|
||||
const Resolution = struct { width: u32, height: u32, refresh_hz: u32 };
|
||||
|
||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||
/// and read the resulting graphics mode into our own framebuffer description.
|
||||
@@ -128,6 +124,10 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||
// Each pixel is 32 bits, so the byte pitch is 4 * pixels-per-row.
|
||||
.pitch = info.pixels_per_scan_line * 4,
|
||||
.format = try pixelFormat(info.pixel_format),
|
||||
// The refresh rate rides the EDID preferred timing. If the firmware kept a
|
||||
// non-native mode it may not describe that mode exactly — but it is the panel's
|
||||
// own clock, a far better frame-clock seed than a hardcoded 60 Hz.
|
||||
.refresh_hz = if (native) |n| n.refresh_hz else 0,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -176,10 +176,12 @@ fn nativeResolution(bs: *uefi.tables.BootServices, handles: []uefi.Handle) ?Reso
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Parse the native resolution from a raw EDID block. The first Detailed Timing
|
||||
/// Descriptor (at byte 54) is the preferred — i.e. native — mode by convention;
|
||||
/// its active pixel counts are split across low bytes and the high nibbles of
|
||||
/// later bytes.
|
||||
/// Parse the native resolution and refresh rate from a raw EDID block. The first
|
||||
/// Detailed Timing Descriptor (at byte 54) is the preferred — i.e. native — mode by
|
||||
/// convention; its active pixel counts are split across low bytes and the high nibbles
|
||||
/// of later bytes. The refresh rate is derived, not stored: the descriptor carries the
|
||||
/// pixel clock (10 kHz units) and the active+blanking extents, and
|
||||
/// refresh = clock / (horizontal total × vertical total).
|
||||
fn edidNative(edid: []const u8) ?Resolution {
|
||||
if (edid.len < 128) return null;
|
||||
// Every EDID begins with this fixed 8-byte header.
|
||||
@@ -193,7 +195,13 @@ fn edidNative(edid: []const u8) ?Resolution {
|
||||
const w = @as(u32, dtd[2]) | (@as(u32, dtd[4] & 0xf0) << 4);
|
||||
const h = @as(u32, dtd[5]) | (@as(u32, dtd[7] & 0xf0) << 4);
|
||||
if (w == 0 or h == 0) return null;
|
||||
return .{ .width = w, .height = h };
|
||||
|
||||
const clock_hz = (@as(u64, dtd[0]) | (@as(u64, dtd[1]) << 8)) * 10_000;
|
||||
const h_blank = @as(u64, dtd[3]) | (@as(u64, dtd[4] & 0x0f) << 8);
|
||||
const v_blank = @as(u64, dtd[6]) | (@as(u64, dtd[7] & 0x0f) << 8);
|
||||
const total = (@as(u64, w) + h_blank) * (@as(u64, h) + v_blank);
|
||||
const refresh: u32 = if (total == 0) 0 else @intCast((clock_hz + total / 2) / total);
|
||||
return .{ .width = w, .height = h, .refresh_hz = refresh };
|
||||
}
|
||||
|
||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||
@@ -357,11 +365,37 @@ fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) nor
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Read a whole file off the boot volume into a pool buffer that outlives the
|
||||
/// loader. The buffer is deliberately NOT freed: it's LoaderData, which the
|
||||
/// memory-map conversion classifies as reserved, so the kernel identity-maps it
|
||||
/// and reads from there. Returns the buffer (pointer + length).
|
||||
fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
// --- the /system and /test trees -> initial_ramdisk --------------------------
|
||||
|
||||
/// Cap on bundled binaries. Generous: the tree carries ~30 today.
|
||||
const maximum_bundled = 64;
|
||||
|
||||
/// How deep the walk goes below a tree root ("/system/services/x" is depth 1,
|
||||
/// "/test/system/services/x" is depth 2).
|
||||
const maximum_tree_depth = 3;
|
||||
|
||||
/// One binary discovered under a walked tree: its FHS path (UTF-8,
|
||||
/// '/'-separated, NUL-free) and its contents in a transient pool buffer.
|
||||
const Bundled = struct {
|
||||
path: [initial_ramdisk.maximum_name]u8,
|
||||
path_len: usize,
|
||||
data: []align(8) u8,
|
||||
};
|
||||
|
||||
/// Gather the boot volume's user binaries into an in-RAM v2 initial_ramdisk
|
||||
/// image, entries named by full FHS path — the volume's file structure is the
|
||||
/// single source of truth (no packed ramdisk artifact; init travels in the
|
||||
/// table like everything else).
|
||||
///
|
||||
/// Two strategies, most portable first:
|
||||
/// 1. /system/manifest (written by the build): each listed path is opened BY
|
||||
/// NAME — the case-insensitive lookup every firmware FAT driver gets
|
||||
/// right, and the only file access the pre-tree loader ever used.
|
||||
/// 2. No manifest: ENUMERATE the /system and /test trees. Portable in
|
||||
/// principle, but firmware differs in what names enumeration returns
|
||||
/// (bare 8.3 entries come back uppercase on some drivers), so this is
|
||||
/// the fallback for hand-assembled sticks, not the primary path.
|
||||
fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
@@ -371,40 +405,217 @@ fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
const root = try fs.openVolume();
|
||||
defer _ = root.close() catch {};
|
||||
|
||||
const file = try root.open(name, .read, .{});
|
||||
defer _ = file.close() catch {};
|
||||
// Unconditional breadcrumb (con_out, independent of -Dserial): this phase
|
||||
// is where a slow firmware stalls, and a silent black screen here already
|
||||
// cost a real-hardware debugging session.
|
||||
log("EFI: loading the system...\r\n");
|
||||
|
||||
// The capsule (boot\system.img) first: one open + one sequential read is
|
||||
// the only firmware file I/O shape that is fast everywhere. It is already
|
||||
// the kernel's wire format — hand it over as-is.
|
||||
if (loadCapsule(bs, root, boot_information)) {
|
||||
log("EFI: system image loaded, starting the kernel\r\n");
|
||||
return;
|
||||
}
|
||||
|
||||
var list: [maximum_bundled]Bundled = undefined;
|
||||
var count: usize = 0;
|
||||
|
||||
loadByManifest(bs, root, &list, &count) catch {
|
||||
count = 0; // a torn manifest read leaves partial entries; start over
|
||||
};
|
||||
if (count == 0) {
|
||||
const system_directory = try root.open(system_directory_name, .read, .{});
|
||||
defer _ = system_directory.close() catch {};
|
||||
try walkDirectory(bs, system_directory, "/system", 0, &list, &count);
|
||||
// The /test tree is optional: a stick without fixtures still boots.
|
||||
if (root.open(test_directory_name, .read, .{})) |test_directory| {
|
||||
defer _ = test_directory.close() catch {};
|
||||
try walkDirectory(bs, test_directory, "/test", 0, &list, &count);
|
||||
} else |_| {}
|
||||
}
|
||||
if (count == 0) return error.NoBinaries;
|
||||
|
||||
// Assemble the v2 image: header, entry table, then the blobs.
|
||||
const table_end = @sizeOf(initial_ramdisk.Header) + count * @sizeOf(initial_ramdisk.Entry);
|
||||
var total: usize = table_end;
|
||||
for (list[0..count]) |e| total += e.data.len;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, total); // survives the handoff
|
||||
std.mem.bytesAsValue(initial_ramdisk.Header, image[0..@sizeOf(initial_ramdisk.Header)]).* = .{
|
||||
.magic = initial_ramdisk.magic,
|
||||
.count = @intCast(count),
|
||||
};
|
||||
var offset: usize = table_end;
|
||||
for (list[0..count], 0..) |e, i| {
|
||||
var record = initial_ramdisk.Entry{ .name = @splat(0), .offset = offset, .len = e.data.len };
|
||||
@memcpy(record.name[0..e.path_len], e.path[0..e.path_len]);
|
||||
const slot = image[@sizeOf(initial_ramdisk.Header) + i * @sizeOf(initial_ramdisk.Entry) ..][0..@sizeOf(initial_ramdisk.Entry)];
|
||||
std.mem.bytesAsValue(initial_ramdisk.Entry, slot).* = record;
|
||||
@memcpy(image[offset..][0..e.data.len], e.data);
|
||||
offset += e.data.len;
|
||||
_ = bs.freePool(e.data.ptr) catch {};
|
||||
}
|
||||
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = total;
|
||||
log("EFI: boot tree loaded, starting the kernel\r\n");
|
||||
}
|
||||
|
||||
/// The boot capsule: the bundled binaries as one v2 initial_ramdisk image.
|
||||
const capsule_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\system.img");
|
||||
|
||||
/// Load boot\system.img whole and hand it to the kernel unmodified — it is
|
||||
/// already the initial_ramdisk wire format. Returns false (capsule absent or
|
||||
/// unreadable or wrong magic) to let the caller fall back to per-file loading.
|
||||
fn loadCapsule(bs: *uefi.tables.BootServices, root: *uefi.protocol.File, boot_information: *BootInformation) bool {
|
||||
const file = root.open(capsule_file_name, .read, .{}) catch return false;
|
||||
defer _ = file.close() catch {};
|
||||
const image = readWholeFile(bs, file) catch return false;
|
||||
if (image.len < @sizeOf(initial_ramdisk.Header) or
|
||||
std.mem.bytesToValue(initial_ramdisk.Header, image[0..@sizeOf(initial_ramdisk.Header)]).magic != initial_ramdisk.magic)
|
||||
{
|
||||
_ = bs.freePool(image.ptr) catch {};
|
||||
return false;
|
||||
}
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The manifest path, and a scratch limit for its UTF-16 conversion.
|
||||
const manifest_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\manifest");
|
||||
|
||||
/// Load every binary the manifest lists, opening each path by name from the
|
||||
/// volume root. A listed-but-unopenable file is skipped (the kernel reports the
|
||||
/// absence); a missing manifest errors so the caller falls back to the walk.
|
||||
fn loadByManifest(bs: *uefi.tables.BootServices, root: *uefi.protocol.File, list: *[maximum_bundled]Bundled, count: *usize) !void {
|
||||
const manifest_handle = try root.open(manifest_file_name, .read, .{});
|
||||
var manifest_open = true;
|
||||
defer if (manifest_open) {
|
||||
_ = manifest_handle.close() catch {};
|
||||
};
|
||||
const manifest = try readWholeFile(bs, manifest_handle);
|
||||
_ = manifest_handle.close() catch {};
|
||||
manifest_open = false;
|
||||
defer _ = bs.freePool(manifest.ptr) catch {};
|
||||
|
||||
var lines = std.mem.tokenizeAny(u8, manifest, "\r\n");
|
||||
while (lines.next()) |line| {
|
||||
if (line.len < 2 or line[0] != '/') continue;
|
||||
if (line.len >= initial_ramdisk.maximum_name) continue;
|
||||
if (count.* == maximum_bundled) return;
|
||||
|
||||
// "/system/services/init" -> UTF-16 "system\services\init".
|
||||
var name16: [initial_ramdisk.maximum_name]u16 = undefined;
|
||||
var i: usize = 0;
|
||||
for (line[1..]) |c| {
|
||||
name16[i] = if (c == '/') '\\' else c;
|
||||
i += 1;
|
||||
}
|
||||
name16[i] = 0;
|
||||
|
||||
const file = root.open(@ptrCast(name16[0..i :0]), .read, .{}) catch continue;
|
||||
defer _ = file.close() catch {};
|
||||
const data = readWholeFile(bs, file) catch continue;
|
||||
|
||||
var entry: *Bundled = &list[count.*];
|
||||
@memcpy(entry.path[0..line.len], line);
|
||||
entry.path_len = line.len;
|
||||
entry.data = data;
|
||||
count.* += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Recursively collect the regular files below `directory` into `list`. Top-level
|
||||
/// files (depth 0) are skipped: the only one is /system/kernel, which loadKernel
|
||||
/// has already consumed and which is not a spawnable user binary (/test has no
|
||||
/// top-level files, so the skip is a no-op there).
|
||||
fn walkDirectory(
|
||||
bs: *uefi.tables.BootServices,
|
||||
directory: *uefi.protocol.File,
|
||||
prefix: []const u8,
|
||||
depth: usize,
|
||||
list: *[maximum_bundled]Bundled,
|
||||
count: *usize,
|
||||
) !void {
|
||||
// Each read() on a directory yields one EFI_FILE_INFO; zero bytes means done.
|
||||
var info_buffer: [1024]u8 align(8) = undefined;
|
||||
while (true) {
|
||||
const n = try directory.read(&info_buffer);
|
||||
if (n == 0) return;
|
||||
const info: *const uefi.protocol.File.Info.File = @ptrCast(@alignCast(&info_buffer));
|
||||
const name16 = info.getFileName();
|
||||
|
||||
// Convert the (ASCII in practice) UTF-16 name. A hostile-shaped entry
|
||||
// (too long, non-ASCII) is SKIPPED, never fatal — one odd file on a
|
||||
// hand-written stick must not cost the whole boot. Names are lowered:
|
||||
// the danos tree is canonically lowercase and FAT lookups are
|
||||
// case-insensitive, but firmware ENUMERATION returns whatever the
|
||||
// directory stores — an 8.3 short entry comes back uppercase ("INIT"),
|
||||
// which would otherwise poison every path comparison downstream.
|
||||
var name_buffer: [initial_ramdisk.maximum_name]u8 = undefined;
|
||||
var name_length: usize = 0;
|
||||
var name_ok = true;
|
||||
while (name16[name_length] != 0) : (name_length += 1) {
|
||||
if (name_length == name_buffer.len) {
|
||||
name_ok = false;
|
||||
break;
|
||||
}
|
||||
const c = name16[name_length];
|
||||
if (c > 0x7F) {
|
||||
name_ok = false;
|
||||
break;
|
||||
}
|
||||
name_buffer[name_length] = std.ascii.toLower(@intCast(c));
|
||||
}
|
||||
if (!name_ok) continue;
|
||||
const name = name_buffer[0..name_length];
|
||||
// Skip dot entries: "." / ".." and host-OS litter (macOS "._*" AppleDouble
|
||||
// resource forks, ".fseventsd", ".Spotlight-V100") a copied-onto stick
|
||||
// accumulates — none of it is a danos binary.
|
||||
if (name.len == 0 or name[0] == '.') continue;
|
||||
|
||||
if (info.attribute.directory) {
|
||||
if (depth == maximum_tree_depth) continue;
|
||||
var child_prefix: [initial_ramdisk.maximum_name]u8 = undefined;
|
||||
const child = try std.fmt.bufPrint(&child_prefix, "{s}/{s}", .{ prefix, name });
|
||||
const child_directory = try directory.open(name16, .read, .{});
|
||||
defer _ = child_directory.close() catch {};
|
||||
try walkDirectory(bs, child_directory, child, depth + 1, list, count);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (depth == 0) continue; // /system/kernel — already loaded, not bundled
|
||||
if (count.* == maximum_bundled) return error.TooManyBinaries;
|
||||
|
||||
var entry: *Bundled = &list[count.*];
|
||||
const path = std.fmt.bufPrint(&entry.path, "{s}/{s}", .{ prefix, name }) catch continue; // path too long: skip the file, keep the boot
|
||||
entry.path_len = path.len;
|
||||
|
||||
const file = directory.open(name16, .read, .{}) catch continue;
|
||||
defer _ = file.close() catch {};
|
||||
entry.data = readWholeFile(bs, file) catch continue; // unreadable/empty: skip
|
||||
count.* += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Read an open file completely into a fresh pool buffer that survives the
|
||||
/// handoff (LoaderData is classified reserved, so the kernel identity-maps it).
|
||||
fn readWholeFile(bs: *uefi.tables.BootServices, file: *uefi.protocol.File) ![]align(8) u8 {
|
||||
try file.setPosition(seek_end);
|
||||
const size: usize = @intCast(try file.getPosition());
|
||||
try file.setPosition(0);
|
||||
if (size == 0) return error.EmptyFile;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||
|
||||
const buffer = try bs.allocatePool(.loader_data, size);
|
||||
var read_total: usize = 0;
|
||||
while (read_total < size) {
|
||||
const n = try file.read(image[read_total..]);
|
||||
const n = try file.read(buffer[read_total..]);
|
||||
if (n == 0) return error.UnexpectedEof;
|
||||
read_total += n;
|
||||
}
|
||||
return image[0..size];
|
||||
}
|
||||
|
||||
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
progress("EFI: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
progress("EFI: initial_ramdisk loaded\r\n");
|
||||
return buffer[0..size];
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
|
||||
@@ -56,70 +56,63 @@ fn timestamp(b: *std.Build) []const u8 {
|
||||
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||
///
|
||||
/// The compilation root is not the program's own file but the shared shim
|
||||
/// library/runtime/root.zig, which supplies the root declarations (`main`
|
||||
/// library/kernel/root.zig, which supplies the root declarations (`main`
|
||||
/// re-export, panic handler, `_start` pull) so a program only defines
|
||||
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||
/// imports; reach it through `programModule` to add per-binary imports.
|
||||
fn addUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
default_imports: []const std.Build.Module.Import,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, false);
|
||||
return addUserBinaryImpl(b, target, default_imports, name, root, false);
|
||||
}
|
||||
|
||||
/// As `addUserBinary`, but built multi-threaded (`single_threaded = false`) so real
|
||||
/// atomics/TLS work — required before a binary may call `runtime.Thread.spawn`
|
||||
/// atomics/TLS work — required before a binary may call `Thread.spawn`
|
||||
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||
fn addThreadedUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
default_imports: []const std.Build.Module.Import,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, true);
|
||||
return addUserBinaryImpl(b, target, default_imports, name, root, true);
|
||||
}
|
||||
|
||||
/// The module registered under `name` in `imports` — the root shim reaches the
|
||||
/// couple of concern modules it needs (start, logging) out of the default set.
|
||||
fn findImport(imports: []const std.Build.Module.Import, name: []const u8) *std.Build.Module {
|
||||
for (imports) |import| {
|
||||
if (std.mem.eql(u8, import.name, name)) return import.module;
|
||||
}
|
||||
@panic("default_imports is missing a module the root shim needs");
|
||||
}
|
||||
|
||||
fn addUserBinaryImpl(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
default_imports: []const std.Build.Module.Import,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
threaded: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Settings (target, optimize, code model, ...) live on the root module only;
|
||||
// the program and runtime modules leave theirs null and inherit them.
|
||||
// Every user binary gets the same default set of importable modules — the library/kernel
|
||||
// concern modules (ipc, memory, process, time, logging, file-system, ...), the device/
|
||||
// service clients (driver, block, display, input), mmio, acpi-ids, and xkeyboard-config.
|
||||
// Per-binary extras go through programModule(exe).addImport. Settings (target, optimize,
|
||||
// code model, ...) live on the root module only; the program module inherits them.
|
||||
const program_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
// to any program that wants it. See library/xkeyboard-config/.
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
},
|
||||
.imports = default_imports,
|
||||
});
|
||||
const exe = b.addExecutable(.{
|
||||
.name = name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/root.zig"),
|
||||
.root_source_file = b.path("library/kernel/root.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
@@ -127,13 +120,16 @@ fn addUserBinaryImpl(
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
// The root shim itself imports only start (_start + panic) and logging
|
||||
// (std_options); the program's own file reaches the full default set.
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
.{ .name = "start", .module = findImport(default_imports, "start") },
|
||||
.{ .name = "logging", .module = findImport(default_imports, "logging") },
|
||||
.{ .name = "program", .module = program_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("library/runtime/user.ld"));
|
||||
exe.setLinkerScript(b.path("library/kernel/user.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
exe.image_base = 0x7000_0000_0000;
|
||||
exe.use_llvm = true;
|
||||
@@ -195,6 +191,7 @@ fn addKernel(
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.strip = optimize != .Debug, // DWARF info doubles the flashable image; keep it only for debug builds
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = modules.boot_handoff },
|
||||
.{ .name = "abi", .module = modules.abi },
|
||||
@@ -221,18 +218,23 @@ fn addKernel(
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||
/// bundle its own serial kernel while sharing the loader, init, and ramdisk — all
|
||||
/// built once per invocation (the loader's boot breadcrumbs and init's heartbeat
|
||||
/// both follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
/// One user binary and its FHS home on the boot volume (and in zig-out).
|
||||
const BundledBinary = struct { path: []const u8, binary: std.Build.LazyPath };
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding the
|
||||
/// EFI stub, the kernel, and every user binary at its FHS path — the volume's
|
||||
/// /system tree IS the system image; the EFI loader walks it at boot and builds
|
||||
/// the in-RAM initial_ramdisk from it. Factored so the serial-enabled
|
||||
/// `run-x86-64` variant can bundle its own serial kernel while sharing the
|
||||
/// loader and user tree (the loader's boot breadcrumbs and init's heartbeat both
|
||||
/// follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
init_bin: std.Build.LazyPath,
|
||||
initial_ramdisk_img: std.Build.LazyPath,
|
||||
manifest: std.Build.LazyPath,
|
||||
capsule: std.Build.LazyPath,
|
||||
bundled: []const BundledBinary,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
@@ -242,10 +244,14 @@ fn addBootImage(
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/services/init");
|
||||
mk_fat.addFileArg(init_bin);
|
||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||
mk_fat.addFileArg(initial_ramdisk_img);
|
||||
mk_fat.addArg("system/manifest");
|
||||
mk_fat.addFileArg(manifest);
|
||||
mk_fat.addArg("boot/system.img");
|
||||
mk_fat.addFileArg(capsule);
|
||||
for (bundled) |item| {
|
||||
mk_fat.addArg(item.path);
|
||||
mk_fat.addFileArg(item.binary);
|
||||
}
|
||||
return fat_image;
|
||||
}
|
||||
|
||||
@@ -269,14 +275,21 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
// The devices sub-project's public interface (the flat wire types), exposed as
|
||||
// its own module like vfs-protocol — importable by user space, unlike the
|
||||
// kernel-internal device model it also feeds (system/devices/device-model.zig).
|
||||
// kernel-internal device model it also feeds (system/kernel/device-model.zig).
|
||||
const device_abi_module = b.addModule("device-abi", .{
|
||||
.root_source_file = b.path("system/devices/device-abi.zig"),
|
||||
.root_source_file = b.path("library/device/model/device-abi.zig"),
|
||||
});
|
||||
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference data,
|
||||
// shared by kernel discovery (the device-tree dump) and any user-space PCI tool.
|
||||
const pci_class_module = b.addModule("pci-class", .{
|
||||
.root_source_file = b.path("system/devices/pci-class.zig"),
|
||||
.root_source_file = b.path("library/device/pci/pci-class.zig"),
|
||||
});
|
||||
// The device registry: parse /etc/devices.csv into match rules and bind a
|
||||
// reported device to a driver — the data-driven, authoritative replacement for
|
||||
// the manager's hand-written switch tables. Pure logic (no hardware, no
|
||||
// syscalls), so it unit-tests with plain `zig test`; the manager imports it.
|
||||
const device_registry_module = b.addModule("device-registry", .{
|
||||
.root_source_file = b.path("library/device/registry/device-registry.zig"),
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||
// nodes. Also shared reference data.
|
||||
@@ -284,11 +297,11 @@ pub fn build(b: *std.Build) void {
|
||||
// same parser the kernel does (docs/discovery.md — the shared AML module).
|
||||
// Pure Zig, no kernel imports — one source, two builds.
|
||||
const aml_module = b.addModule("aml", .{
|
||||
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||
.root_source_file = b.path("library/device/acpi/aml/aml.zig"),
|
||||
});
|
||||
|
||||
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||
.root_source_file = b.path("library/device/acpi/acpi-ids.zig"),
|
||||
});
|
||||
|
||||
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||
@@ -296,21 +309,21 @@ pub fn build(b: *std.Build) void {
|
||||
// reference the xHCI bus driver, the USB class drivers, and the device
|
||||
// manager's identity matcher all share. Pure data, like pci-class/acpi-ids.
|
||||
const usb_abi_module = b.addModule("usb-abi", .{
|
||||
.root_source_file = b.path("system/devices/usb-abi.zig"),
|
||||
.root_source_file = b.path("library/device/usb/usb-abi.zig"),
|
||||
});
|
||||
const usb_ids_module = b.addModule("usb-ids", .{
|
||||
.root_source_file = b.path("system/devices/usb-ids.zig"),
|
||||
.root_source_file = b.path("library/device/usb/usb-ids.zig"),
|
||||
});
|
||||
// The USB transfer protocol: what a USB class driver says to the xHCI bus
|
||||
// driver to drive its device (open / control / interrupt / bulk). A protocol
|
||||
// module like vfs-protocol, shared by the bus driver and every class driver.
|
||||
const usb_transfer_protocol_module = b.addModule("usb-transfer-protocol", .{
|
||||
.root_source_file = b.path("system/drivers/usb-xhci-bus/usb-transfer-protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/usb-transfer/usb-transfer-protocol.zig"),
|
||||
});
|
||||
// The block-device protocol: read/write of fixed-size blocks, spoken between a
|
||||
// filesystem and a block driver (usb-storage). A protocol module like the rest.
|
||||
const block_protocol_module = b.addModule("block-protocol", .{
|
||||
.root_source_file = b.path("system/services/block/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/block/block-protocol.zig"),
|
||||
});
|
||||
|
||||
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||
@@ -342,15 +355,13 @@ pub fn build(b: *std.Build) void {
|
||||
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
||||
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
||||
// architecture module applies to CPU code. The backend is selected at runtime from
|
||||
// the boot handoff (see system/devices/platform.zig).
|
||||
// the boot handoff (see system/kernel/platform.zig).
|
||||
const platform_module = b.addModule("platform", .{
|
||||
.root_source_file = b.path("system/devices/platform.zig"),
|
||||
.root_source_file = b.path("system/kernel/platform.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module }, // BootInformation (carries the ACPI RSDP), physicalToVirtual
|
||||
.{ .name = "abi", .module = abi_module }, // acpi.zig works in page_size units
|
||||
.{ .name = "device-abi", .module = device_abi_module }, // device-model's DeviceClass/ResourceKind live here
|
||||
.{ .name = "pci-class", .module = pci_class_module }, // decode PCI class codes in the device dump
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module }, // decode ACPI _HID names in the device dump
|
||||
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||
},
|
||||
});
|
||||
@@ -361,14 +372,14 @@ pub fn build(b: *std.Build) void {
|
||||
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||
// expose theirs the same way.
|
||||
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/vfs/vfs-protocol.zig"),
|
||||
});
|
||||
|
||||
// The input wire protocol: the input service's public interface, exposed as its own
|
||||
// module the same way vfs-protocol is. Shared by the input service, the runtime's
|
||||
// `input` helper (subscribe/publish), and every source and subscriber.
|
||||
const input_protocol_module = b.addModule("input-protocol", .{
|
||||
.root_source_file = b.path("system/services/input/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/input/input-protocol.zig"),
|
||||
});
|
||||
|
||||
// The danos-native user-space runtime: system_call wrappers, the C-convention
|
||||
@@ -380,53 +391,165 @@ pub fn build(b: *std.Build) void {
|
||||
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
|
||||
// server. It never touches `boot-handoff` — user space has no business with the
|
||||
// loader↔kernel handoff.
|
||||
const runtime_module = b.addModule("runtime", .{
|
||||
.root_source_file = b.path("library/runtime/runtime.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||
// own module like the other protocol modules. Imported through the runtime.
|
||||
// Wire-protocol modules the kernel-library client wrappers (device-manager/block/display)
|
||||
// and the services speak. The `runtime` module itself is defined below, after the
|
||||
// library/kernel concern modules it shims over.
|
||||
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/device-manager/device-manager-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The USB transfer protocol, so runtime.usb (the class-driver client) can speak
|
||||
// it, the way runtime.input speaks the input protocol.
|
||||
runtime_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
// The block protocol, so runtime.block (the block-device client) can speak it.
|
||||
runtime_module.addImport("block-protocol", block_protocol_module);
|
||||
|
||||
// The display protocol, so runtime.display (the compositor client) and the display
|
||||
// service both speak it through the runtime, like the other protocol modules.
|
||||
const display_protocol_module = b.addModule("display-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/display/display-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("display-protocol", display_protocol_module);
|
||||
|
||||
// The scanout protocol: the compositor's outbound present channel to a native scanout
|
||||
// driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md).
|
||||
// No runtime client speaks it — imported directly by the compositor and the scanout driver.
|
||||
const scanout_protocol_module = b.addModule("scanout-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/scanout-protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/scanout/scanout-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("scanout-protocol", scanout_protocol_module);
|
||||
|
||||
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||
// The power protocol: system power's domain-named surface (docs/power.md). No runtime
|
||||
// client speaks it — imported directly by init and the acpi discovery service.
|
||||
const power_protocol_module = b.addModule("power-protocol", .{
|
||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/power/power-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("power-protocol", power_protocol_module);
|
||||
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||
// no target set, so it inherits each driver's. See library/device/mmio/mmio.zig.
|
||||
const mmio_module = b.addModule("mmio", .{
|
||||
.root_source_file = b.path("library/mmio/mmio.zig"),
|
||||
.root_source_file = b.path("library/device/mmio/mmio.zig"),
|
||||
});
|
||||
|
||||
// --- library/kernel: the userspace private-ABI library (kernel32-style), split by
|
||||
// concern into directly-importable modules. `system.zig` (the old dumping ground) and
|
||||
// `runtime.zig` (the old aggregator) are compatibility shims re-exporting these until the
|
||||
// consumers migrate to direct imports (reorg C1–C5). The graph is a DAG: memory depends on
|
||||
// thread (heap needs Thread.Mutex), and thread does its own raw mmap so there is no cycle.
|
||||
const system_call_module = b.addModule("system-call", .{
|
||||
.root_source_file = b.path("library/kernel/system-call.zig"),
|
||||
.imports = &.{.{ .name = "abi", .module = abi_module }},
|
||||
});
|
||||
const ipc_module = b.addModule("ipc", .{
|
||||
.root_source_file = b.path("library/kernel/ipc.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
|
||||
});
|
||||
const time_module = b.addModule("time", .{
|
||||
.root_source_file = b.path("library/kernel/time.zig"),
|
||||
.imports = &.{.{ .name = "system-call", .module = system_call_module }},
|
||||
});
|
||||
const thread_module = b.addModule("thread", .{
|
||||
.root_source_file = b.path("library/kernel/thread.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
|
||||
});
|
||||
const logging_module = b.addModule("logging", .{
|
||||
.root_source_file = b.path("library/kernel/logging.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
|
||||
});
|
||||
const process_module = b.addModule("process", .{
|
||||
.root_source_file = b.path("library/kernel/process.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
},
|
||||
});
|
||||
const file_system_module = b.addModule("file-system", .{
|
||||
.root_source_file = b.path("library/kernel/file-system.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
},
|
||||
});
|
||||
const memory_module = b.addModule("memory", .{
|
||||
.root_source_file = b.path("library/kernel/memory/memory.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "thread", .module = thread_module },
|
||||
},
|
||||
});
|
||||
const service_module = b.addModule("service", .{
|
||||
.root_source_file = b.path("library/kernel/service.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "process", .module = process_module },
|
||||
},
|
||||
});
|
||||
const start_module = b.addModule("start", .{
|
||||
.root_source_file = b.path("library/kernel/start.zig"),
|
||||
.imports = &.{ .{ .name = "process", .module = process_module }, .{ .name = "logging", .module = logging_module } },
|
||||
});
|
||||
// The driver author's interface (library/device/driver): device access + the
|
||||
// device-manager hello handshake, folded together.
|
||||
const driver_module = b.addModule("driver", .{
|
||||
.root_source_file = b.path("library/device/driver/driver.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "device-manager-protocol", .module = device_manager_protocol_module },
|
||||
},
|
||||
});
|
||||
// The block-device client — a device type, so library/device/block.
|
||||
const block_client_module = b.addModule("block", .{
|
||||
.root_source_file = b.path("library/device/block/block.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "block-protocol", .module = block_protocol_module },
|
||||
},
|
||||
});
|
||||
// Userspace-service clients live in library/client (they talk to services, not the kernel).
|
||||
const display_client_module = b.addModule("display", .{
|
||||
.root_source_file = b.path("library/client/display/display.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "display-protocol", .module = display_protocol_module },
|
||||
},
|
||||
});
|
||||
const input_client_module = b.addModule("input", .{
|
||||
.root_source_file = b.path("library/client/input/input.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||
},
|
||||
});
|
||||
// A device driver's view of its claimed PCI function: config-space header fields, BAR
|
||||
// decode + map, and the capability walk (library/device/pci/pci.zig). The generic PCI
|
||||
// mechanics every leaf PCI driver used to re-derive inline. Imports the driver (device
|
||||
// access) client + mmio + the pci-class data module (config-space layout constants).
|
||||
const pci_module = b.addModule("pci", .{
|
||||
.root_source_file = b.path("library/device/pci/pci.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "driver", .module = driver_module },
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
.{ .name = "pci-class", .module = pci_class_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The USB class-driver transfer client (library/device/usb/usb.zig): open a device on
|
||||
// the xHCI bus and drive it (control / interrupt / bulk). Bus-family logic a class
|
||||
// driver imports directly — over ipc + time. Re-exports usb-abi / usb-ids as
|
||||
// usb.abi / usb.ids for a single USB import.
|
||||
const usb_module = b.addModule("usb", .{
|
||||
.root_source_file = b.path("library/device/usb/usb.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "usb-transfer-protocol", .module = usb_transfer_protocol_module },
|
||||
.{ .name = "usb-abi", .module = usb_abi_module },
|
||||
.{ .name = "usb-ids", .module = usb_ids_module },
|
||||
},
|
||||
});
|
||||
|
||||
// Keyboard layouts compiled from the X11 xkeyboard-config database into native Zig
|
||||
@@ -443,8 +566,9 @@ pub fn build(b: *std.Build) void {
|
||||
},
|
||||
});
|
||||
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and
|
||||
// the EFI loader (packs it in RAM from the boot volume's /system tree). No
|
||||
// dependencies.
|
||||
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||
});
|
||||
@@ -458,6 +582,10 @@ pub fn build(b: *std.Build) void {
|
||||
// QEMU test harness (test/qemu_test.py, which asserts on serial markers) turn
|
||||
// it on; a flashable `zig build` image leaves it out. See serial.zig.
|
||||
const serial = b.option(bool, "serial", "Compile the serial-console log sink into the kernel (default: off; run-x86-64 and the test harness enable it)") orelse false;
|
||||
// The diagnose boot: init skips the display service (and demo), so the
|
||||
// on-screen boot transcript is never suppressed — the full timestamped
|
||||
// timeline stays on the screen for real-hardware debugging by eye.
|
||||
const diagnose = b.option(bool, "diagnose", "Boot without the display service so the timestamped boot transcript stays on screen (real-hardware debugging)") orelse false;
|
||||
|
||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||
@@ -490,10 +618,34 @@ pub fn build(b: *std.Build) void {
|
||||
b.getInstallStep().dependOn(&kernel_install.step);
|
||||
|
||||
// --- init: the first user-space program (a system service) ---
|
||||
// The default module set every user binary can import directly: the library/kernel
|
||||
// concern modules, the device/service clients, mmio, the keyboard layouts, and the ACPI
|
||||
// id registry. Per-binary extras are added with programModule(exe).addImport.
|
||||
const default_imports = [_]std.Build.Module.Import{
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "memory", .module = memory_module },
|
||||
.{ .name = "process", .module = process_module },
|
||||
.{ .name = "thread", .module = thread_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "logging", .module = logging_module },
|
||||
.{ .name = "file-system", .module = file_system_module },
|
||||
.{ .name = "service", .module = service_module },
|
||||
.{ .name = "start", .module = start_module },
|
||||
.{ .name = "driver", .module = driver_module },
|
||||
.{ .name = "block", .module = block_client_module },
|
||||
.{ .name = "display", .module = display_client_module },
|
||||
.{ .name = "input", .module = input_client_module },
|
||||
};
|
||||
|
||||
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||
// linked into the kernel's user region against the library/kernel modules, and
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||
const init_exe = addUserBinary(b, kernel_target, &default_imports, "init", "system/services/init/init.zig");
|
||||
programModule(init_exe).addImport("power-protocol", power_protocol_module);
|
||||
// init reads the same `serial` flag the kernel does: its liveness heartbeat is a
|
||||
// serial/test-build diagnostic (the QEMU harness's init tests assert on it, and
|
||||
// -Dserial images emit it), so a flashable image runs a purely event-driven PID 1
|
||||
@@ -501,20 +653,21 @@ pub fn build(b: *std.Build) void {
|
||||
// the heartbeat stays present under test.
|
||||
const init_options = b.addOptions();
|
||||
init_options.addOption(bool, "serial", serial);
|
||||
init_options.addOption(bool, "diagnose", diagnose);
|
||||
programModule(init_exe).addImport("build_options", init_options.createModule());
|
||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||
b.getInstallStep().dependOn(&init_install.step);
|
||||
|
||||
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||
// Each is built by the same user-binary recipe, then packed into one image by
|
||||
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
// --- the rest of the boot tree: /system services and drivers, /test fixtures ---
|
||||
// Each is built by the same user-binary recipe and laid out at its FHS path on
|
||||
// the boot volume (see `bundled` below). The EFI loader walks the tree at boot
|
||||
// and hands the kernel an in-RAM initial_ramdisk of it (system/initial-ramdisk.zig).
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, &default_imports, "vfs-test", "test/system/services/vfs-test/vfs-test.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, &default_imports, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, &default_imports, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
programModule(ps2_keyboard_exe).addImport("input-protocol", input_protocol_module);
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, &default_imports, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
programModule(ps2_mouse_exe).addImport("input-protocol", input_protocol_module);
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, &default_imports, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
programModule(usb_xhci_bus_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||
// usb-ids.packTriple.
|
||||
@@ -524,31 +677,47 @@ pub fn build(b: *std.Build) void {
|
||||
// The USB HID class drivers: keyboard and mouse. They own no hardware — each
|
||||
// opens its device through runtime.usb (the transfer protocol) and publishes to
|
||||
// the input service. They build chapter-9 class requests from usb-abi.
|
||||
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, &default_imports, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||
programModule(usb_hid_keyboard_exe).addImport("usb", usb_module);
|
||||
programModule(usb_hid_keyboard_exe).addImport("usb-abi", usb_abi_module);
|
||||
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||
programModule(usb_hid_keyboard_exe).addImport("input-protocol", input_protocol_module);
|
||||
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, &default_imports, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||
programModule(usb_hid_mouse_exe).addImport("usb", usb_module);
|
||||
programModule(usb_hid_mouse_exe).addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_hid_mouse_exe).addImport("input-protocol", input_protocol_module);
|
||||
// The USB mass-storage class driver: opens its device via runtime.usb, drives it
|
||||
// with Bulk-Only Transport + SCSI, and serves the block protocol under `.block`.
|
||||
const usb_storage_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||
const usb_storage_exe = addUserBinary(b, kernel_target, &default_imports, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||
programModule(usb_storage_exe).addImport("usb", usb_module);
|
||||
programModule(usb_storage_exe).addImport("block-protocol", block_protocol_module);
|
||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||
const display_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||
const shm_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-server", "system/services/shm-server/shm-server.zig");
|
||||
const shm_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-client", "system/services/shm-client/shm-client.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
const fat_exe = addUserBinary(b, kernel_target, &default_imports, "fat", "system/services/fat/fat.zig");
|
||||
programModule(fat_exe).addImport("vfs-protocol", vfs_protocol_module);
|
||||
// Threaded: the display runs a mouse-listener thread alongside its compositor loop
|
||||
// (docs/threading.md, docs/display.md), so it opts into real atomics/TLS.
|
||||
const display_exe = addThreadedUserBinary(b, kernel_target, &default_imports, "display", "system/services/display/display.zig");
|
||||
programModule(display_exe).addImport("display-protocol", display_protocol_module);
|
||||
programModule(display_exe).addImport("scanout-protocol", scanout_protocol_module);
|
||||
const display_demo_exe = addUserBinary(b, kernel_target, &default_imports, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, &default_imports, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||
programModule(virtio_gpu_exe).addImport("pci", pci_module); // library/device/pci — the claimed-function view
|
||||
programModule(virtio_gpu_exe).addImport("display-protocol", display_protocol_module);
|
||||
programModule(virtio_gpu_exe).addImport("scanout-protocol", scanout_protocol_module);
|
||||
const shared_memory_server_exe = addUserBinary(b, kernel_target, &default_imports, "shared-memory-server", "test/system/services/shared-memory-server/shared-memory-server.zig");
|
||||
const shared_memory_client_exe = addUserBinary(b, kernel_target, &default_imports, "shared-memory-client", "test/system/services/shared-memory-client/shared-memory-client.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, &default_imports, "fat-test", "test/system/services/fat-test/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, &default_imports, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
programModule(pci_bus_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The PCI bus driver decodes each function's class triple to human names in its
|
||||
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||
programModule(pci_bus_exe).addImport("pci-class", pci_class_module);
|
||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||
// what the driver-restart scenario drives the crash-loop cap with.
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, &default_imports, "crash-test", "test/system/services/crash-test/crash-test.zig");
|
||||
programModule(crash_test_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
const device_list_exe = addUserBinary(b, kernel_target, &default_imports, "device-list", "test/system/services/device-list/device-list.zig");
|
||||
programModule(device_list_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The discovery service: one swappable process per firmware
|
||||
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||
// "discovery" so the device manager never learns which firmware it is on.
|
||||
@@ -562,113 +731,111 @@ pub fn build(b: *std.Build) void {
|
||||
.acpi => "system/services/acpi/acpi.zig",
|
||||
.fdt => "system/services/fdt/fdt.zig",
|
||||
};
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
const discovery_exe = addUserBinary(b, kernel_target, &default_imports, "discovery", discovery_source);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("aml", aml_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||
programModule(device_manager_exe).addImport("pci-class", pci_class_module);
|
||||
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||
programModule(device_manager_exe).addImport("usb-ids", usb_ids_module);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("power-protocol", power_protocol_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, &default_imports, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
programModule(device_manager_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// Driver matching is data-driven: the manager parses /etc/devices.csv into this
|
||||
// module's rules and binds each reported device by most-specific match.
|
||||
programModule(device_manager_exe).addImport("device-registry", device_registry_module);
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||
const input_exe = addUserBinary(b, kernel_target, &default_imports, "input", "system/services/input/input.zig");
|
||||
programModule(input_exe).addImport("input-protocol", input_protocol_module);
|
||||
const input_source_exe = addUserBinary(b, kernel_target, &default_imports, "input-source", "test/system/services/input-source/input-source.zig");
|
||||
const input_test_exe = addUserBinary(b, kernel_target, &default_imports, "input-test", "test/system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, &default_imports, "args-echo", "test/system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, &default_imports, "process-test", "test/system/services/process-test/process-test.zig");
|
||||
const logger_exe = addUserBinary(b, kernel_target, &default_imports, "logger", "system/services/logger/logger.zig");
|
||||
// The first multi-threaded binary: exercises runtime.Thread over the thread ABI
|
||||
// (docs/threading.md). Built threaded so its shared-memory poll is real.
|
||||
const thread_test_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "thread-test", "system/services/thread-test/thread-test.zig");
|
||||
const thread_test_exe = addThreadedUserBinary(b, kernel_target, &default_imports, "thread-test", "test/system/services/thread-test/thread-test.zig");
|
||||
|
||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||
mk_run.addArg("vfs");
|
||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||
mk_run.addArg("vfs-test");
|
||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-bus");
|
||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-keyboard");
|
||||
mk_run.addFileArg(ps2_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-mouse");
|
||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-xhci-bus");
|
||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-keyboard");
|
||||
mk_run.addFileArg(usb_hid_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-mouse");
|
||||
mk_run.addFileArg(usb_hid_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-storage");
|
||||
mk_run.addFileArg(usb_storage_exe.getEmittedBin());
|
||||
mk_run.addArg("fat");
|
||||
mk_run.addFileArg(fat_exe.getEmittedBin());
|
||||
mk_run.addArg("fat-test");
|
||||
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
||||
mk_run.addArg("display");
|
||||
mk_run.addFileArg(display_exe.getEmittedBin());
|
||||
mk_run.addArg("display-demo");
|
||||
mk_run.addFileArg(display_demo_exe.getEmittedBin());
|
||||
mk_run.addArg("virtio-gpu");
|
||||
mk_run.addFileArg(virtio_gpu_exe.getEmittedBin());
|
||||
mk_run.addArg("shm-server");
|
||||
mk_run.addFileArg(shm_server_exe.getEmittedBin());
|
||||
mk_run.addArg("shm-client");
|
||||
mk_run.addFileArg(shm_client_exe.getEmittedBin());
|
||||
mk_run.addArg("pci-bus");
|
||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("crash-test");
|
||||
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||
mk_run.addArg("thread-test");
|
||||
mk_run.addFileArg(thread_test_exe.getEmittedBin());
|
||||
mk_run.addArg("device-list");
|
||||
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||
mk_run.addArg("discovery");
|
||||
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||
mk_run.addArg("device-manager");
|
||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||
mk_run.addArg("input");
|
||||
mk_run.addFileArg(input_exe.getEmittedBin());
|
||||
mk_run.addArg("input-source");
|
||||
mk_run.addFileArg(input_source_exe.getEmittedBin());
|
||||
mk_run.addArg("input-test");
|
||||
mk_run.addFileArg(input_test_exe.getEmittedBin());
|
||||
mk_run.addArg("args-echo");
|
||||
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||
mk_run.addArg("process-test");
|
||||
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||
mk_run.addArg("log-flush");
|
||||
mk_run.addFileArg(log_flush_exe.getEmittedBin());
|
||||
// Every user binary and its FHS home on the boot volume. There is no packed
|
||||
// ramdisk artifact any more: make-fat-image.py lays each binary out at this
|
||||
// path on the image, and the EFI loader walks /system and /test at boot and
|
||||
// builds the in-RAM initial_ramdisk table from the trees — the volume's file
|
||||
// structure is the single source of truth. Entry names (and hence argv[0] and
|
||||
// task names) are these paths with a leading slash. Test fixtures mirror their
|
||||
// repo home: test/system/services/<name> in the source tree IS the boot path.
|
||||
const bundled = [_]BundledBinary{
|
||||
.{ .path = "system/services/init", .binary = init_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/fat", .binary = fat_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/display", .binary = display_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/display-demo", .binary = display_demo_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/device-manager", .binary = device_manager_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/input", .binary = input_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/discovery", .binary = discovery_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/logger", .binary = logger_exe.getEmittedBin() },
|
||||
// A data file, not a binary: the device registry the manager reads at boot.
|
||||
// Packing it under /etc makes the kernel auto-mount /etc as a read-only
|
||||
// initrd tree (system/kernel/vfs.zig setInitialRamdisk), so the manager can
|
||||
// fs.open("/etc/devices.csv") with no filesystem service running.
|
||||
.{ .path = "etc/devices.csv", .binary = b.path("etc/devices.csv") },
|
||||
.{ .path = "system/drivers/ps2-bus", .binary = ps2_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-keyboard", .binary = ps2_keyboard_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-mouse", .binary = ps2_mouse_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-xhci-bus", .binary = usb_xhci_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-hid-keyboard", .binary = usb_hid_keyboard_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-hid-mouse", .binary = usb_hid_mouse_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-storage", .binary = usb_storage_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/virtio-gpu", .binary = virtio_gpu_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/pci-bus", .binary = pci_bus_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/vfs-test", .binary = vfstest_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/fat-test", .binary = fat_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/shared-memory-server", .binary = shared_memory_server_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/shared-memory-client", .binary = shared_memory_client_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/crash-test", .binary = crash_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/device-list", .binary = device_list_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/input-source", .binary = input_source_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/input-test", .binary = input_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/args-echo", .binary = args_echo_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/process-test", .binary = process_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/thread-test", .binary = thread_test_exe.getEmittedBin() },
|
||||
};
|
||||
|
||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||
.{ vfs_exe, "system/services" },
|
||||
.{ device_manager_exe, "system/services" },
|
||||
.{ input_exe, "system/services" },
|
||||
.{ ps2_bus_exe, "system/drivers" },
|
||||
.{ ps2_keyboard_exe, "system/drivers" },
|
||||
.{ ps2_mouse_exe, "system/drivers" },
|
||||
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||
.{ usb_hid_keyboard_exe, "system/drivers" },
|
||||
.{ usb_hid_mouse_exe, "system/drivers" },
|
||||
.{ usb_storage_exe, "system/drivers" },
|
||||
.{ fat_exe, "system/services" },
|
||||
.{ display_exe, "system/services" },
|
||||
.{ log_flush_exe, "system/services" },
|
||||
}) |entry| {
|
||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||
b.getInstallStep().dependOn(&step.step);
|
||||
// The boot manifest: the FHS path of every bundled binary, one per line. The
|
||||
// EFI loader reads THIS by name and opens each listed path by name — FAT
|
||||
// name lookup is case-insensitive and firmware-portable, unlike directory
|
||||
// ENUMERATION, whose returned names vary by firmware (bare 8.3 entries come
|
||||
// back uppercase on some FAT drivers). The tree walk remains only as the
|
||||
// loader's fallback for hand-assembled sticks without a manifest.
|
||||
var manifest_text: std.ArrayListUnmanaged(u8) = .empty;
|
||||
for (bundled) |item| {
|
||||
manifest_text.append(b.allocator, '/') catch @panic("OOM");
|
||||
manifest_text.appendSlice(b.allocator, item.path) catch @panic("OOM");
|
||||
manifest_text.append(b.allocator, '\n') catch @panic("OOM");
|
||||
}
|
||||
const manifest_files = b.addWriteFiles();
|
||||
const manifest_file = manifest_files.add("manifest", manifest_text.items);
|
||||
const manifest_install = b.addInstallFileWithDir(manifest_file, .prefix, "system/manifest");
|
||||
b.getInstallStep().dependOn(&manifest_install.step);
|
||||
|
||||
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||
// The boot capsule: the same bundled list packed into ONE file (v2
|
||||
// initial_ramdisk format), because a single open + sequential read is the
|
||||
// only firmware file I/O shape that is fast everywhere — a per-file tree
|
||||
// walk measured MINUTES on real firmware. The loader tries this first,
|
||||
// then the manifest, then the walk; the running system cannot tell the
|
||||
// difference (it always receives the same in-RAM table). Derived from the
|
||||
// tree in the same build graph, so the two cannot drift.
|
||||
const mk_capsule = b.addSystemCommand(&.{"python3"});
|
||||
mk_capsule.addFileArg(b.path("tools/pack-system-image.py"));
|
||||
const capsule_img = mk_capsule.addOutputFileArg("system.img");
|
||||
for (bundled) |item| {
|
||||
mk_capsule.addArg(item.path);
|
||||
mk_capsule.addFileArg(item.binary);
|
||||
}
|
||||
const capsule_install = b.addInstallFile(capsule_img, "boot/system.img");
|
||||
b.getInstallStep().dependOn(&capsule_install.step);
|
||||
|
||||
// Install every bundled binary to its FHS home, so zig-out is a true image of
|
||||
// the filesystem — the same tree make-fat-image.py lays out on the boot volume.
|
||||
for (bundled) |item| {
|
||||
const install = b.addInstallFileWithDir(item.binary, .prefix, item.path);
|
||||
b.getInstallStep().dependOn(&install.step);
|
||||
}
|
||||
|
||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
@@ -690,8 +857,10 @@ pub fn build(b: *std.Build) void {
|
||||
}),
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||
// The bootloader speaks the handoff contract and the ramdisk
|
||||
// container it packs the /system tree into — never the user ABI.
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||
.{ .name = "build_options", .module = loader_options_module },
|
||||
},
|
||||
}),
|
||||
@@ -704,11 +873,11 @@ pub fn build(b: *std.Build) void {
|
||||
|
||||
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||
// holding exactly what the firmware and bootloader need off the ESP: the EFI
|
||||
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||
// danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), manifest_file, capsule_img, &bundled);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
@@ -717,7 +886,7 @@ pub fn build(b: *std.Build) void {
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), manifest_file, capsule_img, &bundled);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
@@ -782,8 +951,8 @@ pub fn build(b: *std.Build) void {
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
|
||||
// The FHS zig-out (installed above) *is* the boot volume — no separate ESP to
|
||||
// assemble. QEMU presents it to the guest as a FAT drive below.
|
||||
// The guest boots the self-contained FAT image (attached as USB storage below),
|
||||
// not the installed FHS zig-out — see the run step's drive/device flags.
|
||||
|
||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||
@@ -848,6 +1017,53 @@ pub fn build(b: *std.Build) void {
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// --- run-x86-64-gpu: the same boot plus a virtio-gpu adapter ---
|
||||
// The VGA device still supplies the boot (GOP) framebuffer the compositor starts
|
||||
// on; the virtio-gpu function is discovered by the device-manager stack, its
|
||||
// driver announces a shared scanout, and the compositor upgrades off the GOP
|
||||
// floor to fenced, tear-free native presents (docs/display-v2.md).
|
||||
// This is the interactive twin of the `display-native` test case, and 512M
|
||||
// matches it (the whole driver stack + the compositor's surfaces at once).
|
||||
// QEMU shows one head per adapter: pick the virtio-gpu head in the View menu
|
||||
// to watch the native output.
|
||||
const run_gpu = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"512M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_gpu.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
"-device",
|
||||
"virtio-gpu-pci",
|
||||
});
|
||||
const gpu_serial_log = b.fmt("{s}/run-x86-64-gpu-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_gpu.addArgs(&.{ "-serial", b.fmt("file:{s}", .{gpu_serial_log}) });
|
||||
run_gpu.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_gpu_step = b.step("run-x86-64-gpu", "Boot in QEMU with a virtio-gpu adapter: the compositor upgrades to fenced (tear-free) native presents; watch the virtio-gpu head in QEMU's View menu");
|
||||
run_gpu_step.dependOn(&run_gpu.step);
|
||||
|
||||
// const run_cmd = b.addRunArtifact(exe);
|
||||
// const run_step = b.step("run", "Run the app");
|
||||
// run_step.dependOn(&run_cmd.step);
|
||||
@@ -865,24 +1081,25 @@ pub fn build(b: *std.Build) void {
|
||||
for ([_][]const u8{
|
||||
"system/boot-handoff.zig",
|
||||
"system/abi.zig",
|
||||
"system/devices/device-abi.zig",
|
||||
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||
"system/devices/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
"system/initial-ramdisk.zig", // v2 path-named entries: find/basename/magic
|
||||
"library/device/model/device-abi.zig",
|
||||
"library/device/pci/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"library/device/acpi/acpi-ids.zig", // _HID name decoding
|
||||
"library/device/acpi/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||
"library/device/usb/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"library/device/usb/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/device/registry/device-registry.zig", // /etc/devices.csv parse + most-specific driver match
|
||||
"library/device/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||
"system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||
"system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||
"system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian)
|
||||
"system/services/vfs/path.zig", // mount-prefix path matching
|
||||
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"library/protocol/vfs/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||
"system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile
|
||||
"system/services/display/protocol.zig", // pack(): native pixel encoding per format
|
||||
"library/protocol/display/display-protocol.zig", // pack(): native pixel encoding per format
|
||||
"system/drivers/virtio-gpu/virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||
"system/drivers/virtio-gpu/virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||
}) |root| {
|
||||
@@ -911,12 +1128,27 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
|
||||
// The tagged kernel log ring: append/wrap/reclaim/sequence-gap behavior over
|
||||
// a RAM buffer. Needs the `abi` module (record header layout), so it doesn't
|
||||
// fit the plain loop above.
|
||||
const log_ring_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/log-ring.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(log_ring_tests).step);
|
||||
|
||||
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||
// loop above.
|
||||
const time_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/time.zig"),
|
||||
.root_source_file = b.path("library/kernel/time.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
@@ -932,7 +1164,7 @@ pub fn build(b: *std.Build) void {
|
||||
// system.zig (syscall wrappers), which needs the `abi` module.
|
||||
const thread_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/thread.zig"),
|
||||
.root_source_file = b.path("library/kernel/thread.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
|
||||
+140
-93
@@ -3,89 +3,102 @@
|
||||
Notes on how danos boots and draws, written to explain the *why* behind the code
|
||||
rather than restate it. Roughly in the order things happen at runtime:
|
||||
|
||||
1. **[efi.md](efi.md) — EFI / the boot process.** How UEFI firmware finds and
|
||||
1. **[efi.md](os-development/efi.md) — EFI / the boot process.** How UEFI firmware finds and
|
||||
runs the bootloader, what the loader gathers before `ExitBootServices`, how it
|
||||
loads the kernel ELF, and the ABI contract for the jump into the kernel. Start
|
||||
here.
|
||||
2. **[gop.md](gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics
|
||||
2. **[system-image.md](os-development/system-image.md) — system.img, the boot capsule.** The
|
||||
bundled user binaries packed into one file in the initial-ramdisk wire
|
||||
format, because one open + one sequential read is the only file I/O shape
|
||||
firmware is fast at. The trivial container format, the three artifacts one
|
||||
build list derives (tree, manifest, capsule), the loader's three-strategy
|
||||
fallback chain, and the capsule's kernel-side life as both the spawn table
|
||||
and the read-only `/system` mount.
|
||||
3. **[gop.md](os-development/gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics
|
||||
modes (unlike fixed VGA modes), how we detect the monitor's native resolution
|
||||
from EDID and switch to it, and the pixel formats we accept or reject.
|
||||
3. **[framebuffer.md](framebuffer.md) — the framebuffer.** What the linear
|
||||
4. **[framebuffer.md](os-development/framebuffer.md) — the framebuffer.** What the linear
|
||||
framebuffer the loader hands over actually is, and what **pitch** (stride)
|
||||
means versus width — the detail you have to get right to avoid a skewed image.
|
||||
4. **[memory-map.md](memory-map.md) — the memory map.** How the loader learns what
|
||||
5. **[memory-map.md](os-development/memory-map.md) — the memory map.** How the loader learns what
|
||||
physical RAM exists and hands it to the kernel in danos's own neutral format,
|
||||
rather than leaking UEFI's memory descriptors across the boundary.
|
||||
5. **[frame-allocator.md](frame-allocator.md) — the physical frame allocator.** The
|
||||
6. **[frame-allocator.md](os-development/frame-allocator.md) — the physical frame allocator.** The
|
||||
bitmap allocator that hands out and reclaims 4 KiB physical frames from that
|
||||
map — the primitive page tables and the heap are built on.
|
||||
6. **[interrupts.md](interrupts.md) — interrupts and exceptions.** The GDT, IDT and
|
||||
7. **[interrupts.md](os-development/interrupts.md) — interrupts and exceptions.** The GDT, IDT and
|
||||
TSS, the exception stubs, and the handler that reports a CPU fault in red instead
|
||||
of letting it triple-fault into a silent reset.
|
||||
7. **[paging.md](paging.md) — the kernel's page tables.** Building our own 4-level
|
||||
8. **[paging.md](os-development/paging.md) — the kernel's page tables.** Building our own 4-level
|
||||
page tables, identity-mapping the low 4 GiB, and switching CR3 off the firmware's
|
||||
tables onto ours.
|
||||
8. **[device-interrupts.md](device-interrupts.md) — device interrupts.** The Local
|
||||
9. **[device-interrupts.md](device-driver-development/device-interrupts.md) — device interrupts.** The Local
|
||||
APIC and its timer — the kernel's first interrupt that is *handled and returned
|
||||
from*, giving it a heartbeat.
|
||||
9. **[heap.md](heap.md) — the kernel heap.** A growable free-list allocator built on
|
||||
10. **[heap.md](os-development/heap.md) — the kernel heap.** A growable free-list allocator built on
|
||||
the VMM, exposed as a `std.mem.Allocator` so std containers work — dynamic
|
||||
allocation for the kernel.
|
||||
10. **[scheduling.md](scheduling.md) — the scheduler.** Fixed-priority preemptive
|
||||
11. **[scheduling.md](os-development/scheduling.md) — the scheduler.** Fixed-priority preemptive
|
||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||
blocking (sleep, wait queues) — the leap to a running system.
|
||||
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||
12. **[ipc.md](device-driver-development/ipc.md) — inter-process communication.** Bounded blocking
|
||||
message-passing channels, then synchronous call/reply between *processes* over
|
||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
13. **[syscall.md](os-development/syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](os-development/vdso.md) designs
|
||||
the public boundary that will hide them.
|
||||
13. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
14. **[vfs-protocol.md](file-system-development/vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||
the operation table, mount routing, and the append-only evolution rules — the
|
||||
first IPC protocol documented as public ABI.
|
||||
14. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
15. **[drivers.md](device-driver-development/drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
15. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
16. **[driver-model.md](device-driver-development/driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
16. **[process-management.md](process-management.md) — process management.** The
|
||||
17. **[usb-hub.md](device-driver-development/usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled
|
||||
*inside* the `usb-xhci-bus` driver rather than a separate hub class driver — a
|
||||
device behind a hub is reached by the **controller**, programmed with a route
|
||||
string in its slot context — plus the compound-hub reality (a USB 3.0 hub is
|
||||
physically two hubs) and detection via the hub's status-change interrupt endpoint.
|
||||
18. **[process-management.md](os-development/process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
17. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
19. **[process-lifecycle.md](os-development/process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||
`process` module interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
18. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
20. **[device-manager.md](device-driver-development/device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
[resilience.md](resilience.md)'s restart goal into increments.
|
||||
19. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
[resilience.md](os-development/resilience.md)'s restart goal into increments.
|
||||
21. **[input.md](device-driver-development/input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||
service layered on top.
|
||||
20. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||
22. **[display.md](device-driver-development/display.md) — the display service.** The display half of the GUI
|
||||
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||
that tear-free doesn't. Plan: [display-plan.md](display-plan.md). **v2** (complete) makes
|
||||
that tear-free doesn't. Plan: [display-plan.md](device-driver-development/display-plan.md). **v2** (complete) makes
|
||||
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||
[display-v2.md](display-v2.md), plan [display-v2-plan.md](display-v2-plan.md). Looking
|
||||
further out, two research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
|
||||
21. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
[display-v2.md](device-driver-development/display-v2.md), plan [display-v2-plan.md](device-driver-development/display-v2-plan.md). Looking
|
||||
further out, three research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](device-driver-development/nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere), [amd-gpus.md](device-driver-development/amd-gpus.md) (RX 6600 / RDNA2), and [intel-igpu.md](device-driver-development/intel-igpu.md)
|
||||
(Intel iGPU).
|
||||
23. **[halting.md](os-development/halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||
|
||||
Start with the north star:
|
||||
@@ -95,28 +108,28 @@ Start with the north star:
|
||||
**resilience** (restartable components). Win condition: runs on the author's PC and
|
||||
both Raspberry Pis, ideally with a GUI. Real-time is an option to explore, not a
|
||||
requirement. The *why* that shapes everything below.
|
||||
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
||||
- **[resilience.md](os-development/resilience.md) — resilience.** A design note (not built yet) on
|
||||
fault isolation + live restart — the reincarnation-server + capability model that
|
||||
makes "if I break it, I can restart it" real. danos's core motivation.
|
||||
- **[zig-self-hosting.md](zig-self-hosting.md) — running Zig on danos.** A design note
|
||||
(not built yet) on making danos a real Zig target (`-target x86_64-danos`) and
|
||||
eventually running the compiler on it. The key realisation: Zig 0.16 reduces an OS
|
||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||
port to **one seam** (`std.os.danos`), so we build an `os` seam module (→ that seam) plus
|
||||
the thin `file-system` module, retire the `posix` shim, and follow a phased path to
|
||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||
- **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
`runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
- **[threading.md](os-development/threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
the `thread` module's `Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||
a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the
|
||||
native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](threading-plan.md).
|
||||
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
native type and not literal `std.Thread` (the [private ABI](os-development/syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](os-development/resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](os-development/threading-plan.md).
|
||||
- **[vdso.md](os-development/vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
|
||||
stable boundary without danos growing a dynamic linker. danos's public ABI = the
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](vfs-protocol.md) first).
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](file-system-development/vfs-protocol.md) first).
|
||||
|
||||
Cutting across all of these:
|
||||
|
||||
@@ -124,69 +137,69 @@ Cutting across all of these:
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[release-iso.md](release-iso.md) — the release ISO.** The flashable boot
|
||||
- **[release-iso.md](os-development/release-iso.md) — the release ISO.** The flashable boot
|
||||
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
|
||||
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
|
||||
flash to USB or a burner writes to disc — built by an in-repo pure-Python
|
||||
tool, like the FAT image itself.
|
||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||
- **[architecture.md](os-development/architecture.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
- **[arm.md](arm.md) — ARM targets.** The Raspberry Pi landscape the arch split is
|
||||
- **[arm.md](os-development/arm.md) — ARM targets.** The Raspberry Pi landscape the arch split is
|
||||
aiming at: `arm` (32-bit, Pi Zero W) vs `aarch64` (64-bit, Pi 3-5), UEFI vs
|
||||
device-tree boot, and what each layer needs.
|
||||
- **[discovery.md](discovery.md) — device discovery.** A design note on learning what
|
||||
- **[discovery.md](os-development/discovery.md) — device discovery.** A design note on learning what
|
||||
hardware exists via ACPI (x86) or device tree (ARM) behind one neutral device model —
|
||||
when to build it, and how to keep it architecture-agnostic.
|
||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
||||
- **[acpi.md](os-development/acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
how the loader captures the **RSDP**, hands its physical address across in `BootInformation`,
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||
- **[power.md](os-development/power.md) — the power service.** System power as a domain-named
|
||||
service: button/lid/battery events published to subscribers, and init's orderly
|
||||
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||
shutdown composing the [lifecycle](os-development/process-lifecycle.md) stop sequence with an ACPI
|
||||
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
- **[timers.md](os-development/timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-driver-development/device-interrupts.md).
|
||||
- **[smp.md](os-development/smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||
right choice depends on whether danos is chasing real-time or resilience.
|
||||
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
- **[sysv.md](os-development/sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||
QEMU and asserting on its serial output — reproducibly, and structured so the
|
||||
same tests run across architectures.
|
||||
- **[logging.md](logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
- **[logging.md](os-development/logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
0xE9 debugcon, file later) kept separate from the framebuffer display, plus the
|
||||
robustness path: optional framebuffer, POST-code checkpoints, and a persistent
|
||||
panic breadcrumb so the kernel survives — and can be diagnosed — with no output.
|
||||
|
||||
## How the pieces relate
|
||||
|
||||
The boot flow ties them together: UEFI runs the loader ([efi.md](efi.md)), which
|
||||
queries the **GOP** to pick a graphics mode ([gop.md](gop.md)), hands the kernel a
|
||||
**framebuffer** to draw into ([framebuffer.md](framebuffer.md)) and a **memory
|
||||
map** of physical RAM ([memory-map.md](memory-map.md)); the kernel turns that map
|
||||
into a **frame allocator** ([frame-allocator.md](frame-allocator.md)), installs
|
||||
its **descriptor tables** so CPU faults are caught ([interrupts.md](interrupts.md)),
|
||||
builds its own **page tables** and switches onto them ([paging.md](paging.md)),
|
||||
brings up the **heap** for dynamic allocation ([heap.md](heap.md)), starts the
|
||||
**scheduler** ([scheduling.md](scheduling.md)) and the **timer** that preempts it
|
||||
([device-interrupts.md](device-interrupts.md)) — with tasks blocking, sleeping and
|
||||
passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](halting.md)).
|
||||
The boot flow ties them together: UEFI runs the loader ([efi.md](os-development/efi.md)), which
|
||||
queries the **GOP** to pick a graphics mode ([gop.md](os-development/gop.md)), hands the kernel a
|
||||
**framebuffer** to draw into ([framebuffer.md](os-development/framebuffer.md)) and a **memory
|
||||
map** of physical RAM ([memory-map.md](os-development/memory-map.md)); the kernel turns that map
|
||||
into a **frame allocator** ([frame-allocator.md](os-development/frame-allocator.md)), installs
|
||||
its **descriptor tables** so CPU faults are caught ([interrupts.md](os-development/interrupts.md)),
|
||||
builds its own **page tables** and switches onto them ([paging.md](os-development/paging.md)),
|
||||
brings up the **heap** for dynamic allocation ([heap.md](os-development/heap.md)), starts the
|
||||
**scheduler** ([scheduling.md](os-development/scheduling.md)) and the **timer** that preempts it
|
||||
([device-interrupts.md](device-driver-development/device-interrupts.md)) — with tasks blocking, sleeping and
|
||||
passing messages over **[IPC](device-driver-development/ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [architecture](os-development/architecture.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](os-development/halting.md)).
|
||||
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](os-development/discovery.md),
|
||||
[acpi.md](os-development/acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](os-development/syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](device-driver-development/ipc.md)), and a **[driver](device-driver-development/drivers.md)** claims
|
||||
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
|
||||
@@ -195,7 +208,7 @@ the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
|
||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md)):
|
||||
what you see under `system/` in the source is what a running danos represents under
|
||||
`/system`.
|
||||
|
||||
@@ -207,10 +220,11 @@ addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||
| `test/system/services/vfs-test/vfs-test.zig` | `test/system/services/vfs-test` → `/test/system/services/vfs-test` |
|
||||
| `library/device/pci/pci.zig` | `library/device/pci` (the `pci` module) |
|
||||
|
||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||
`init/init.zig`, beside it `vfs/vfs-test.zig`, `vfs/protocol.zig`, and so on. When
|
||||
`fat/fat.zig`, beside it `fat/engine.zig`, `fat/on-disk.zig`, and so on. When
|
||||
**addressed or installed**, that collapses to the single canonical path: the `init`
|
||||
binary installs to `/system/services/init` (a file at that path), not
|
||||
`/system/services/init/init`. The repeated leaf exists only in source; the directory is
|
||||
@@ -221,28 +235,58 @@ A sub-project's extra files are reached through the module, never as separate pa
|
||||
```
|
||||
system/ → /system danos's own internals (the self-representation)
|
||||
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||
abi.zig the private kernel↔runtime syscall ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig shared contracts
|
||||
kernel/ IPC, memory, scheduling, the private syscall dispatch
|
||||
abi.zig the private kernel↔userspace syscall ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig shared contracts
|
||||
kernel/ IPC, memory, scheduling, the VFS root, the private syscall dispatch
|
||||
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||
devices/ the device model /system/devices reflects (+ aml/)
|
||||
device-abi.zig the device wire types (the `device-abi` module)
|
||||
drivers/ hpet/ bus/ one sub-project per driver → /system/drivers
|
||||
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||
vfs.zig, vfs-test.zig, protocol.zig)
|
||||
devices-broker.zig the syscall-facing device table
|
||||
platform.zig acpi.zig fdt.zig device-model.zig firmware discovery + the kernel's
|
||||
device model — the implementation of what /system/devices reflects
|
||||
drivers/ pci-bus/ ps2-bus/ usb-xhci-bus/ one sub-project per driver → /system/drivers
|
||||
services/ init/ fat/ device-manager/ system servers → /system/services (fat/ holds
|
||||
fat.zig, engine.zig, on-disk.zig)
|
||||
library/ → /lib libraries, one sub-directory each
|
||||
runtime/ the danos-native runtime + file API (fs) — the stable application ABI
|
||||
kernel/ the danos-native system library (kernel32-style): the syscall
|
||||
surface split by concern — ipc, memory (heap/dma/shared-memory),
|
||||
process, time, logging, file-system, thread, service, plus the
|
||||
system-call stubs and the start/root entry shim
|
||||
device/ device code by domain — mmio/ model/ pci/ usb/ acpi/ driver/
|
||||
block/ — each a shareable data module (device-abi, pci-class,
|
||||
usb-abi/ids, acpi-ids) plus a logic module (mmio, pci, usb, aml,
|
||||
driver — the device-access + device-manager-hello client)
|
||||
client/ userspace service clients (display, input) — a program's view of
|
||||
a service, layered over that service's protocol
|
||||
protocol/ driver↔service wire contracts (vfs block display scanout input
|
||||
power device-manager usb-transfer), one module per directory
|
||||
boot/ → /boot the loaders
|
||||
tools/ test/ host-side build + QEMU test harness
|
||||
test/ → /test the test tree: the QEMU harness (qemu_test.py, host-side)
|
||||
system/services/ beside the on-image test fixtures — vfs-test/ thread-test/
|
||||
crash-test/ … — whose repo path IS their boot-volume path
|
||||
(/test/system/services/<name>)
|
||||
tools/ host-side build scripts
|
||||
```
|
||||
|
||||
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the runtime's
|
||||
file API (`runtime.fs`) imports by name. `usb`/`block` drivers expose their protocols the
|
||||
same way.
|
||||
**Wire protocols live in `library/protocol/`**, one module per directory
|
||||
(`library/protocol/vfs/vfs-protocol.zig` is the `vfs-protocol` module), imported by module
|
||||
name. A protocol is the seam between a low-level driver and the higher-level service it
|
||||
serves — block ↔ the filesystem, a scanout driver ↔ the compositor — so both sides depend
|
||||
on the contract, not on each other, and the contract belongs to neither sub-project. A
|
||||
client module may *wrap* one for application convenience (the `file-system` module over
|
||||
`vfs-protocol`, and the `block`, `display`, `input` clients over theirs), but the protocol
|
||||
module is the boundary — a client re-exports no protocol, it imports it by name. A driver's
|
||||
private wire to its *hardware* (virtio-gpu's command set) is not a service seam and stays a
|
||||
driver-private file, beside the transport that reaches the same device.
|
||||
|
||||
**Device code lives in `library/device/<domain>/`**, grouped by what it is about (pci, usb,
|
||||
acpi, and the cross-cutting device model) and split by dependency weight: a data module of
|
||||
enums and wire types that is `std`-only and cheap for anyone to import, and a logic module
|
||||
that needs `mmio` or IPC. This is what keeps the microkernel out of device business — it
|
||||
imports exactly one `library/` module, `device-abi` (the descriptor types its broker
|
||||
marshals across the syscall boundary), and nothing with logic or a taxonomy in it. That
|
||||
lone pure-data import is the only edge from `system/kernel/` into `library/`.
|
||||
|
||||
There is **no POSIX/C compatibility layer today**: danos programs do file I/O through the
|
||||
danos-native `runtime.fs` (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||
danos-native `file-system` module (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||
(`library/posix/`) was retired as premature — the real POSIX/C surface will come later
|
||||
from the `std.os.danos` seam (and, eventually, musl) when danos becomes a Zig target (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). When it does, the foreign-ABI naming
|
||||
@@ -254,9 +298,9 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
|------|------|
|
||||
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||
| Loader↔kernel handoff (`BootInfo`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||
| Private kernel↔runtime syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the runtime speaks it, not apps | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||
| Loader↔kernel handoff (`BootInformation`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||
| Private kernel↔userspace syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the system library speaks it, not apps | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `library/device/model/device-abi.zig` |
|
||||
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||
@@ -264,14 +308,17 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||
| VFS root: mount table + kernel-served nodes (`fs_resolve`/`fs_node`); wire protocol in `library/protocol/vfs/vfs-protocol.zig` | `system/kernel/vfs.zig` |
|
||||
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/kernel/platform.zig` |
|
||||
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||
| danos-native system library (kernel32-style): the syscall surface by concern — `ipc`, `memory`, `process`, `time`, `logging`, `file-system`, `thread`, `service` — the stable application ABI | `library/kernel/` |
|
||||
| Service clients (a program's view of a service) and device clients | `library/client/` (display, input), `library/device/driver` |
|
||||
| System services (init, the `fat` filesystem, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| On-image test fixtures for the QEMU cases (`vfs-test`, `crash-test`, `thread-test`, …) → `/test/system/services` | `test/system/services/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
-105
@@ -1,105 +0,0 @@
|
||||
# Architecture split
|
||||
|
||||
danos targets x86_64 today, but is meant to grow onto other systems later — a
|
||||
Raspberry Pi, say, which is AArch64 and has no UEFI. To keep that possible without
|
||||
a rewrite, CPU-specific kernel code lives behind a boundary: the generic kernel
|
||||
never names an architecture, and each architecture plugs in behind it.
|
||||
|
||||
## The seam is a build-time module named `arch`
|
||||
|
||||
The mechanism is deliberately boring — no vtables, no function-pointer tables, no
|
||||
runtime dispatch. `build.zig` exposes one architecture's code as a module called
|
||||
`arch`:
|
||||
|
||||
```zig
|
||||
const arch_mod = b.addModule("arch", .{
|
||||
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||
});
|
||||
```
|
||||
|
||||
and the generic kernel imports it by that name:
|
||||
|
||||
```zig
|
||||
const arch = @import("arch");
|
||||
// ...
|
||||
arch.halt(); // never says "x86_64"
|
||||
```
|
||||
|
||||
Adding a second architecture is then a build-time choice: create
|
||||
`system/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
||||
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
||||
boundary _is_ the architecture interface** — when a new arch is missing a function
|
||||
the generic kernel calls, the build fails and names exactly what's missing.
|
||||
|
||||
## What's arch-specific vs generic
|
||||
|
||||
The split follows a simple test: does it name a CPU instruction, a hardware
|
||||
register, or a memory-management structure? If so, it's arch-specific.
|
||||
|
||||
| Arch-specific — `system/kernel/architecture/x86_64/` | Generic — kernel core |
|
||||
|---|---|
|
||||
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
||||
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
||||
| (future) interrupt controller, MMU setup | `root.zig` — the neutral handoff contract |
|
||||
|
||||
Notice the framebuffer console is *generic*: it just writes pixels into whatever
|
||||
framebuffer it's handed, so it needs no per-arch version. Most of the kernel
|
||||
should end up on the generic side; the arch module stays small.
|
||||
|
||||
## Two axes, kept separate
|
||||
|
||||
There are really two independent questions, and it's worth not conflating them:
|
||||
|
||||
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
||||
`system/kernel/arch/<cpu>/`.
|
||||
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
||||
*separately*, because loaders are their own binaries. `boot/efi.zig` builds
|
||||
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
||||
separate loader at all — the firmware jumps straight into the kernel with a
|
||||
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
||||
Either path converges on the same neutral [`BootInfo`](memory-map.md).
|
||||
|
||||
## Current x86_64 contents
|
||||
|
||||
- **`system/kernel/architecture/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
||||
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
||||
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
||||
trap frame.
|
||||
- **`system/kernel/architecture/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
||||
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables (see
|
||||
[paging.md](paging.md)).
|
||||
- **`system/kernel/architecture/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
||||
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
||||
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
||||
port-I/O + MSR primitives.
|
||||
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
||||
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
||||
express them.
|
||||
- **`system/kernel/architecture/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||
address, one PT_LOAD per permission set).
|
||||
|
||||
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
||||
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
||||
x86_64 [SysV](sysv.md), via the shared `system.kernel_abi`), but it's three lines
|
||||
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
||||
up a stack and reading a device-tree pointer from a register — the entry work will
|
||||
be substantial and per-arch, and *that* is when we extract an entry interface into
|
||||
the arch modules.
|
||||
|
||||
## The discipline
|
||||
|
||||
The thing that makes this help rather than hurt: **only extract what's provably
|
||||
architecture-specific, and let the interface emerge with the second
|
||||
implementation.** With a single architecture you're guessing at the seam, and a
|
||||
wrong guess encoded as elaborate abstraction is expensive to undo. So:
|
||||
|
||||
- Move code into `arch/` only when it genuinely names CPU-specific machinery.
|
||||
- Grow the `arch` surface one function at a time, as steps need it.
|
||||
- Don't pre-design the interrupt or paging interfaces before writing them.
|
||||
|
||||
Directory hygiene is cheap and reversible; premature abstraction is neither. When
|
||||
arch #2 lands and something doesn't fit, reshaping a few hundred lines is nothing —
|
||||
unwinding an abstraction empire is not.
|
||||
+12
-10
@@ -5,8 +5,8 @@ Conventions for danos source. The overriding one, from which most of the rest fo
|
||||
> **Names are spelled out in full. An identifier is not abbreviated unless the
|
||||
> abbreviation is an acronym.**
|
||||
|
||||
`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len`
|
||||
is a Zig idiom — see the exceptions). `devices_broker`, not `devices_broker`. `scheduler`, not
|
||||
`interruptDispatch`, not `intDisp`. `message_len`, not `msg_len` (`msg` expands, `len`
|
||||
is a Zig idiom — see the exceptions). `devices_broker`, not `dev_broker`. `scheduler`, not
|
||||
`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a
|
||||
cryptic one is paid every time the code is read, by everyone who reads it. In a
|
||||
microkernel whose whole argument is that a human can hold each piece in their head,
|
||||
@@ -67,7 +67,7 @@ Three, and only three.
|
||||
|
||||
**This exception is scoped to a file that *is* a foreign ABI, and nothing else.**
|
||||
danos has no such file today: the old `library/posix/` compatibility shim was retired
|
||||
once its callers moved to the danos-native `runtime.fs`, since a hand-rolled POSIX
|
||||
once its callers moved to the danos-native `file_system`, since a hand-rolled POSIX
|
||||
layer is premature until danos actually needs it (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). The exception will apply again to the
|
||||
`std.os.danos` seam when danos becomes a real Zig target — that module *is* the C-ABI
|
||||
@@ -97,9 +97,10 @@ Three, and only three.
|
||||
|
||||
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||
redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory
|
||||
lives in `system/drivers/`, a service in `system/services/`, and a test fixture in
|
||||
`test/system/services/` (the repo path *is* its path on the boot volume), so the
|
||||
*location* already says what it is. Encoding the role in the name too (`busd`, `fatd`) is
|
||||
redundant — the program is just `ps2-bus`, `fat`. Don't put in a name what its directory
|
||||
already tells you.
|
||||
|
||||
## A note on collisions
|
||||
@@ -135,15 +136,16 @@ Within those spelling rules, follow Zig's own conventions:
|
||||
`notify_badge_bit`.
|
||||
|
||||
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
`device-model.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||
|
||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||
resolving away. See the repository-layout section of [README.md](README.md).
|
||||
`pci/pci.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/device/pci`,
|
||||
`test/system/services/vfs-test`), with the repeated leaf resolving away. See the
|
||||
repository-layout section of [README.md](README.md).
|
||||
|
||||
## Named values, not magic numbers
|
||||
|
||||
|
||||
@@ -0,0 +1,344 @@
|
||||
# Native AMD GPU support — feasibility and roadmap
|
||||
|
||||
**Status: research snapshot, not implemented.** This records what a *minimal, display-only*
|
||||
native driver for a real discrete AMD GPU — specifically an **RX 6600-class card (Navi 23,
|
||||
RDNA2, DCN 3.0.2)**, the market analog of the RTX 3060 — would take, and how it slots into
|
||||
danos's pluggable scanout architecture. It is a survey of primary sources (the Linux
|
||||
[amdgpu Display Core](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/amd/display)
|
||||
driver and its [kernel documentation](https://docs.kernel.org/gpu/amdgpu/display/index.html),
|
||||
the AtomBIOS interpreter in
|
||||
[drivers/gpu/drm/amd](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/amd),
|
||||
linux-firmware's `LICENSE.amdgpu`, and Haiku's
|
||||
[radeon_hd](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/radeon_hd)),
|
||||
not an implementation. It completes the trilogy with [nvidia-gpus.md](nvidia-gpus.md) and
|
||||
[intel-igpu.md](intel-igpu.md) and should be read against both — AMD lands *between* them:
|
||||
NVIDIA-class discrete-card mechanics, but Intel-class (better, in one way) reference material.
|
||||
|
||||
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the
|
||||
v2 model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- **AMD's decisive advantage is that the vendor's own display driver is the register manual, and
|
||||
it's MIT-licensed.** The entire Display Core (DC) — hardware sequencer, per-block code for
|
||||
OTG/OPTC, HUBP, DPP, MPC, DIO/link encoders, plus the `asic_reg` register headers — ships in
|
||||
the Linux tree under MIT/X11, deliberately written OS-agnostic because AMD shares it across
|
||||
operating systems. You can study it, port it, even copy from it into a danos driver without
|
||||
license contamination. NVIDIA has no analog (nouveau is GPL); Intel has PRM prose but you
|
||||
still write the code yourself.
|
||||
- **The firmware wall is one small blob, not a GSP.** The only display-side firmware the Linux
|
||||
driver hard-requires is **DMCUB** (the display microcontroller), and only on **DCN 2.1
|
||||
through 4.x** — which includes Navi 23. It is redistributable from linux-firmware, and it is
|
||||
a display helper, not a full-card resource manager: on DCN 3.0.x, hardware init
|
||||
(`dcn30_init_hw`) is **host-driven direct register programming** — the only DMUB call in it
|
||||
is a capability query. All DCE generations, DCN 1.0 (Raven), and DCN 2.0 (Navi 10/12/14) run
|
||||
display with **no display firmware at all**.
|
||||
- **Whether the *silicon* (vs. the Linux driver) needs DMCUB for a bare GOP-inheriting modeset
|
||||
is unproven** — Linux fails init with `-EINVAL` if the blob is missing on a DMUB ASIC, but
|
||||
what it's *used for* at minimum scope (vs. PSR/ABM/offloaded DP link training) isn't
|
||||
documented. The safe plan ships the blob; it's legally and practically cheap to do so.
|
||||
- **Programming model is direct MMIO, not channel DMA.** DCN mode-set is ordered register-write
|
||||
sequences (the DC "hardware sequencer") against named, header-documented registers — no
|
||||
pushbuffers, no method streams, no RAMHT, no supervisor-interrupt handshake. This deletes the
|
||||
hardest structural layer of the NVIDIA path.
|
||||
- **Scanout is VRAM-only on discrete cards** — the claim that DCN can scan out of GTT/system
|
||||
memory was checked and *refuted* for dGPUs (Linux allows GTT scanout only on select APUs). So
|
||||
a small VRAM allocator + BAR CPU mapping is required, same as NVIDIA. Pitch-linear surfaces
|
||||
are supported; no DCC/tiling needed.
|
||||
- **danos's GOP boot helps here too, with a caveat.** DC explicitly models taking over a
|
||||
VBIOS/GOP-lit pipe (`dc_validate_boot_timing` reads back live DIG/OTG/pixel-clock state), so
|
||||
"repoint the surface on the running pipe" is demonstrably hardware-feasible — but Linux's
|
||||
seamless-boot path is **eDP-only and default-off on discrete cards**, so plan on a full
|
||||
self-owned modeset (including DP retrain) right after first light rather than living on the
|
||||
inherited link.
|
||||
- **AMD has real non-Linux prior art — but only for the old hardware.** Haiku's MIT `radeon_hd`
|
||||
mode-sets by executing VBIOS **AtomBIOS command tables** through AMD's own MIT interpreter;
|
||||
its compiled-in ceiling is **DCE 8.5 (Hawaii, ~2013)** — every Polaris/Vega/Navi entry sits
|
||||
in a `#if 0` block. There is zero non-Linux DCN precedent; a danos DCN driver would be first.
|
||||
- **Effort tier ≈ high-3 to 4** for a native DCN 3.0.x display-only driver on Navi 23 — the raw
|
||||
register surface is GA106-class (tier 4), but the MIT vendor reference, the one-blob firmware
|
||||
wall, and the absence of channel-DMA plumbing pull real risk out. The AtomBIOS-interpreter
|
||||
route is tier ≈ 3 but dead-ends at pre-2016 silicon.
|
||||
- **Recommendation:** the same sober conclusion as the other two docs — GOP already gives
|
||||
native-res scanout for zero code — but if danos ever does drive real discrete silicon
|
||||
natively, **an RDNA2 card is the best target of the three**: modern, mainstream, in-warranty
|
||||
hardware with a legally clean, vendor-authored reference. That combination exists nowhere
|
||||
else.
|
||||
|
||||
## The firmware wall (a fence, next to NVIDIA's wall)
|
||||
|
||||
AMD GPUs carry a zoo of firmware: PSP (security processor), SMU (power/clock management), CP/RLC
|
||||
(graphics), SDMA, VCN (media) — and, on the display side, DMCU (legacy) then **DMCUB**
|
||||
("Display Micro-Controller Unit, version B"), a per-generation blob in linux-firmware
|
||||
(`navi23_dmcub.bin` etc.). The display-only question is: which of these does a scanout driver
|
||||
actually need?
|
||||
|
||||
The Linux answer is precise and readable in
|
||||
[`amdgpu_dm.c`](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c):
|
||||
`dm_init_microcode()` switches on the display IP version — **DCN 2.1 (Renoir) through DCN
|
||||
3.0.x / 3.1.x / 3.2 / 3.5 / 4.x** request a DMCUB blob as `AMDGPU_UCODE_REQUIRED`, and
|
||||
`dm_dmub_hw_init()` fails driver init with `-EINVAL` if it's absent. Everything earlier — **all
|
||||
of DCE (Southern Islands through Vega), DCN 1.0 (Raven), and DCN 2.0 (Navi 10/12/14)** — hits
|
||||
the `default:` case, `dmub_srv` stays NULL, and display runs with no display firmware at all
|
||||
([kernel display-manager doc](https://docs.kernel.org/6.2/gpu/amdgpu/display/display-manager.html)).
|
||||
|
||||
Two nuances survive verification:
|
||||
|
||||
- **The requirement is Linux-driver enforcement backed by real functional need, and it's
|
||||
version-sensitive.** AMD force-switched all Renoir ASICs to DMUB to fix a USB-C/resume bug
|
||||
(kernel commit `652de07addd2`, "with new dmub f/w dmcu is superseded"), which regressed users
|
||||
on old blobs and had to be patched with explicit `dmcub_fw_version` gating (`91adec9e0709`).
|
||||
What DMUB is *used for* varies with blob version. Ship a current blob.
|
||||
- **DMCUB is an architectural fixture, not a bolt-on** — the DCN hardware itself contains a DMU
|
||||
block housing the microcontroller
|
||||
([DCN overview](https://docs.kernel.org/gpu/amdgpu/display/dcn-overview.html)) — but it is
|
||||
**not a mediator of the programming model** on DCN 3.0: `dcn30_init_hw()` initializes clocks,
|
||||
disables power gating, and powers up link encoders via direct register writes; its sole DMUB
|
||||
interaction is `dc_dmub_srv_query_caps_cmd`. Firmware-*assisted* PHY/link bring-up appears
|
||||
from **DCN 3.1** onward — one more reason to target 3.0.x. Features like PSR and ABM are
|
||||
DMUB-offloaded on all generations; a minimal driver simply doesn't enable them.
|
||||
|
||||
**Contrast with NVIDIA's GSP:** the GSP is a full resource manager with a signed multi-stage
|
||||
boot chain and a firmware ABI that breaks every driver release. DMCUB is a display helper blob
|
||||
you copy onto the boot image once, load into a reserved buffer, and mostly ignore. There is no
|
||||
signature fuse-matching, no WPR carve-out, no RPC-only register access. The one genuinely open
|
||||
question — whether a GOP-inheriting minimal modeset could skip DMCUB entirely on DCN 3.0.2 —
|
||||
doesn't need answering, because shipping the blob costs nothing (see [Licensing](#licensing)).
|
||||
|
||||
**PSP/SMU remain the flagged risk.** Nothing display-only touches CP/RLC/SDMA (those gate the
|
||||
graphics rings, exactly like NVIDIA's PGRAPH — irrelevant here). But `dcn30_init_hw` calls into
|
||||
the clock manager, and on discrete cards the clock manager may message the SMU to change display
|
||||
clocks (DISPCLK/DPPCLK). Whether inherited GOP boot clocks suffice for a same-or-lower mode —
|
||||
avoiding SMU (and hence PSP firmware-load) entirely — is the largest unverified assumption in
|
||||
the milestone list below. The survey produced no confirmed claim either way.
|
||||
|
||||
## The display engine landscape
|
||||
|
||||
Two eras, one boundary that matters:
|
||||
|
||||
| Generation | Display IP | Cards | Display firmware | Route |
|
||||
|---|---|---|---|---|
|
||||
| GCN 1–4 (SI→Polaris) | DCE 6/8/10/11 | HD 7000 → RX 580 | none | AtomBIOS tables or direct DCE registers |
|
||||
| Vega / Raven | DCE 12 / DCN 1.0 | Vega 56/64, APUs | none | DC code (first DCN) |
|
||||
| Navi 1x (RDNA1) | DCN 2.0 | RX 5500–5700 | none | DC code |
|
||||
| Renoir APU | DCN 2.1 | 4000-series APUs | **DMCUB required** | DC code |
|
||||
| **Navi 2x (RDNA2)** | **DCN 3.0.x** | **RX 6600–6900** | **DMCUB required** | **DC code, host-driven init** |
|
||||
| RDNA3/RDNA4+ | DCN 3.1+/3.2/3.5/4.x | RX 7000/9000 | DMCUB required, fw-assisted PHY | DC code, more DMUB offload |
|
||||
|
||||
The best modern first-pixel target is **DCN 3.0.x**: it has the full MIT block stack from the
|
||||
June 2020 Sienna Cichlid patch series (207 patches, Linux 5.9; Navi 23 reuses the dcn30
|
||||
sequencer), host-driven hardware init, and sits *before* the DCN 3.1 shift toward
|
||||
firmware-assisted link management. Older DCE cards are even simpler (no firmware at all, plus
|
||||
the AtomBIOS escape hatch) but are 2013–2016 hardware; newer DCN 3.5/4.x pushes more into DMUB.
|
||||
|
||||
The DCN pipe, in one line each (the vocabulary the DC code speaks —
|
||||
[programming model](https://docs.kernel.org/next/gpu/amdgpu/display/programming-model-dcn.html)):
|
||||
**HUBP** fetches and unpacks the surface from memory (this is where the scanout address and
|
||||
pitch live), **DPP** scales/converts colors, **MPC** blends planes (bypassable for one plane),
|
||||
**OPP** packs output, **OTG/OPTC** generates raster timings (the CRTC), and the **DIO** block's
|
||||
DIG encoders + PHY drive the connector. Mode-set is the DC *hardware sequencer* walking these
|
||||
blocks with ordered register writes — plain MMIO with polling, no pushbuffer channels, no
|
||||
supervisor interrupts. Structurally this is Intel-shaped, not NVIDIA-shaped.
|
||||
|
||||
## Two routes: AtomBIOS interpreter vs. native DC-derived registers
|
||||
|
||||
**AtomBIOS** is AMD's VBIOS bytecode: every card's ROM carries *data tables* (connector
|
||||
topology, clock limits — the DCB equivalent) and *command tables* (`SetPixelClock`,
|
||||
`SetCRTC_Timing`, `EnableCRTC`, DIG encoder/transmitter control), executed by a small
|
||||
interpreter the driver embeds (`atom.c`, ~1.5k lines). The classic radeon driver and Haiku's
|
||||
`radeon_hd` mode-set this way: parse the tables, execute them, and the VBIOS does the
|
||||
register-level work for you — inherently per-board correct, since the tables come from the
|
||||
card's own ROM.
|
||||
|
||||
- **Where it's proven:** through DCE 8.5 (Haiku's ceiling, below) and in Linux's pre-DC code
|
||||
through Polaris (DCE 11.2). AMD's interpreter itself is MIT (Haiku ships AMD's own
|
||||
`atom.cpp`, "Copyright 2008 Advanced Micro Devices").
|
||||
- **Where it's unproven:** DCN. amdgpu's DC still *uses* AtomBIOS for init sub-steps
|
||||
(`bios_golden_init` in `dcn30_init_hw` executes host-interpreted tables) and reads the data
|
||||
tables for connector topology — but nobody drives a full DCN modeset from command tables, and
|
||||
whether RDNA2 VBIOSes still carry a complete modeset path or vestigial init-only tables is an
|
||||
open question no source answers. Do not bet on it.
|
||||
|
||||
**The native route** is: port the relevant slice of DC. Not wholesale — in-tree DC has
|
||||
accumulated Linux-isms (kernel-FPU guards around DML, the bandwidth-calculation library, which a
|
||||
single-plane fixed-mode driver can largely sidestep) — but the DC core is *designed* to be
|
||||
retargeted: the kernel docs state outright that DC "is shared with other OSes" and holds the
|
||||
OS-agnostic hardware programming behind a `dm_services` shim (register access, memory, delays,
|
||||
firmware loading). Dave Airlie initially rejected the DAL/DC merge in 2016 *because* it was
|
||||
AMD's cross-OS codebase — hostile-witness confirmation that this exact code runs outside Linux.
|
||||
Reimplement the shim in Zig, and the dcn30 sequences sit on top.
|
||||
|
||||
## The memory floor
|
||||
|
||||
Same shape as NVIDIA's, and the survey *hardened* one assumption:
|
||||
|
||||
- **VRAM-only scanout on discrete cards.** The documented DCN fetch path is VRAM → Data Fabric
|
||||
(SDP) → DCHUB → HUBP; the claim that display buffers can live in GTT/system memory was
|
||||
refuted for dGPUs in verification — Linux permits GTT scanout only on select APUs. The NVIDIA
|
||||
doc's "maybe sysmem ctxdma?" hope has a firm *no* here. Budget for a small VRAM allocator.
|
||||
- **Pitch-linear is fine.** HUBP programs a surface address + pitch; linear (untiled, no DCC)
|
||||
surfaces are first-class for scanout. No tiling math.
|
||||
- **CPU access via the VRAM BAR.** Compositing writes go through the PCI VRAM aperture;
|
||||
resizable BAR helps but isn't needed — one pitch-linear surface fits comfortably in a
|
||||
fixed 256 MB small-BAR window.
|
||||
- **No GPU VMM.** Display addresses are physical VRAM addresses programmed into HUBP; no page
|
||||
tables, no GEM/TTM, no eviction.
|
||||
|
||||
**Net:** (1) a contiguous aligned VRAM allocator, (2) a BAR CPU mapping, (3) a small reserved
|
||||
buffer for the DMCUB firmware regions. That's the whole memory story.
|
||||
|
||||
## Inheriting GOP state
|
||||
|
||||
danos's GOP boot pays off again, with sharper edges than on NVIDIA:
|
||||
|
||||
- **DC models pipe takeover explicitly.** `dc_validate_boot_timing()`
|
||||
([dc.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/amd/display/dc/core/dc.c))
|
||||
reads back *live* hardware — `is_dig_enabled` on the link encoder, OTG timing registers,
|
||||
pixel clock within tolerance — and keeps the VBIOS/GOP-lit pipe running until first flip.
|
||||
This is a vendor-blessed recipe for milestone 2 below: the exact register set that tells you
|
||||
which pipe is alive and how it's configured.
|
||||
- **But Linux's seamless path is eDP-only and APU-gated.** The code comment is blunt: "Support
|
||||
seamless boot on EDP displays only", and the enabling check requires an APU with DCN ≥ 3.0
|
||||
unless forced with `amdgpu.seamless=1`. On a discrete RX 6600 with DP/HDMI, Linux does a full
|
||||
modeset at takeover. Read that as a warning, not a prohibition: repointing HUBP at your own
|
||||
surface on the live pipe should still work (the readback code proves the state is
|
||||
inspectable), but plan the full self-owned modeset — **including DP retraining** — as the
|
||||
immediate next step, not a someday.
|
||||
- **No supervisor handshake exists to re-learn.** The NVIDIA doc's SV1/SV2/SV3 open question has
|
||||
no AMD counterpart; commit sequencing is ordered register writes + vblank/lock waits in the
|
||||
hwseq, all visible in MIT source.
|
||||
|
||||
## Licensing
|
||||
|
||||
The inverse of the NVIDIA situation, and the single strongest argument for AMD:
|
||||
|
||||
- **The reference code is MIT.** `amdgpu_dm.c` carries `SPDX-License-Identifier: MIT`;
|
||||
`dc/core/dc.c` and `dmub_srv.h` carry the full X11-style grant (use, copy, modify, merge,
|
||||
publish, distribute, sell). The `asic_reg` register headers ship under the same terms. One
|
||||
diligence note: SPDX tagging isn't uniform across the tree, so header-check each file before
|
||||
copying from it — but no GPL files are known inside `dc/`. Where nouveau forces a
|
||||
GPL-or-clean-room choice, here the *easy technical path and the permissive path are the same
|
||||
path*.
|
||||
- **Firmware redistribution is a solved problem.** linux-firmware's `LICENSE.amdgpu` grants
|
||||
anyone a royalty-free right to reproduce and distribute the blobs, binary-only, with the
|
||||
license text attached — no OSI-license gate like NVIDIA's, no AMD agreement needed. danos can
|
||||
ship `navi23_dmcub.bin` (and PSP/SMU blobs if ever needed) on its boot image today. The same
|
||||
license **prohibits reverse-engineering the blobs** — all programming knowledge must come
|
||||
from the MIT source, never from blob disassembly. (VBIOS images aren't in linux-firmware;
|
||||
they're read from the card's own ROM, as Haiku does.)
|
||||
- **Prose register docs are a DCE-era artifact.** AMD's classic X.Org-hosted PDFs cover the old
|
||||
families — and the famous `R6xx_3D_Registers.pdf` turns out to be 3D-only (verified: zero
|
||||
display content; the display material lives in the separate per-ASIC Register Reference
|
||||
Guides). For DCN there is **no prose display spec at all**: the MIT DC source plus the
|
||||
`asic_reg` headers *are* the register manual. Plan accordingly.
|
||||
|
||||
## Prior art
|
||||
|
||||
AMD, unlike NVIDIA, has genuine working non-Linux precedent — with a hard generational ceiling:
|
||||
|
||||
- **Haiku `radeon_hd`** (MIT, still in the tree): a real, shipping, from-scratch display driver
|
||||
that executes AtomBIOS command tables via AMD's own MIT interpreter. Verified ceiling:
|
||||
the last *enabled* device entry is **Hawaii (DCE 8.5, 0x67be)**; everything newer —
|
||||
Tonga/Fiji, Carrizo/Polaris, Vega/Raven, and every Navi/RDNA2 entry up to the RX 6900 XT —
|
||||
sits inside one `#if 0 /* disabled for R1/beta5 */` block under the comment "WARN: DCE
|
||||
versions below here get sketchy."
|
||||
- **AmigaOS/MorphOS RadeonHD drivers** (hdrlab): commercial non-Linux Radeon display drivers,
|
||||
again for the DCE era.
|
||||
- **FreeBSD** `drm-kmod`: a port of Linux amdgpu (DC and all), not independent prior art — but
|
||||
proof the DC codebase transplants.
|
||||
|
||||
**Nobody has driven DCN outside Linux-derived code.** A danos DCN 3.0.x driver would be a
|
||||
first — but a first with the vendor's MIT code as its map, which is a different proposition
|
||||
from nouveau-as-only-reference.
|
||||
|
||||
## Alternatives
|
||||
|
||||
| Option | What you get | The tradeoff |
|
||||
|---|---|---|
|
||||
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; no runtime mode change, no hardware vsync, no multihead |
|
||||
| **AtomBIOS interpreter on an old DCE card** | Proven end-to-end (Haiku); interpreter is small + MIT; board-correct by construction | 2013–2016 hardware ceiling; tier ≈ 3; teaches AtomBIOS, not modern DCN |
|
||||
| **Native DCN 3.0.x on RX 6600** (this doc) | Runtime modeset, vsync, multihead on modern silicon; MIT vendor reference; one redistributable blob | Tier ≈ high-3–4; SMU/clock question open; no non-Linux precedent |
|
||||
| **Port DC wholesale** (reimplement `dm_services`) | Vendor-maintained sequences verbatim; designed-for-porting seam | Big codebase to carry (DML, abstractions); Linux-isms to shear off; overkill for one plane |
|
||||
| **RDNA3+/DCN 3.5+** | Newer cards | More DMUB offload (fw-assisted PHY from DCN 3.1); strictly harder than 3.0.x for no display-only gain |
|
||||
|
||||
## "First light" milestones (native DCN 3.0.x path)
|
||||
|
||||
Framed as a danos `.scanout` service, inheriting the GOP-initialized display:
|
||||
|
||||
1. **PCI/BAR bring-up** — enumerate Navi 23, map the register BAR and the VRAM BAR via danos
|
||||
MMIO grants; prove the pipe is GOP-live by writing pixels into the *existing* GOP
|
||||
framebuffer through the VRAM BAR.
|
||||
2. **Read back the live pipe** — port the `dc_validate_boot_timing` register set: which OTG is
|
||||
running, its timings, which DIG/link encoder is enabled, current HUBP surface address/pitch.
|
||||
This is pure reads — zero risk, high information.
|
||||
3. **Repoint the surface** — allocate a danos-owned pitch-linear VRAM surface, program the HUBP
|
||||
surface address/pitch on the live pipe at vblank. First self-owned pixel with **no modeset,
|
||||
no firmware, no clock changes**.
|
||||
4. **DMCUB bring-up** — load `navi23_dmcub.bin` (redistributed per `LICENSE.amdgpu`) into its
|
||||
reserved regions, minimal `dmub_srv` init, verify the caps query answers.
|
||||
5. **Full owned modeset** — port the dcn30 hwseq slice: OTG timing programming, MPC bypass
|
||||
(single plane), DIG/PHY enable, **DP link retrain** (or start on HDMI to defer it, exactly
|
||||
as the NVIDIA doc advises). This is where the SMU/clock question lands — first attempt:
|
||||
reuse inherited boot clocks for a same-or-lower mode.
|
||||
6. **EDID** — AUX (DP) / DDC (HDMI) over the DCN AUX engine registers; parse and build the mode
|
||||
list; connector topology from the VBIOS AtomBIOS data tables.
|
||||
7. **Wire into the compositor** — `attach_scanout`, vsync from the vblank/pageflip interrupt,
|
||||
then multihead.
|
||||
|
||||
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with
|
||||
a working display (the resilience v2 already provides via re-attach).
|
||||
|
||||
## Reading list
|
||||
|
||||
**The DC core (MIT — the register manual for DCN):**
|
||||
- `drivers/gpu/drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c` — hardware init + the modeset
|
||||
sequencer for the target generation (host-driven; one DMUB caps query).
|
||||
- `dc/dcn30/` + `dc/dcn302/` blocks: `dcn30_hubp.c` (surface address/pitch — milestone 3),
|
||||
`dcn30_optc.c` (OTG timings), `dcn30_dio_link_encoder.c` (DIG/PHY), `dcn30_mpc.c` (bypass),
|
||||
`clk_mgr/dcn30/` (the SMU question, read before milestone 5).
|
||||
- `dc/core/dc.c` — `dc_validate_boot_timing()`: the GOP-takeover readback recipe.
|
||||
- `asic_reg/dcn/dcn_3_0_0_{offset,sh_mask}.h` — every register name and bitfield.
|
||||
- `dmub/` (`dmub_srv.h`, `src/dmub_dcn30.c`) — firmware regions + bring-up for milestone 4.
|
||||
|
||||
**The Linux glue (for logic, not porting):**
|
||||
[`amdgpu_dm.c`](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c)
|
||||
— `dm_init_microcode` / `dm_dmub_hw_init` (the firmware-wall switch), seamless-boot gating.
|
||||
|
||||
**Kernel docs (read first):**
|
||||
[DCN overview](https://docs.kernel.org/gpu/amdgpu/display/dcn-overview.html) (the block diagram
|
||||
+ VRAM→DF→DCHUB fetch path) ·
|
||||
[DC programming model](https://docs.kernel.org/next/gpu/amdgpu/display/programming-model-dcn.html)
|
||||
(dc_plane/dc_stream/dc_link objects, hwseq, block APIs) ·
|
||||
[display manager](https://docs.kernel.org/6.2/gpu/amdgpu/display/display-manager.html).
|
||||
|
||||
**AtomBIOS:** `drivers/gpu/drm/amd/amdgpu/atom.c` (the interpreter), `atombios.h` (table
|
||||
formats), [osdev AMD AtomBIOS](https://wiki.osdev.org/AMD_Atombios) (hobby-OS orientation),
|
||||
Haiku [`radeon_hd`](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/radeon_hd)
|
||||
(a complete worked example, MIT, through DCE 8.5).
|
||||
|
||||
**Licensing:** linux-firmware
|
||||
[`LICENSE.amdgpu`](https://github.com/endlessm/linux-firmware/blob/master/LICENSE.amdgpu);
|
||||
DCE-era prose specs at [x.org/docs/AMD](https://www.x.org/docs/AMD/) (Register Reference
|
||||
Guides — display; note `R6xx_3D_Registers.pdf` is 3D-only).
|
||||
|
||||
## Open questions (unresolved by the survey)
|
||||
|
||||
- **Does DCN 3.0.2 silicon need DMCUB for a bare inherit-and-modeset path**, or only for
|
||||
PSR/ABM/offloaded features? (Moot if the blob is shipped regardless — but it decides whether
|
||||
milestone 3 can precede milestone 4.)
|
||||
- **Can a display-only driver avoid PSP and SMU entirely** by inheriting GOP boot clocks — what
|
||||
does the dcn30 clock manager actually require from SMU messaging on Navi 23 for a
|
||||
same-or-lower mode? *The largest open risk in the plan.*
|
||||
- **Do RDNA2 VBIOS command tables still carry a complete modeset path**, or are they vestigial
|
||||
init-only tables? (Would open a Haiku-style route on modern cards; no source answers it.)
|
||||
- **Is the DCN 3.0 AUX/DDC engine and DP retrain fully host-drivable without DMUB**, as
|
||||
`dcn30_hwseq` implies?
|
||||
- Exact HUBP surface alignment/pitch constraints for linear scanout on Navi 23 (in the headers;
|
||||
not captured verbatim in the survey).
|
||||
|
||||
---
|
||||
|
||||
*Research snapshot (2026-07); findings pinned to Linux master and Haiku master as of the survey
|
||||
date. DMUB coverage only grows with new DCN generations — re-verify the firmware-wall switch in
|
||||
`amdgpu_dm.c` against current source before building.*
|
||||
@@ -1,6 +1,6 @@
|
||||
# Device interrupts
|
||||
|
||||
CPU exceptions ([interrupts.md](interrupts.md)) are the kernel reacting to its own
|
||||
CPU exceptions ([interrupts.md](../os-development/interrupts.md)) are the kernel reacting to its own
|
||||
mistakes. **Device interrupts** are the opposite: hardware asking for attention —
|
||||
a timer firing, a key pressed, a packet arriving. They share the IDT, but differ
|
||||
in one fundamental way: an exception here is terminal (we report and halt), while a
|
||||
@@ -10,7 +10,7 @@ back — the same mechanism a scheduler will later use to preempt tasks.
|
||||
|
||||
The first device we bring up is the **timer**, because it's the simplest: it lives
|
||||
entirely on the CPU's local interrupt controller, needing no external routing.
|
||||
It's all x86_64-specific, behind the [arch](arch.md) boundary.
|
||||
It's all x86_64-specific, behind the [architecture](../os-development/architecture.md) boundary.
|
||||
|
||||
## The APIC, not the PIC
|
||||
|
||||
@@ -40,7 +40,7 @@ count that becomes the reload value. From then on it fires vector 32 repeatedly,
|
||||
its own, forever.
|
||||
|
||||
The reload count isn't picked arbitrarily — it's **calibrated to real time**,
|
||||
which the [real-time](vision.md) scheduling guarantees depend on. Since the LAPIC
|
||||
which the [real-time](../vision.md) scheduling guarantees depend on. Since the LAPIC
|
||||
timer's raw rate is bus-clock dependent and unknown up front, `calibrate` runs the
|
||||
LAPIC timer one-shot from its maximum count while a **reference clock** counts out a
|
||||
known 10 ms, then sees how far the LAPIC got — its counts-per-millisecond, from which
|
||||
@@ -53,7 +53,7 @@ a missing PIT would hang the boot):
|
||||
|
||||
1. **CPUID leaf 0x15** — the CPU's TSC frequency directly, needing no external timer
|
||||
at all (the LAPIC is then measured against the TSC).
|
||||
2. The **HPET**, discovered via ACPI (see [discovery](discovery.md) / [acpi](acpi.md)).
|
||||
2. The **HPET**, discovered via ACPI (see [discovery](../os-development/discovery.md) / [acpi](../os-development/acpi.md)).
|
||||
3. The **ACPI PM timer** (a fixed 3.579545 MHz counter from the FADT).
|
||||
4. The **PIT** (legacy 8254, 1.193182 MHz) — last resort, and bounded so it can't hang.
|
||||
|
||||
@@ -87,7 +87,9 @@ because they decide whether we read time with a cheap `rdtsc` or fall back to th
|
||||
frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an
|
||||
**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a
|
||||
CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in
|
||||
`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is
|
||||
`calibrate`, and a TSC that doesn't advertise it is demoted to the HPET clocksource —
|
||||
provided a usable HPET exists (64-bit; a 32-bit one wraps too fast to stay monotonic).
|
||||
With no such fallback the TSC stays, there being nothing steadier to switch to. AMD is
|
||||
why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates
|
||||
the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a
|
||||
measured frequency without the invariance guarantee is not enough.
|
||||
@@ -98,7 +100,7 @@ values (a second socket, some firmware), so a thread migrating from a core readi
|
||||
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||
come up one at a time ([smp.md](smp.md)).
|
||||
come up one at a time ([smp.md](../os-development/smp.md)).
|
||||
|
||||
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||
@@ -127,6 +129,10 @@ if (state.vector < 32) {
|
||||
// else: spurious/unhandled — deliberately no EOI
|
||||
```
|
||||
|
||||
(A third branch has since joined for user mode, elided here:
|
||||
`state.vector == system_call_vector` (128) hands the trap frame to the ring-3
|
||||
syscall handler.)
|
||||
|
||||
Two things make device interrupts *return* where exceptions don't:
|
||||
|
||||
1. **The handler returns.** The timer handler just bumps a tick counter. Control
|
||||
@@ -151,8 +157,10 @@ needs, so only the handler can sequence it. See [drivers.md](drivers.md), where
|
||||
device is quieted by a driver in ring 3, long after the ISR has returned.
|
||||
|
||||
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
||||
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
||||
a handler must not use them; ours don't.)
|
||||
the interrupted registers. (The stubs originally didn't save the SSE/vector
|
||||
registers, so a handler couldn't use them; `isr_common` now does an
|
||||
`fxsave`/`fxrstor` of the full SSE/x87 state around dispatch — see
|
||||
[interrupts.md](../os-development/interrupts.md).)
|
||||
|
||||
## Turning them on
|
||||
|
||||
@@ -160,11 +168,11 @@ Exceptions can't be masked, which is why they worked all along. Maskable device
|
||||
interrupts don't fire until the CPU's interrupt flag is set — so the final step is
|
||||
`sti` (`arch.enableInterrupts()`), after the APIC and timer are configured. From
|
||||
that instant the kernel has a heartbeat, and its idle `hlt` loop
|
||||
([halting.md](halting.md)) wakes on every tick and dozes off again.
|
||||
([halting.md](../os-development/halting.md)) wakes on every tick and dozes off again.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `timer` test (see [testing.md](testing.md)) is the proof that an interrupt both
|
||||
The `timer` test (see [testing.md](../testing.md)) is the proof that an interrupt both
|
||||
*fires* and *returns*: it records the tick count, busy-waits, and checks the count
|
||||
advanced on its own.
|
||||
|
||||
@@ -180,21 +188,23 @@ spinning in unrelated code — is the whole mechanism working end to end.
|
||||
## Since (done elsewhere)
|
||||
|
||||
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||
reason a *returning* interrupt matters. See [scheduling.md](../os-development/scheduling.md).
|
||||
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||
[drivers.md](drivers.md).
|
||||
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||
user drivers — see [paging.md](paging.md).
|
||||
user drivers — see [paging.md](../os-development/paging.md).
|
||||
|
||||
## What's next (not done here)
|
||||
## What's next (partly done since)
|
||||
|
||||
- **The keyboard**: the PS/2 controller is port-mapped (`0x60`/`0x64`), and port I/O is
|
||||
now available to ring 3 via the claim-gated `io_read`/`io_write` syscalls
|
||||
([drivers.md](drivers.md)) — so the first *input* device is unblocked; it just needs
|
||||
writing (claim the controller, `irq_bind` GSI 1, read scancodes from `0x60`).
|
||||
- **MSI-X**: `msi_bind` gives one per-device edge-triggered vector (M15); MSI-X's
|
||||
multi-vector table (many queues per device, e.g. NVMe) is the remaining extension.
|
||||
- **The LAPIC's own page** is still mapped writeback-cacheable like the rest of the
|
||||
identity map. QEMU tolerates it; real hardware wants it uncacheable.
|
||||
- **The keyboard** — done, exactly as sketched: the PS/2 bus driver
|
||||
(`system/drivers/ps2-bus/`) claims the port-mapped 8042 controller through the
|
||||
claim-gated `io_read`/`io_write` syscalls ([drivers.md](drivers.md)), binds
|
||||
IRQ 1 (and the aux mouse's IRQ 12), reads scancodes from `0x60`, and decodes
|
||||
them into HID events for the [input service](input.md).
|
||||
- **MSI-X** — still open: `msi_bind` gives one per-device edge-triggered vector
|
||||
(M15); MSI-X's multi-vector table (many queues per device, e.g. NVMe) is the
|
||||
remaining extension.
|
||||
- **The LAPIC's own page** — still mapped writeback-cacheable like the rest of
|
||||
the identity map. QEMU tolerates it; real hardware wants it uncacheable.
|
||||
@@ -10,19 +10,19 @@ mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||
the manager is now the one answer to "what devices exist" for applications.
|
||||
The primitives underneath are real ([process-management.md](process-management.md):
|
||||
The primitives underneath are real ([process-management.md](../os-development/process-management.md):
|
||||
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||
policy process that turns [resilience.md](../os-development/resilience.md)'s restart goal into practice
|
||||
for drivers.
|
||||
|
||||
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||
document: that is the universal lifecycle every danos process speaks —
|
||||
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||
`runtime.process` interface. The device manager is that design's first serious
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md), signals over IPC and the stable
|
||||
`process` interface. The device manager is that design's first serious
|
||||
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||
driver is stopped, health-checked, and buried exactly like any other process.
|
||||
|
||||
@@ -35,7 +35,7 @@ The device tree is two things fused: *information* (what exists, how it nests) a
|
||||
claims, resource containment on `device_register`, the
|
||||
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||
dies** (settled; it is increment 1 of
|
||||
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md)). The three invariants in
|
||||
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||
buggy one would un-earn everything the microkernel bought.
|
||||
@@ -51,7 +51,7 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid
|
||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||
depends on when it lands. (It landed: [discovery.md](../os-development/discovery.md), M19–M20.)
|
||||
|
||||
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||
identical (parent, class, identity, resources) tuple returns the existing id
|
||||
@@ -73,8 +73,8 @@ one world.
|
||||
| Direction | Message | Purpose |
|
||||
|---|---|---|
|
||||
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||
| bus → manager | `child_added { parent, identity, resources }` | one node the bus discovered |
|
||||
| bus → manager | `child_removed { id }` | unplug, or the bus lost it |
|
||||
| bus → manager | `child_added { parent, bus_address, identity, device_id, hid }` | one node the bus discovered |
|
||||
| bus → manager | `child_removed { parent, bus_address }` | unplug, or the bus lost it |
|
||||
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||
| app → manager | `subscribe` | receive published add/remove events |
|
||||
|
||||
@@ -82,7 +82,7 @@ one world.
|
||||
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
|
||||
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||
@@ -91,13 +91,16 @@ mechanism), replacing first-come-first-served `device_claim` with policy. Identi
|
||||
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||
(Since the registry landed, `child_added` also carries a `bus` discriminator and
|
||||
the numeric `vendor`/`device`/`subsystem` ids the finer match levels need —
|
||||
see [/etc/devices.csv](devices-csv.md).)
|
||||
|
||||
## Supervision and restart
|
||||
|
||||
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||
On a death notification:
|
||||
|
||||
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||
1. **Read the reason** ([process-lifecycle.md](../os-development/process-lifecycle.md) increment 2).
|
||||
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||
@@ -136,8 +139,8 @@ way.
|
||||
## Increments
|
||||
|
||||
Increments 1–4 are the lifecycle prerequisites and live in
|
||||
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `runtime.process`). On top of those:
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `process`). On top of those:
|
||||
|
||||
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||
usb-xhci-bus becomes the first conforming driver.
|
||||
@@ -147,8 +150,10 @@ published exit events, signals + `runtime.process`). On top of those:
|
||||
to a manager-internal seam.
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||
the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds
|
||||
only the host bridge and the acpi-tables node. Matching moved with it:
|
||||
the acpi service (M20), see [discovery.md](../os-development/discovery.md); of the enumerable
|
||||
devices, the kernel seeds only the host bridge and the acpi-tables node (the
|
||||
non-enumerable platform nodes — processors, interrupt controllers, the HPET,
|
||||
the loader's framebuffer — stay kernel-seeded too). Matching moved with it:
|
||||
`child_added` grew a `device_id` (the kernel-registered id, `no_device` for
|
||||
unregistered leaves like USB ports) and a firmware `hid`, and the manager now
|
||||
matches drivers from those **reports** rather than its boot-time snapshot. The
|
||||
@@ -167,6 +172,15 @@ published exit events, signals + `runtime.process`). On top of those:
|
||||
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||
the world (above). Checkpointing driver state with the manager is deferred until
|
||||
something demonstrates the need.
|
||||
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` are
|
||||
honest at two bus types; the third triggers the manifest (a driver declares what
|
||||
it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||
- **Matching is a registry, not code (resolved 2026-07-26).** `driverFor`/
|
||||
`pciDriverFor` were honest at two bus types; the third (USB) was matched in code
|
||||
too, and then the switch tables started to hurt — they keyed PCI matches on the
|
||||
class triple alone, so a virtio-gpu could only be matched as a generic display
|
||||
function and the driver had to re-confirm its `1AF4:1050` identity from config
|
||||
space after being spawned. The manifest the earlier note anticipated landed as a
|
||||
human-readable registry: **[/etc/devices.csv](devices-csv.md)**, parsed by the
|
||||
pure `device-registry` module and read by the manager at boot. A row binds a
|
||||
driver to a device by any of base / subclass / prog-IF / vendor / device /
|
||||
subsystem / `_HID`, most-specific match winning; it is authoritative (no
|
||||
compiled-in fallback — an unmatched device is logged, never guessed).
|
||||
`pciDriverForIdentity`, `hidDriverFor`, and `usbDriverForIdentity` are gone.
|
||||
@@ -0,0 +1,102 @@
|
||||
# /etc/devices.csv — the device registry
|
||||
|
||||
**Status: built (2026-07-26).** The device manager reads `/etc/devices.csv` at
|
||||
boot and binds every device a bus driver reports to the driver the registry
|
||||
names. It replaces the three hand-written `switch` tables that used to live in
|
||||
the manager (`pciDriverForIdentity`, `hidDriverFor`, `usbDriverForIdentity`) —
|
||||
the "manifest" [device-manager.md](device-manager.md) anticipated once code
|
||||
matching started to hurt. The parser and matcher are the pure, unit-tested
|
||||
`device-registry` module (`library/device/registry/device-registry.zig`).
|
||||
|
||||
## Why a registry
|
||||
|
||||
The switch tables keyed PCI matches on the 24-bit class/subclass/prog-IF triple
|
||||
alone. That is too coarse: a virtio-gpu is just "display / other" by class, so it
|
||||
could only be *class-matched* and the driver had to re-confirm its real
|
||||
`1AF4:1050` identity from config space **after** the manager had already spawned
|
||||
it. The registry lets a rule bind on the full identity — down to vendor, device,
|
||||
and subsystem — so the manager makes the precise decision itself, and the driver
|
||||
comes up already knowing it is the right one.
|
||||
|
||||
It is also **data, not code**: teaching the system new hardware is a line in a
|
||||
file, not an edit-and-recompile of the manager. And it is **greppable** — one
|
||||
place to read "what binds what," the same idea as Linux's `modules.alias`.
|
||||
|
||||
## The file
|
||||
|
||||
One rule per line, nine comma-separated fields; `#` starts a comment (whole-line
|
||||
or trailing); blank lines are ignored. Whitespace around a field is trimmed, so
|
||||
columns may be padded for readability.
|
||||
|
||||
```
|
||||
# bus base class prog_if vendor device subsystem hid driver
|
||||
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||
pci, 03, 00, 00, *, *, *, *, /system/drivers/display
|
||||
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||
acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||
```
|
||||
|
||||
| Field | Meaning | Notes |
|
||||
|---|---|---|
|
||||
| `bus` | `pci` \| `usb` \| `acpi` | which bus reported the device; picks the namespace for the id columns |
|
||||
| `base` | PCI base class / USB class | hex |
|
||||
| `class` | PCI subclass / USB subclass | hex |
|
||||
| `prog_if` | PCI prog-IF / USB protocol | hex |
|
||||
| `vendor` | PCI vendor / USB idVendor | hex |
|
||||
| `device` | PCI device / USB idProduct | hex |
|
||||
| `subsystem` | PCI subsystem, `(ssvid<<16)\|ssid` | hex; blank for usb/acpi |
|
||||
| `hid` | ACPI `_HID` (e.g. `PNP0303`) | blank for pci/usb |
|
||||
| `driver` | full ramdisk path to spawn | e.g. `/system/drivers/virtio-gpu` |
|
||||
|
||||
`*` or an empty field is a **wildcard** — it matches anything and adds nothing to
|
||||
a rule's specificity.
|
||||
|
||||
## Levels of detection: most-specific-wins
|
||||
|
||||
Several rows may match one device. The manager picks the **most specific** — the
|
||||
one that pins the finest-grained fields. Specificity weights double from the
|
||||
coarsest level so each outweighs all coarser levels combined:
|
||||
|
||||
```
|
||||
base(1) < class(2) < prog_if(4) < vendor(8) < subsystem(16) < device(32) ≈ hid(32)
|
||||
```
|
||||
|
||||
So the generic `pci, 03, 00, 00, …/display` rule and the precise
|
||||
`pci, 03, 80, *, 1AF4, 1050, …/virtio-gpu` rule coexist: the virtio card
|
||||
(vendor 1AF4, device 1050) takes the specific rule; a plain VGA adapter still
|
||||
falls to the generic one. Two rules that match a device with the *same*
|
||||
specificity are a registry authoring error — the manager logs it loudly and binds
|
||||
the first, so the shadowed rule is visible rather than silently dropped.
|
||||
|
||||
## Authoritative — no code fallback
|
||||
|
||||
There is no compiled-in default table behind the registry. A device that no row
|
||||
matches goes **unbound** and is logged; the manager never guesses. A missing or
|
||||
empty `/etc/devices.csv` therefore means nothing matches — which is loud at boot,
|
||||
not a silent half-working system.
|
||||
|
||||
## How the manager reads it
|
||||
|
||||
`/etc/devices.csv` is bundled into the initial ramdisk (`build.zig`'s `bundled`
|
||||
list). The kernel serves the initrd's `/etc` tree directly — the `fat` service is
|
||||
spawned *after* the device manager and is irrelevant to `/etc` — so the manager
|
||||
reads the file with a plain `fs.open("/etc/devices.csv")` + `read`, with no
|
||||
filesystem service running and no boot-ordering dependency. It parses the bytes
|
||||
once in `initialise`, before any bus driver can report a device to match.
|
||||
|
||||
## Feeding the matcher: the widened report
|
||||
|
||||
Finer-grained matching needs identity the old ABI threw away. Two things carry it
|
||||
now: `child_added` (and `DeviceDescriptor`) grew `vendor` / `device` /
|
||||
`subsystem` fields, filled by the PCI bus driver from config space (offsets
|
||||
0x00 and 0x2C); and each bus driver states its `bus` in the report (a `BusKind`),
|
||||
so the manager reads a PCI class triple and a USB class triple — the same 24 bits
|
||||
in different namespaces — against the right `bus` column.
|
||||
|
||||
## Adding a driver
|
||||
|
||||
1. Build the driver binary and bundle it at `/system/drivers/<name>` (build.zig).
|
||||
2. Add a row to `etc/devices.csv` naming the identity it binds and its full path.
|
||||
|
||||
No device-manager change is required — the registry is the seam.
|
||||
@@ -7,7 +7,8 @@ Read [display.md](display.md) first for the *why*; this is the *what* and the *o
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **Handoff = device node + write-combining `mmio_map`.** The kernel seeds a synthetic
|
||||
`display0` node from `BootInformation.framebuffer`; the service claims + WC-maps it.
|
||||
display node (found by class, not name) from `BootInformation.framebuffer`; the
|
||||
service claims + WC-maps it.
|
||||
(Not a bespoke `framebuffer_map` syscall — the device route inherits ownership,
|
||||
release-on-death, and re-claim-on-restart.)
|
||||
- **v1 = the full compositor pipeline on the dumb framebuffer.** One `display` service
|
||||
@@ -17,9 +18,9 @@ Read [display.md](display.md) first for the *why*; this is the *what* and the *o
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations in
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations in
|
||||
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||
go through `addUserBinary` in [build.zig](../build.zig) and get packed into the
|
||||
go through `addUserBinary` in [build.zig](../../build.zig) and get packed into the
|
||||
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||
`runtime` module.
|
||||
|
||||
@@ -28,7 +29,7 @@ initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported i
|
||||
- `zig build test` — host unit tests (compositor math: layer clipping, damage merge,
|
||||
pitch/format blits are all host-testable with a fake framebuffer).
|
||||
- `python3 test/qemu_test.py <case>` — boots the real kernel in QEMU; assert on the
|
||||
serial log ([tests.zig](../system/kernel/tests.zig) is the registry).
|
||||
serial log ([tests.zig](../../system/kernel/tests.zig) is the registry).
|
||||
- The `run-efi` target renders to QEMU's display (`-device VGA,edid=on,xres=1280,yres=720`)
|
||||
— a screenshot confirms pixels for the milestones whose gate is visual.
|
||||
|
||||
@@ -38,22 +39,22 @@ initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported i
|
||||
|
||||
Make the boot framebuffer reachable and mappable **write-combining** from user space.
|
||||
|
||||
- [x] [device-abi.zig](../system/devices/device-abi.zig): added `DeviceClass.display`; a
|
||||
- [x] [device-abi.zig](../../library/device/model/device-abi.zig): added `DeviceClass.display`; a
|
||||
`DisplayInfo{ width, height, pitch, format }` carried on the descriptor; a
|
||||
`flags` field on `ResourceDescriptor` + `resource_flag_write_combining`.
|
||||
- [x] [devices-broker.zig](../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||
- [x] [devices-broker.zig](../../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||
pitch, format)` publishes a root-level `display` node with one WC-flagged `memory`
|
||||
resource `[base, height*pitch]` + the `DisplayInfo`; `displayDevice()` /
|
||||
`displayClaimed()`. Seeded from `kmain` after `devices_broker.init`.
|
||||
- [x] [process.zig](../system/kernel/process.zig) `systemMmioMap` + paging
|
||||
- [x] [process.zig](../../system/kernel/process.zig) `systemMmioMap` + paging
|
||||
(`mapUserDeviceInto` gains a `write_combining` bool): a resource's WC flag maps it
|
||||
through the WC PAT slot (`setupPat`) instead of strong-uncacheable.
|
||||
- [x] [console.zig](../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||
- [x] [console.zig](../../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||
the display device is claimed (driven from `systemDeviceClaim` / release); the
|
||||
terminal panic + exception paths clear it first so a dying machine still draws.
|
||||
|
||||
**Gate (met, automated):** the `display` kernel test (`python3 test/qemu_test.py display`,
|
||||
`displayTest` in [tests.zig](../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||
`displayTest` in [tests.zig](../../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||
and geometry, then walks the real claim + `mmio_map` path into a throwaway address space
|
||||
and verifies the leaf is **write-combining** (PAT entry 4: PAT bit set, PCD/PWT clear) —
|
||||
with an uncacheable-still-uncacheable regression guard. Chosen over the original
|
||||
@@ -66,10 +67,10 @@ Regression-checked: `discovery`, `ioport`, `claim-release`, `supervision`, `devi
|
||||
|
||||
Stand up the named service and the double-buffer, no layers yet.
|
||||
|
||||
- [x] `system/services/display/protocol.zig`: `Operation{ info, create_layer,
|
||||
- [x] `library/protocol/display/display-protocol.zig`: `Operation{ info, create_layer,
|
||||
configure_layer, destroy_layer, fill_rect, blit_tile, damage, present }`; `extern`
|
||||
`Request`/`Reply`; size + `maximum_payload` consts. (Model: block/protocol.zig.)
|
||||
- [x] [abi.zig](../system/abi.zig): `ServiceId.display = 9`.
|
||||
- [x] [abi.zig](../../system/abi.zig): `ServiceId.display = 9`.
|
||||
- [x] `system/services/display/display.zig`: `main` → enumerate + claim + WC-map the LFB
|
||||
(front) → `mmap` a cacheable back buffer of `height*pitch` → `runtime.service.run`.
|
||||
`info` and a whole-screen `present` (back → front) are live; layer ops fail-stub
|
||||
@@ -77,8 +78,8 @@ Stand up the named service and the double-buffer, no layers yet.
|
||||
- [x] [library/runtime/display.zig](../library/runtime/runtime.zig) (+ barrel export of
|
||||
`display` and `display_protocol`): `info()` and `present()`, cached `.display`
|
||||
lookup with retry (model: block.zig).
|
||||
- [x] [init.zig](../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||
- [x] [build.zig](../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||
- [x] [init.zig](../../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||
- [x] [build.zig](../../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||
via `addUserBinary`; packed into the initial-ramdisk; installed to
|
||||
`/system/services/display`.
|
||||
- [x] **Kernel fix the back buffer surfaced:** `mmap` was capped at 256 pages (1 MiB) by
|
||||
@@ -104,7 +105,7 @@ The heart: composite an ordered layer stack, present only what changed.
|
||||
- [x] `create_layer` / `configure_layer` (damages old + new footprints) / `destroy_layer`,
|
||||
`fill_rect`, `blit_tile` (reads the inline tile from the IPC payload, unaligned-safe),
|
||||
`damage`, `present`.
|
||||
- [x] Pure, host-tested [compositor.zig](../system/services/display/compositor.zig): `Rect`
|
||||
- [x] Pure, host-tested [compositor.zig](../../system/services/display/compositor.zig): `Rect`
|
||||
(intersect/unite), `Surface`, `fillRect`, `composite` (opaque, clipped to a damage
|
||||
rect), `blitTile`. `present` clears the damaged region to the wallpaper, paints the
|
||||
visible layers bottom-to-top (z-sorted), and flushes just that rect back → front (WC).
|
||||
@@ -134,7 +135,7 @@ Prove the pipeline end-to-end from a separate process.
|
||||
every IPC message at `MESSAGE_MAXIMUM` = 256 — so `replyWait` rejected the oversized
|
||||
receive buffer with `-E2BIG` and the serve loop had been *spinning* since D2 (unseen,
|
||||
as D2/D3 matched init-time heartbeats). Set it to 256; `blit_tile` is now explicitly
|
||||
a small-tile path (≤ 54 px inline), larger bitmaps being the deferred shm surface.
|
||||
a small-tile path (≤ 54 px inline), larger bitmaps being the deferred shared-memory surface.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display-demo` spawns the service + `display-demo`;
|
||||
the demo drives a run of frames of motion through the layer client API and logs
|
||||
@@ -147,10 +148,10 @@ still pass, and the default `zig build` is clean.
|
||||
- [x] The three integration cases exist and pass: `display` (D1 handoff, kernel),
|
||||
`display-service` (D2/D3 compositor + self-check), and `display-demo` (D4 full
|
||||
pipeline: spawn `display` + `display-demo`, match `display-demo: ok`) —
|
||||
[tests.zig](../system/kernel/tests.zig) + [qemu_test.py](../test/qemu_test.py). Plus
|
||||
[tests.zig](../../system/kernel/tests.zig) + [qemu_test.py](../../test/qemu_test.py). Plus
|
||||
the pure host tests (`zig build test`).
|
||||
- [x] [display.md](display.md) updated to the built state (the "Verifying it" section names
|
||||
the real cases); [README index](README.md) entry present (#19); the `display-track`
|
||||
the real cases); [README index](../README.md) entry present (#19); the `display-track`
|
||||
memory marked DONE with the commits.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display display-service display-demo` all pass,
|
||||
@@ -172,7 +173,7 @@ documented (docs/display.md): no runtime mode-setting (native backend) and no tr
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Shared-memory surfaces** — generalize M13 capability passing to memory objects
|
||||
(`shm_create`/`shm_map`), so bitmap clients hand the compositor a rendered surface
|
||||
(`shared_memory_create`/`shared_memory_map`), so bitmap clients hand the compositor a rendered surface
|
||||
instead of drawing commands. The compositor's layer model already anticipates it.
|
||||
- **Native backend (Bochs DISPI, then virtio-gpu)** — behind the same internal backend
|
||||
interface as the dumb framebuffer: EDID mode list + runtime resolution/bpp change +
|
||||
@@ -6,27 +6,27 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **First native backend = virtio-gpu** (VM standard: mode-set + present/flush + vsync).
|
||||
- **First native backend = virtio-gpu** (VM standard: mode-set + fenced present/flush).
|
||||
- **Dynamic hot-attach**: boot on GOP, upgrade to native when the driver **announces**
|
||||
(push, not polling); re-attach across driver restarts; GOP is the floor for "no driver
|
||||
ever," not a live fall-back after a reprogram.
|
||||
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||
- **v2 builds the shared-memory capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
- The compositor's layers/back-buffer/damage are **unchanged**; only scanout is pluggable.
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||
[abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
**Every gate is serial-checkable — no screenshots** (this plan is built to run unattended).
|
||||
Where "does it actually display" would otherwise need a human eyeball, the code **reads its
|
||||
own pixels back**: the scanout resource is CPU-visible RAM (shm-backed) and the back buffer
|
||||
own pixels back**: the scanout resource is CPU-visible RAM (shared-memory-backed) and the back buffer
|
||||
is cacheable, so a driver/compositor can write a known value, read it back, and log a
|
||||
pass/fail — and a virtio `resource_flush` is confirmed by the device **acking it on the
|
||||
used ring**. Those two together (pixel-readback + flush-ack) are the automated stand-in for
|
||||
@@ -46,7 +46,7 @@ Extract scanout from the compositor so today's path becomes one backend among fu
|
||||
|
||||
- [x] `system/services/display/backend.zig`: a `Backend` tagged union with `info()`,
|
||||
`surface()` (the cacheable compose target), `present(damage)`, and capability flags
|
||||
(`canModeSet`/`hasVsync`, both false for GOP).
|
||||
(`canModeSet`/`hasFencedPresent`, both false for GOP).
|
||||
- [x] The v1 GOP path is now `backend.Gop` (claims the `display` node, WC-maps the LFB,
|
||||
keeps the cacheable back buffer, `present` = the damage-rect WC copy). display.zig
|
||||
composes into `backend.surface()` and calls `backend.present(damage)` — no LFB or
|
||||
@@ -57,25 +57,25 @@ Extract scanout from the compositor so today's path becomes one backend among fu
|
||||
**Gate (met):** `display-service` + `display-demo` pass **unchanged** (pure refactor; GOP
|
||||
is the only backend), and `zig build test` stays green.
|
||||
|
||||
## V2 — The `shm` cross-process memory capability (kernel) ✅
|
||||
## V2 — The shared-memory cross-process capability (kernel) ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a
|
||||
`shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous,
|
||||
- [x] [abi.zig](../../system/abi.zig): `shared_memory_create` (34) / `shared_memory_map` (35) syscalls + a
|
||||
`shared_memory_test` service id. Handlers in process.zig: `shared_memory_create(len)` allocates contiguous,
|
||||
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||
handle, maps them into the caller's shm arena → returns virtual_address + handle; `shm_map(cap)`
|
||||
handle, maps them into the caller's shared-memory arena → returns virtual_address + handle; `shared_memory_map(cap)`
|
||||
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
||||
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
||||
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
||||
dispatch by kind, so an `ShmObject` rides an `ipc_call` `send_cap` exactly like an
|
||||
dispatch by kind, so a `SharedMemoryObject` rides an `ipc_call` `send_cap` exactly like an
|
||||
endpoint and frees only when its last capability drops. `mapUserSharedInto` (paging)
|
||||
maps WB-cacheable + `device_grant`, so a sharer's teardown never frees the shared
|
||||
frames — the object owns them.
|
||||
- [x] `library/runtime/shm.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||
- [x] `library/runtime/shared-memory.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||
`map(handle) -> ptr`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py shm` — `shm-client` creates a region, writes a
|
||||
pattern, and passes its capability to `shm-server` as an `ipc_call` send_cap; the server
|
||||
`shm_map`s it and reads the **same bytes** back → `shm: shared 4096 bytes ok`. Guardrail:
|
||||
**Gate (met):** `python3 test/qemu_test.py shared-memory` — `shared-memory-client` creates a region, writes a
|
||||
pattern, and passes its capability to `shared-memory-server` as an `ipc_call` send_cap; the server
|
||||
`shared_memory_map`s it and reads the **same bytes** back → `shared-memory: shared 4096 bytes ok`. Guardrail:
|
||||
`ipc`/`ipc-call`/`ipc-cap`, `supervision`, `dma`, `usermem`, `display-service`, and host
|
||||
tests all still pass — the handle-table change broke no existing IPC.
|
||||
|
||||
@@ -88,7 +88,7 @@ tests all still pass — the handle-table change broke no existing IPC.
|
||||
VERSION_1, and stand up the control virtqueue in coherent DMA. `virtio-gpu-protocol.zig`
|
||||
+ `virtio-pci.zig` for the control/transport structs (host-tested sizes).
|
||||
- [x] Create a 2D scanout resource backed by a coherent DMA region (V4 swaps this for the
|
||||
shm-shared surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||
shared-memory surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||
+ `resource_flush` of a test pattern, and wait on the used ring.
|
||||
- [x] Register a `scanout` service (`ServiceId.scanout` = 11).
|
||||
|
||||
@@ -102,7 +102,7 @@ end without a screenshot (the used-ring ack is the device confirming it consumed
|
||||
|
||||
## V4 — The native backend + hot-attach ✅
|
||||
|
||||
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared `shm` scanout surface
|
||||
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared-memory scanout surface
|
||||
(the compositor composes straight into the device's resource backing; x86 DMA is
|
||||
coherent, so the cacheable shared pages need no flush), `present(damage)` = a `present`
|
||||
request over the driver's `.scanout` endpoint (→ transfer-to-host + resource flush).
|
||||
@@ -112,7 +112,7 @@ end without a screenshot (the used-ring ack is the device confirming it consumed
|
||||
driver registered it), switches backend, and re-composites the current frame full-screen.
|
||||
The present is deferred to a one-shot timer so it runs *after* the reply unblocks the
|
||||
driver and it serves `.scanout` — presenting inline would deadlock.
|
||||
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shm_physical` (a
|
||||
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shared_memory_physical` (a
|
||||
new syscall) gives the driver the guest-physical of the shared surface for `attach_backing`.
|
||||
|
||||
**Gate (met):** the `display-native` case (QEMU `-device virtio-gpu-pci`, `mem` bumped since it
|
||||
@@ -123,7 +123,7 @@ confirm the composited frame landed (`display: native present verified`), while
|
||||
ok` still fires — checked order-independently. Without `-device virtio-gpu-pci` nothing is
|
||||
announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pass unchanged.
|
||||
|
||||
## V5 — Mode-setting, EDID, and vsync ✅
|
||||
## V5 — Mode-setting, EDID, and fenced presents ✅
|
||||
|
||||
- [x] The driver negotiates `VIRTIO_GPU_F_EDID` (when offered) and reads the monitor's EDID,
|
||||
logging its preferred mode; it offers a small mode list over `.scanout` `get_modes`. The
|
||||
@@ -131,14 +131,16 @@ announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pas
|
||||
scanout rectangle (no resource/surface churn) — a runtime resolution change. `runtime.display`
|
||||
gains `modes()` / `setMode()` (display-protocol `get_modes`/`set_mode`, forwarded to the backend).
|
||||
- [x] Every `resource_flush` is issued fenced (`VIRTIO_GPU_FLAG_FENCE`); the device signals the
|
||||
fence when the frame is on screen, which the used-ring ack the synchronous present waits on
|
||||
already gates — a tear-free present.
|
||||
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasVsync` = true.
|
||||
fence when it has consumed the frame, which the used-ring ack the synchronous present waits
|
||||
on already gates — a tear-free present. (Completion feedback, **not vblank**: base
|
||||
virtio-gpu 2D has no display-refresh event, so nothing paces presents to the monitor —
|
||||
see the "Fenced is not vsync" note in [display-v2.md](display-v2.md).)
|
||||
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasFencedPresent` = true.
|
||||
|
||||
**Gate (met):** the `display-modeset` case (reusing the display-native boot) upgrades to
|
||||
virtio-gpu, queries the driver's modes, `setMode`s to a different resolution, and confirms the
|
||||
change by reading the backend's geometry back (`display: mode set to {w}x{h}, verified`); the
|
||||
fenced present path is exercised and confirmed (`display: vsync present ok`) — both from serial,
|
||||
fenced present path is exercised and confirmed (`display: fenced present ok`) — both from serial,
|
||||
passing 3/3. The driver also logs the EDID preferred mode (`virtio-gpu: EDID preferred mode …`).
|
||||
|
||||
## V6 — Resilience (restart + re-attach) + tests + docs ✅
|
||||
@@ -157,14 +159,14 @@ kills the virtio-gpu driver once after it hellos; the restart policy respawns it
|
||||
re-announces, and the compositor logs `display: scanout re-attached` after the initial
|
||||
`display: scanout upgraded to virtio-gpu`, with no CPU exception / panic (the compositor
|
||||
survives) — passing 3/3. All v1 + v2 cases (host tests, `ipc`/`ipc-call`/`ipc-cap`,
|
||||
`supervision`, `shm`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||
`supervision`, `shared-memory`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||
`display-modeset`) pass; default `zig build` is clean.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Client-rendered surfaces** — now unblocked by the `shm` capability (V2): an app renders
|
||||
- **Client-rendered surfaces** — now unblocked by the shared-memory capability (V2): an app renders
|
||||
its own bitmap and hands the compositor a reference. A natural follow-on.
|
||||
- **Bochs DISPI backend** — a simpler second native backend (mode-set only, dumb scanout);
|
||||
slots behind the same interface if wanted.
|
||||
@@ -1,8 +1,8 @@
|
||||
# The display service v2: a pluggable scanout backend
|
||||
|
||||
**Status: complete (V1–V6).** The compositor boots on the GOP framebuffer and, when a
|
||||
virtio-gpu driver announces itself, hot-attaches a native backend over the shared `shm`
|
||||
scanout surface — with runtime mode-setting, EDID, and fenced (vsync) presents, and it
|
||||
virtio-gpu driver announces itself, hot-attaches a native backend over the shared-memory
|
||||
scanout surface — with runtime mode-setting, EDID, and fenced presents, and it
|
||||
re-attaches across driver restarts. All serial-gated (see [display-v2-plan.md](display-v2-plan.md)).
|
||||
|
||||
v1 ([display.md](display.md)) is a compositor that owns the **GOP framebuffer** — it
|
||||
@@ -25,11 +25,11 @@ The compositor itself (layers, back buffer, damage) does not change. Only the la
|
||||
scanout backend (selected at runtime — GOP by default, native when it appears)
|
||||
│
|
||||
├─ GopBackend the v1 path: WC copy back→front to the firmware LFB.
|
||||
│ Always available. No mode-set, no vsync. THE FLOOR.
|
||||
│ Always available. No mode-set, no present fence. THE FLOOR.
|
||||
│
|
||||
└─ VirtioGpuBackend talks to a virtio-gpu driver process over a `scanout`
|
||||
service: present via a shared resource + flush (real vsync),
|
||||
EDID mode list, runtime mode-set.
|
||||
service: present via a shared resource + fenced flush,
|
||||
a fixed mode list, runtime mode-set, EDID refresh rate.
|
||||
```
|
||||
|
||||
A **backend** is a small interface the compositor calls:
|
||||
@@ -37,8 +37,8 @@ A **backend** is a small interface the compositor calls:
|
||||
- `surface()` → the pixels to compose into and their geometry `{ptr, pitch, format, w, h}`
|
||||
(the LFB for GOP; a shared scanout resource for virtio-gpu),
|
||||
- `present(damage: Rect)` → make the damaged region visible (a no-op-ish WC copy for GOP;
|
||||
a virtio flush, optionally vsync-fenced, for the native path),
|
||||
- capability queries — `canModeSet`, `hasVsync` — and, when supported, `modes()` /
|
||||
a fenced virtio flush for the native path),
|
||||
- capability queries — `canModeSet`, `hasFencedPresent` — and, when supported, `modes()` /
|
||||
`setMode(m)`.
|
||||
|
||||
The compositor composes into `surface()` and calls `present(damage)` exactly as it does
|
||||
@@ -53,10 +53,12 @@ manager brings it up after boot), and because danos is meant to be resilient:
|
||||
blank screen while drivers load — the exact v1 behaviour.
|
||||
2. **Upgrade on announce.** When the virtio-gpu driver has claimed its device and set up a
|
||||
scanout, it **announces itself to the display service** (a `push`: the driver looks up
|
||||
`.display` and sends an *attach-scanout* message carrying its `scanout` endpoint as a
|
||||
capability). The compositor switches to `VirtioGpuBackend` and re-presents the current
|
||||
frame full-screen. Push beats polling — the compositor doesn't know a priori which
|
||||
driver, if any, exists, and danos has no service-registration pub/sub.
|
||||
`.display` and sends an *attach-scanout* message carrying the shared scanout **surface**
|
||||
as a capability; the compositor maps it and reaches the driver's present/mode channel by
|
||||
looking up the registered `scanout` service). The compositor switches to
|
||||
`VirtioGpuBackend` and re-presents the current frame full-screen. Push beats polling —
|
||||
the compositor doesn't know a priori which driver, if any, exists, and danos has no
|
||||
service-registration pub/sub.
|
||||
3. **Native is restartable, not fallback-on-crash.** Once a native driver has reprogrammed
|
||||
the device, the firmware's GOP framebuffer is **stale** — "native → GOP" is not a clean
|
||||
fall-back. So a native driver that **crashes** is *restarted* by its supervisor (the
|
||||
@@ -64,9 +66,10 @@ manager brings it up after boot), and because danos is meant to be resilient:
|
||||
(native → native). The screen freezes on the last frame during the gap — acceptable.
|
||||
4. **GOP is the floor for "no driver was ever there."** On a real GPU (NVIDIA/AMD/Intel)
|
||||
the class-0x03 device matches nothing in the driver table, no `scanout` is ever
|
||||
announced, and the compositor stays on GOP forever — no special-casing. Only if a
|
||||
native driver *permanently* gives up (crash-loop cap) does the compositor attempt GOP
|
||||
again, and even then only if the LFB is still mappable.
|
||||
announced, and the compositor stays on GOP forever — no special-casing. If a native
|
||||
driver *permanently* gives up (the device manager's crash-loop cap), nothing tells the
|
||||
compositor and there is no path back to GOP — the screen stays frozen on the last
|
||||
frame. A GOP revert (sensible only while the LFB is still mappable) is not built.
|
||||
|
||||
## The shared-memory primitive this needs
|
||||
|
||||
@@ -77,9 +80,9 @@ deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural gen
|
||||
of M13 capability-passing from *endpoints* to *memory objects* —
|
||||
|
||||
```
|
||||
shm_create(len) -> {handle, virtual_address} // a shareable, page-aligned RAM region
|
||||
shared_memory_create(len) -> {handle, virtual_address} // a shareable, page-aligned RAM region
|
||||
… pass `handle` as the send_cap on an ipc_call …
|
||||
shm_map(cap) -> virtual_address // the receiver maps the same physical pages
|
||||
shared_memory_map(cap) -> virtual_address // the receiver maps the same physical pages
|
||||
```
|
||||
|
||||
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
||||
@@ -91,36 +94,49 @@ reference instead of drawing by command). One piece of kernel work, two features
|
||||
A new ring-3 driver process (the topology v1 anticipated — "split the driver from the
|
||||
compositor when a second backend arrives"). It claims the virtio-gpu PCI function, and:
|
||||
|
||||
- sets up the **virtqueues** (control + cursor) and the device's config space,
|
||||
- creates a **2D scanout resource** backed by an `shm` region, `attach_backing`s it,
|
||||
`set_scanout`s it to a CRTC, and `resource_flush`es damaged rectangles,
|
||||
- reads **EDID** (the `GET_EDID` control command) for the mode list, and `set_scanout`
|
||||
at a chosen mode for **runtime mode-setting**,
|
||||
- sets up the **control virtqueue** (queue 0 — the only queue it uses; no cursor queue)
|
||||
and the device's config space,
|
||||
- creates a **2D scanout resource** backed by a shared-memory region, `attach_backing`s it,
|
||||
`set_scanout`s it to a CRTC, and on each present transfers and `resource_flush`es the
|
||||
full current-mode rectangle (damage-narrowed flushes are a later refinement),
|
||||
- reads **EDID** (the `GET_EDID` control command) to log the monitor's preferred timing
|
||||
and derive the refresh rate it announces (the compositor's frame-clock seed); the mode
|
||||
list it offers is a fixed pair — 640×480 and 800×600 — and `set_scanout` at a chosen
|
||||
mode gives **runtime mode-setting**,
|
||||
- registers a `scanout` service and announces to the display service.
|
||||
|
||||
Its `resource_flush` is the real **present** — and gives a genuine **vsync/tear-free**
|
||||
path a dumb GOP framebuffer can't.
|
||||
Its `resource_flush` is the real **present** — and gives a **fenced, tear-free** path a
|
||||
dumb GOP framebuffer can't.
|
||||
|
||||
**Fenced is not vsync.** The fence completes when the device has *consumed* the frame:
|
||||
real completion feedback, and tear-freedom by snapshot semantics (the host displays
|
||||
discrete transferred frames, never a half-written surface). It is **not** a vblank —
|
||||
base virtio-gpu 2D has no display-refresh event at all (Linux's driver for this device
|
||||
fakes one with a software timer), so nothing paces presents to the monitor's refresh.
|
||||
Refresh-paced presents need either a native driver's vblank interrupt (delivered over
|
||||
the existing IRQ-as-IPC path) or the compositor's own frame clock.
|
||||
|
||||
## What v2 unlocks — and its honest scope
|
||||
|
||||
Behind the abstraction, a native backend gives runtime **mode-setting** (resolution /
|
||||
refresh / bpp), **EDID** enumeration, and **vsync**. But only on devices we have a driver
|
||||
Behind the abstraction, a native backend gives runtime **mode-setting** (resolution only —
|
||||
the scanout protocol carries neither refresh nor bpp), an **EDID**-derived refresh rate, and
|
||||
**fenced presents**. But only on devices we have a driver
|
||||
for — realistically **VMs** (virtio-gpu, and later maybe Bochs DISPI). Real discrete GPUs
|
||||
need per-vendor KMS-class drivers that aren't getting written, so they **stay on GOP** —
|
||||
which is genuinely fine (v1 on the NVIDIA box is smooth). So v2's real value is twofold:
|
||||
the **pluggable architecture** (a driver slots in when one exists) and a **rich, vsync'd
|
||||
the **pluggable architecture** (a driver slots in when one exists) and a **rich, fenced
|
||||
path in VMs**, where danos development happens. The framebuffer floor never goes away.
|
||||
|
||||
## Locked decisions
|
||||
|
||||
- **First native backend: virtio-gpu** — the VM standard; gives mode-set + a real
|
||||
present/flush (and vsync), and exercises the whole pluggable design. Tested with QEMU
|
||||
present/flush (fenced), and exercises the whole pluggable design. Tested with QEMU
|
||||
`-device virtio-gpu`.
|
||||
- **Dynamic hot-attach** — boot on GOP, upgrade to native on the driver's announce,
|
||||
re-attach across driver restarts; GOP is the floor for "no driver ever," not a live
|
||||
fall-back after a reprogram.
|
||||
- **Detection = push** (the driver announces to `.display`), not compositor polling.
|
||||
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||
- **v2 builds the shared-memory capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
|
||||
## See also
|
||||
@@ -128,4 +144,4 @@ path in VMs**, where danos development happens. The framebuffer floor never goes
|
||||
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||
- [resilience.md](resilience.md) — the restart machinery the hot-attach leans on.
|
||||
- [resilience.md](../os-development/resilience.md) — the restart machinery the hot-attach leans on.
|
||||
@@ -1,12 +1,12 @@
|
||||
# The display service: a framebuffer compositor
|
||||
|
||||
The [framebuffer](framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../system/kernel/console.zig) draws text
|
||||
The [framebuffer](../os-development/framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../../system/kernel/console.zig) draws text
|
||||
into it directly. That console is a stop-gap. The **display service**
|
||||
(`system/services/display/`) is the real thing: an ordinary ring-3 process that *owns*
|
||||
the framebuffer, composes a stack of **layers** into an off-screen back buffer, and
|
||||
**presents** finished frames to the screen — the display half of the GUI track
|
||||
([vision.md](vision.md)), the sibling of the [input service](input.md).
|
||||
([vision.md](../vision.md)), the sibling of the [input service](input.md).
|
||||
|
||||
This note is the architecture and the reasoning behind it. The concrete build order
|
||||
lives in [display-plan.md](display-plan.md).
|
||||
@@ -20,20 +20,20 @@ which one you're holding decides what you can do.
|
||||
|
||||
- **GOP is firmware's *temporary* driver** for the display controller. It gives you a
|
||||
linear framebuffer pointer and can set video modes — but only until
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../boot/efi.zig)
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../../boot/efi.zig)
|
||||
reads the monitor's EDID, picks the native mode, and calls `set_mode` **before**
|
||||
exiting ([gop.md](gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
exiting ([gop.md](../os-development/gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
mode list, no EDID. What survives is the frozen snapshot in
|
||||
[`BootInformation.framebuffer`](../system/boot-handoff.zig): `{base, width, height,
|
||||
pitch, format}`, and nothing more.
|
||||
[`BootInformation.framebuffer`](../../system/boot-handoff.zig): `{base, width, height,
|
||||
pitch, format, refresh_hz}`, and nothing more.
|
||||
|
||||
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||
([`-device VGA,edid=on`](../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
([`-device VGA,edid=on`](../../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||
VRAM BAR. danos already decodes this device
|
||||
([pci-class.zig](../system/devices/pci-class.zig) has the full `display` namespace, and
|
||||
([pci-class.zig](../../library/device/pci/pci-class.zig) has the full `display` namespace, and
|
||||
`pci-bus` already reports it to the [device manager](device-manager.md) with its class
|
||||
triple) — but nothing binds it yet.
|
||||
|
||||
@@ -63,48 +63,50 @@ rest of the system hasn't had to face:
|
||||
|
||||
1. **The framebuffer is kernel-only today.** It arrives through the boot handoff, is
|
||||
mapped into the kernel's physmap, and is touched only by
|
||||
[`console.zig`](../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../system/kernel/devices-broker.zig) node, so
|
||||
[`console.zig`](../../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../../system/kernel/devices-broker.zig) node, so
|
||||
`device.claim`/`mmio_map` cannot reach it, and there is no framebuffer
|
||||
[syscall](syscall.md). A user-space display service needs a **new mechanism just to
|
||||
[syscall](../os-development/syscall.md). A user-space display service needs a **new mechanism just to
|
||||
touch the pixels**. (See "The handoff" below — this is built.)
|
||||
|
||||
2. **danos has no cross-process shared memory.** The memory syscalls are `mmap`
|
||||
2. **danos had no cross-process shared memory.** At v1 the memory syscalls were `mmap`
|
||||
(private, zeroed), `mmio_map` (a *claimed device's* MMIO), and `dma_alloc` (new
|
||||
pinned physical). The block driver's "pass a buffer by physical address" trick
|
||||
([block/protocol.zig](../system/services/block/protocol.zig)) works *only because its
|
||||
([block/protocol.zig](../../library/protocol/block/block-protocol.zig)) works *only because its
|
||||
consumer is DMA hardware*. A compositor that CPU-reads and blends client layers can't
|
||||
use it — it would have to *map* another process's memory, which nothing allows. This
|
||||
is deferred (see "What v1 does not do"), because v1 sidesteps it entirely.
|
||||
use it — it would have to *map* another process's memory, which nothing allowed. v1
|
||||
sidesteps it entirely (see "What v1 does not do"); v2 has since built the primitive
|
||||
(`shared_memory_create` / `shared_memory_map` / `shared_memory_physical` —
|
||||
[display-v2.md](display-v2.md)).
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
kernel ── owns the boot framebuffer; bootstrap console only
|
||||
│ seeds a "display0" device node from BootInformation.framebuffer
|
||||
│ seeds a display-class device node from BootInformation.framebuffer
|
||||
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
||||
│ plus DisplayInfo{width, height, pitch, format})
|
||||
│ plus DisplayInfo{width, height, pitch, format, refresh_hz})
|
||||
▼
|
||||
display service (system/services/display/, ServiceId.display) ← the compositor
|
||||
│ device.claim(display0) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||
│ device.claim(display node) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||
│ mmap(cacheable) a BACK buffer of the same geometry
|
||||
│ owns: an ordered LAYER STACK + a per-frame DAMAGE list
|
||||
│ owns: an ordered LAYER STACK + a per-frame DAMAGE tracker (rect list or tile grid)
|
||||
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
||||
│ backend is an INTERNAL interface: {gop-fb} today; {bochs-dispi, virtio-gpu} later
|
||||
│ backend is an INTERNAL interface: {gop-fb} at boot; {virtio-gpu} on hot-attach (v2)
|
||||
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
||||
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||
drawing clients (v1) surface clients (deferred)
|
||||
runtime.display commands: runtime.display surfaces:
|
||||
create_layer / configure_layer shm_create → pass as a capability →
|
||||
display commands: display surfaces:
|
||||
create_layer / configure_layer shared_memory_create → pass as a capability →
|
||||
fill_rect / blit_tile / damage the compositor maps & composites the
|
||||
present client-rendered bitmap directly
|
||||
```
|
||||
|
||||
The bring-up sequence mirrors a hardware driver's — it is the
|
||||
[`usb-xhci-bus` `initialise`](../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
[`usb-xhci-bus` `initialise`](../../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
(claim → `mmio_map` → run loop) — and the request/reply service shell is the
|
||||
[FAT](../system/services/fat/fat.zig) / [input](../system/services/input/input.zig) shape
|
||||
([`runtime.service.run`](../library/runtime/service.zig) with a `protocol.zig` of
|
||||
[FAT](../../system/services/fat/fat.zig) / [input](../../system/services/input/input.zig) shape
|
||||
([`service.run`](../../library/kernel/service.zig) with a `protocol.zig` of
|
||||
`extern struct` messages and an `Operation` tag).
|
||||
|
||||
**One process, for now.** v1 is a *single* service that both owns the framebuffer and
|
||||
@@ -117,20 +119,23 @@ second backend or a second monitor appears; until then it is complexity with no
|
||||
|
||||
The framebuffer crosses into user space through the machinery that already exists for
|
||||
every other device, rather than a bespoke syscall — so it inherits ownership,
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](resilience.md)
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](../os-development/resilience.md)
|
||||
story: a crashed display service returns the LFB to the kernel, and its restart
|
||||
re-claims it).
|
||||
|
||||
- The kernel seeds a synthetic **`display0`** node into the
|
||||
[devices-broker](../system/kernel/devices-broker.zig) at init, from
|
||||
- The kernel seeds a synthetic **display-class** node into the
|
||||
[devices-broker](../../system/kernel/devices-broker.zig) at init (`seedDisplay`), from
|
||||
`BootInformation.framebuffer`: one `ResourceKind.memory` resource spanning
|
||||
`[base, height*pitch]`, tagged **write-combining**, plus a small
|
||||
`DisplayInfo{width, height, pitch, format}` (the memory resource says *where* and *how
|
||||
big*; `DisplayInfo` says how to *interpret* the bytes).
|
||||
`DisplayInfo{width, height, pitch, format, refresh_hz}` (the memory resource says *where*
|
||||
and *how big*; `DisplayInfo` says how to *interpret* the bytes — and `refresh_hz`, the
|
||||
panel refresh the loader computed from EDID before `ExitBootServices`, seeds the
|
||||
compositor's frame clock). The node carries no name or index; it is identified purely by
|
||||
its `display` device class.
|
||||
- The service `device.claim`s it and `mmio_map`s the resource. The map is
|
||||
**write-combining**, not the strong-uncacheable that `mmio_map` uses for register
|
||||
MMIO. The kernel already programs a WC PAT slot for its own console
|
||||
([`setupPat`](../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||
([`setupPat`](../../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||
the user mapping path. **This matters:** an uncacheable framebuffer makes the
|
||||
back→front blit unusably slow.
|
||||
- On `claim`, the kernel's bootstrap console goes quiet, so the two never fight over the
|
||||
@@ -138,8 +143,8 @@ re-claims it).
|
||||
panic on screen wins.
|
||||
|
||||
The display service is a **named boot service**: `init` spawns it by name alongside
|
||||
`vfs`/`input`/`device-manager` ([init.zig](../system/services/init/init.zig)), and it
|
||||
self-discovers `display0` with `device.enumerate`. The [device manager](device-manager.md)
|
||||
`input`/`device-manager`/`fat` ([init.zig](../../system/services/init/init.zig)), and it
|
||||
self-discovers the display node with `device.enumerate` (matching on `DeviceClass.display`). The [device manager](device-manager.md)
|
||||
matching path (PCI class 0x03 → a driver) is reserved for the future *native* backend, not
|
||||
this singleton synthetic node.
|
||||
|
||||
@@ -155,8 +160,8 @@ Two buffers, with deliberately different memory types:
|
||||
|
||||
So a frame is: compose every dirty layer into the cacheable back buffer, then **present**
|
||||
— copy the changed regions back→front in sequential, WC-friendly writes. Two details the
|
||||
[framebuffer](framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](gop.md).
|
||||
[framebuffer](../os-development/framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](../os-development/gop.md).
|
||||
|
||||
## Flicker vs. tearing — what double buffering does and doesn't buy
|
||||
|
||||
@@ -182,6 +187,12 @@ discovered.
|
||||
The compositor holds an **ordered stack of layers**. Each layer has a rectangle, a
|
||||
z-order, a visibility flag, and a surface. Presenting walks the stack bottom-to-top,
|
||||
painting each dirty layer into the back buffer, then flushes the damage to the front.
|
||||
Damage is tracked by one of two interchangeable trackers behind a compile-time
|
||||
`damage_mode` A/B switch ([display.zig](../../system/services/display/display.zig)): a
|
||||
free-form dirty-rectangle **list** (tight bounds, heuristic merging) or a fixed 64-px
|
||||
**tile grid** (exact O(1) merging, tile-quantized repaints) — the grid is the default;
|
||||
[compositor.zig](../../system/services/display/compositor.zig) has both, with the trade-off
|
||||
discussion.
|
||||
|
||||
In v1 the surfaces are **server-owned**, and clients draw into them with a small
|
||||
immediate-mode command protocol — essentially the model early X used, and enough for a
|
||||
@@ -196,20 +207,60 @@ shell, a terminal, a cursor, and a wallpaper:
|
||||
| `fill_rect` | fill a rectangle of a layer with a colour |
|
||||
| `blit_tile` | copy a small client-supplied pixel tile into a layer (inline) |
|
||||
| `damage` | mark a region of a layer dirty |
|
||||
| `present` | composite dirty layers and flush to the screen |
|
||||
| `present` | request a repaint: composited at the next frame-clock tick |
|
||||
|
||||
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||
(the [PSF font](../system/kernel/font.psf) path the console already uses can move into a
|
||||
(the [PSF font](../../system/kernel/font.psf) path the console already uses can move into a
|
||||
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||
policy in the client.
|
||||
|
||||
## `runtime.display`
|
||||
`present` is a *request*, not an immediate flush: the compositor runs a ~60 Hz **frame
|
||||
clock** (a one-shot kernel timer re-armed on demand), and each tick composites all the
|
||||
damage accumulated since the last one. Any number of client presents and cursor moves
|
||||
inside one interval coalesce into a single repaint — the software stand-in for vblank
|
||||
pacing on backends that have none (all of them today; see
|
||||
[display-v2.md](display-v2.md), "Fenced is not vsync"). Bring-up paths that must put
|
||||
pixels on screen synchronously (initialisation, the self-checks) bypass the clock.
|
||||
|
||||
Clients speak the protocol through a new [`library/runtime/display.zig`](../library/runtime/runtime.zig),
|
||||
the [`runtime.block`](../library/runtime/block.zig) shape (a cached `.display` lookup
|
||||
## `display`
|
||||
|
||||
Clients speak the protocol through a new [`library/client/display/display.zig`](../../library/client/display/display.zig),
|
||||
the [`block`](../../library/device/block/block.zig) shape (a cached `.display` lookup
|
||||
with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitTile` /
|
||||
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||
runtime, as with every other danos service.
|
||||
client module, as with every other danos service.
|
||||
|
||||
## The cursor: a mouse-listener thread feeding the compositor
|
||||
|
||||
The compositor is the single owner of the framebuffer — only the main `service.run` loop
|
||||
touches the backend and the layer stack. Tracking the mouse without breaking that
|
||||
ownership is the display's first use of [threads](../os-development/threading.md): the service is built
|
||||
multi-threaded (`addThreadedUserBinary`) and, at startup, spawns a **mouse-listener
|
||||
thread** beside the compositor loop.
|
||||
|
||||
- **Listener thread.** Blocks on the input service's mouse stream
|
||||
(`input.subscribeMouse()`), accumulates the relative `dx`/`dy` motion into an absolute
|
||||
cursor position clamped to the screen, and hands it to the compositor. It never touches
|
||||
the compositor — so no lock guards the framebuffer. A parked `next()` leaves its core
|
||||
free to halt ([halting.md](../os-development/halting.md)).
|
||||
- **The channel.** A single-slot *latest-value* cell (`CursorChannel`) guarded by a
|
||||
`Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
every delta, so a new position overwrites the old. The listener also **pokes** the
|
||||
compositor awake — the main loop is parked in `replyWait`, so the listener posts a
|
||||
zero-payload `ipc.send` to the compositor's endpoint, which arrives as a
|
||||
message-notification ([ipc.md](ipc.md)). The poke is *coalesced*: at most one is queued
|
||||
while the main loop has not drained the last, so a fast mouse cannot flood the endpoint.
|
||||
- **Render.** On the poke, the main loop takes the latest position and moves the cursor —
|
||||
which is just a top-z compositor layer — with the existing `configure` + `present` path
|
||||
(it damages the old and new footprints, so only those two rectangles repaint).
|
||||
|
||||
Two threading facts shape this (both in [threading.md](../os-development/threading.md)). IPC **handles do
|
||||
not cross threads**, so the listener can't reuse the main loop's endpoint handle — it
|
||||
`ipc.lookup(.display)`s its *own* handle to the same endpoint to poke through. And a
|
||||
multi-threaded service doing concurrent IPC is why the kernel's endpoint-create / register
|
||||
/ lookup syscalls now serialize under the big kernel lock. Shared fate applies: a fault in
|
||||
the listener takes the whole display down, and the supervisor restarts the process
|
||||
([resilience.md](../os-development/resilience.md)).
|
||||
|
||||
## What v1 does not do (and why that's fine)
|
||||
|
||||
@@ -218,21 +269,22 @@ both are clean additions behind the interfaces v1 establishes.
|
||||
|
||||
- **Client-rendered surfaces (shared memory).** The fast path for a bitmap-heavy app is
|
||||
to render into its *own* buffer and hand the compositor a *reference*, not a stream of
|
||||
commands. That needs the missing cross-process shared-memory primitive — best built as
|
||||
the natural generalization of the existing M13 [capability passing](driver-model.md)
|
||||
from *endpoints* to *memory objects* (`shm_create(len) → {cap, virtual_address}`, pass `cap` on
|
||||
an `ipc_call`, receiver `shm_map(cap) → virtual_address`). v1 avoids it because server-owned
|
||||
surfaces already prove the whole pipeline.
|
||||
commands. That needs a cross-process shared-memory primitive — the natural
|
||||
generalization of the existing M13 [capability passing](driver-model.md)
|
||||
from *endpoints* to *memory objects* (`shared_memory_create(len) → {cap, virtual_address}`, pass `cap` on
|
||||
an `ipc_call`, receiver `shared_memory_map(cap) → virtual_address`). v1 avoids it because server-owned
|
||||
surfaces already prove the whole pipeline; v2 has since built exactly that primitive
|
||||
([display-v2.md](display-v2.md)) — the client-surface path on top of it is still open.
|
||||
|
||||
- **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing
|
||||
resolution / bpp at runtime needs the raw PCI device. The first native backend is
|
||||
Bochs DISPI — the register interface QEMU's `-device VGA` exposes — behind the same
|
||||
resolution / bpp at runtime needs the raw PCI device. The first native backend — since
|
||||
built by v2 ([display-v2.md](display-v2.md)) — is virtio-gpu, behind the same
|
||||
internal backend interface the dumb framebuffer sits behind. Refresh-rate and colour
|
||||
management (a gamma LUT) are real-GPU-KMS territory, far beyond this.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Three QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||
Four QEMU test cases ([tests.zig](../../system/kernel/tests.zig), `python3
|
||||
test/qemu_test.py <case>`), each layering on the last:
|
||||
|
||||
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||
@@ -245,19 +297,28 @@ test/qemu_test.py <case>`), each layering on the last:
|
||||
layer — logging `display: compositor self-check ok`.
|
||||
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||
[`display-demo`](../system/services/display-demo/) client (the
|
||||
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper, a
|
||||
sliding rectangle, a cursor — through the layer client API and heartbeats
|
||||
[`input-source`](../test/system/services/input-source/) analog) drives layers — a wallpaper and
|
||||
a sliding rectangle — through the layer client API and heartbeats
|
||||
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||
the [input test](input.md) proves an event travels source → service → subscriber. The
|
||||
visible motion itself is a screenshot away via `zig build run-x86-64`.
|
||||
the [input test](input.md) proves an event travels source → service → subscriber. It draws
|
||||
no cursor and reads no input — the cursor is the service's own (below), and the demo
|
||||
animates on its own frame timer, independent of the mouse (the test spawns `input`
|
||||
alongside it to keep that independence honest). The visible motion itself is a screenshot
|
||||
away via `zig build run-x86-64`.
|
||||
- **`display-cursor`** — the mouse-listener thread end to end: with the `input` service up,
|
||||
`input-source mouse` publishes pure motion, and the display's listener thread accumulates
|
||||
it into a cursor position handed to the render loop over the `CursorChannel`. Once the
|
||||
cursor has tracked a run of that motion, the service logs
|
||||
`display: cursor tracking mouse ok`. Runs `smp: 4` — the compositor and listener threads
|
||||
execute on different cores, which is what surfaced the IPC-under-lock requirement above.
|
||||
|
||||
The compositor's pixel math (rectangle clipping, fill, composite, tile blit) and colour
|
||||
packing are additionally covered by pure host unit tests under `zig build test`.
|
||||
|
||||
## See also
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [framebuffer.md](../os-development/framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](../os-development/gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [input.md](input.md) — the sibling service; the async `ipc_send` fan-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map`, capability passing, the trust model.
|
||||
- [device-manager.md](device-manager.md) — matching and supervision (the native backend's route).
|
||||
@@ -33,8 +33,8 @@ plain bus driver with no controller — a USB hub — is also a real thing.
|
||||
## The device table is the spine
|
||||
|
||||
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||
`DeviceDescriptor`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||
seeds it ([discovery.md](../os-development/discovery.md)); `device_register` grows it.
|
||||
|
||||
Three invariants make it a capability system rather than a directory:
|
||||
|
||||
@@ -67,7 +67,7 @@ const base = dev.mmioMap(bus.id, 0).?; // 2. enumerate it — from the har
|
||||
const n = ((cap.* >> 8) & 0x1F) + 1; // GENERAL_CAP says how many children
|
||||
|
||||
for (0..n) |i| { // 3. publish each child
|
||||
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||
var child = std.mem.zeroes(dev.DeviceDescriptor);
|
||||
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = .{ .kind = memory,
|
||||
@@ -95,36 +95,62 @@ A "family" is two modules, not one:
|
||||
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||
*whatever* published its device. This is the part that makes class drivers portable.
|
||||
|
||||
danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||
[`system/services/vfs/protocol.zig`](system/services/vfs/protocol.zig) is a protocol module shared by `system/services/vfs/vfs.zig`
|
||||
and its clients. The pattern generalises directly:
|
||||
danos already has one of each: `library/device/pci/pci.zig` is a logic module (the
|
||||
`Function` view of a claimed PCI function),
|
||||
[`library/protocol/vfs/vfs-protocol.zig`](../../library/protocol/vfs/vfs-protocol.zig) is a
|
||||
protocol module shared by the mount backends (today the fat server) and their clients.
|
||||
(The user-space VFS server it was originally written against has since retired — path
|
||||
routing moved into the kernel, `system/kernel/vfs.zig`'s `fs_resolve` — but the protocol
|
||||
module outlived it, which is rather the point.) The pattern generalises directly:
|
||||
|
||||
```
|
||||
library/
|
||||
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||
bus/
|
||||
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||
usb/ module "usb" — descriptors, control transfers, hubs
|
||||
proto/
|
||||
vfs/ module "vfs-protocol" (today: system/services/vfs/protocol.zig)
|
||||
block/ module "block-protocol"
|
||||
hid/ module "hid-protocol"
|
||||
kernel/ the system library (kernel32-style): the syscall surface split by concern
|
||||
— ipc, memory (heap/dma/shared-memory), process, time, logging,
|
||||
file-system, thread, service, plus system-call stubs + start/root
|
||||
device/ device code grouped by domain; each domain splits into a shareable
|
||||
data module (enums/wire types, std-only) and a logic module (mmio/IPC)
|
||||
mmio/ module "mmio" — typed volatile register access + barriers [M14]
|
||||
model/ module "device-abi" — DeviceDescriptor, DeviceClass, ResourceKind
|
||||
pci/ "pci-class" (data) + "pci" — config/BAR/capability walk (Function)
|
||||
usb/ "usb-abi" + "usb-ids" (data) + "usb" — descriptors, control/interrupt/bulk client
|
||||
acpi/ "acpi-ids" (data) + "aml" — _HID names, the AML interpreter
|
||||
driver/ module "driver" — device-access syscalls + device-manager hello
|
||||
block/ module "block" — the block-device client (a device type)
|
||||
client/ userspace service clients — display, input (a program's view of a service)
|
||||
protocol/ driver <-> service wire contracts, one module per directory
|
||||
vfs/ block/ display/ scanout/ input/ power/ device-manager/ usb-transfer/
|
||||
|
||||
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||
block/ class driver imports runtime, block-protocol
|
||||
usb-xhci-bus/ HCD + bus driver imports usb, mmio, usb-transfer-protocol (+ kernel modules)
|
||||
usb-hid/ class driver imports usb, input-protocol (+ kernel modules)
|
||||
virtio-gpu/ scanout driver imports pci, mmio, display-/scanout-protocol (+ kernel modules)
|
||||
```
|
||||
|
||||
The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one
|
||||
module (`rt_mod`) and injects it. It should take a slice of modules. That's a
|
||||
five-line change, and it's the *entire* mechanism — Zig modules already give you
|
||||
everything else.
|
||||
The split by *dependency weight* is what lets the microkernel stay out of device
|
||||
business: it imports only the `device-abi` data module (the descriptor types its broker
|
||||
marshals across the syscall boundary) — never a logic module, never a taxonomy. That one
|
||||
pure-data import is the only edge from `system/kernel/` into `library/`; decoding a class
|
||||
code or `_HID` to a name is user space's job (the device manager owns those taxonomies).
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
A protocol lives in `library/protocol/` when it is the seam between a low-level driver and
|
||||
a higher-level service (block ↔ filesystem, a scanout driver ↔ the compositor). A driver's
|
||||
private wire to its *hardware* — virtio-gpu's command set — is not that; it stays a
|
||||
driver-private file, like the virtio-pci transport beside it.
|
||||
|
||||
The build side of this has since landed: [`addUserBinary`](build.zig) injects the
|
||||
default modules — the library/kernel concern modules (`ipc`, `memory`, `process`, `time`,
|
||||
`logging`, `file-system`, `thread`, `service`), the device/service clients (`driver`,
|
||||
`block`, `display`, `input`), plus `mmio`, `xkeyboard-config`, `acpi-ids` — into every user
|
||||
binary, and per-binary extras — protocol modules, bus logic — are added with
|
||||
`programModule(exe).addImport(...)`. That's the *entire* mechanism — Zig modules
|
||||
already give you everything else.
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's *hardware*
|
||||
logic module.** `usb-hid` imports `usb` (the transfer client) and `input-protocol`, never
|
||||
`pci` and never `mmio`. If a class driver needs `mmio`, it has become an HCD and should be
|
||||
one. The domain data modules (`usb-abi`, `usb-ids`, `pci-class`) carry no such weight — a
|
||||
class driver, the device manager, or the kernel may share them freely.
|
||||
|
||||
## What exists today
|
||||
|
||||
@@ -132,20 +158,24 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
grants, `device_grant` teardown.
|
||||
- **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before
|
||||
EOI; `irq_ack` is the unmask.
|
||||
- **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment.
|
||||
- **M12** — `parent` in `DeviceDescriptor`, `device_register` with resource containment.
|
||||
- **M13** — capability passing. `ipc_call` / `ipc_reply_wait` grew a `send_cap` argument
|
||||
and a `received_cap` return (r8): an endpoint travels with a message, installed into
|
||||
the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full
|
||||
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
|
||||
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
|
||||
`callCap` and `replyWait(..., send_cap)`; no class driver consumes it yet.
|
||||
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
|
||||
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
|
||||
driver mints a per-device endpoint and hands it to a class driver. The `ipc` module
|
||||
exposes `callCap` and `replyWait(..., send_cap)`, and class drivers consume them now: the
|
||||
PS/2 keyboard and mouse drivers attach to ps2-bus this way, and the `usb` / `input`
|
||||
client modules open their per-device and subscription channels with `callCap`.
|
||||
- **M14** — DMA memory + the memory-ordering layer. `/lib/device/mmio` gives drivers typed
|
||||
volatile access and `memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier` (per-arch); `dma_alloc`/`dma_free` grant
|
||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`;
|
||||
no DMA driver consumes `dma_alloc` yet.
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/device/mmio`,
|
||||
and `dma_alloc` has real consumers now: the xHCI driver's rings and contexts,
|
||||
usb-storage's command/status wrappers, virtio-gpu's virtqueue, and the fat
|
||||
service's bounce buffer.
|
||||
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||
with no new syscall), and `msi_bind(device_id, endpoint) -> address, data` allocates a
|
||||
@@ -169,14 +199,17 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||
([sysv.md](sysv.md)). This is what
|
||||
([sysv.md](../os-development/sysv.md)). This is what
|
||||
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||
spawn capability is future work.
|
||||
|
||||
So: **bus drivers work now, and they're started by the device manager, not the kernel.**
|
||||
HCDs and class drivers do not work yet. Here is exactly why, and exactly what would fix it.
|
||||
So: **all three shapes work now, and they're started by the device manager, not the
|
||||
kernel.** The xHCI driver is the HCD-and-bus proof; the USB HID/storage and PS/2 class
|
||||
drivers reach their devices purely over IPC. What follows are the original design notes
|
||||
for the primitives that unblocked each shape — exactly why each was the blocker, and
|
||||
exactly what fixed it.
|
||||
|
||||
---
|
||||
|
||||
@@ -226,7 +259,7 @@ const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||
|
||||
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
|
||||
|
||||
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
|
||||
*Implemented: `/lib/device/mmio` (typed volatile access + `memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier`, per-arch) and
|
||||
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
|
||||
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
|
||||
PAT, a small follow-up. The rest of this section is the original design note.*
|
||||
@@ -268,31 +301,31 @@ doorbell.* = i; // volatile store to UC MMIO
|
||||
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||
```
|
||||
|
||||
So the rules, which belong in `library/mmio.zig` and behind `arch`:
|
||||
So the rules, which belong in `library/device/mmio/mmio.zig` and behind `arch`:
|
||||
|
||||
| Situation | Required |
|
||||
|---|---|
|
||||
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||
| MMIO write that must complete before the next read | `mb()` |
|
||||
| Fill DMA descriptor, then ring doorbell | `writeMemoryBarrier()` between them |
|
||||
| Woken by IRQ, then read what the device wrote | `readMemoryBarrier()` before the read |
|
||||
| MMIO write that must complete before the next read | `memoryBarrier()` |
|
||||
|
||||
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||
sprinkling of `asm volatile`:
|
||||
|
||||
| | x86_64 | aarch64 |
|
||||
|---|---|---|
|
||||
| `mb()` | `mfence` | `dsb sy` |
|
||||
| `rmb()` | `lfence` | `dsb ld` |
|
||||
| `wmb()` | `sfence` | `dsb st` |
|
||||
| `memoryBarrier()` | `mfence` | `dsb sy` |
|
||||
| `readMemoryBarrier()` | `lfence` | `dsb ld` |
|
||||
| `writeMemoryBarrier()` | `sfence` | `dsb st` |
|
||||
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||
|
||||
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||
with a compiler barrier alone. ARM is not, and [vision.md](../vision.md) makes ARM the win
|
||||
condition. Build the abstraction while there is one caller to fix.
|
||||
|
||||
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||
barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
|
||||
barrier, or per-arch inline asm — which is what `library/device/mmio/mmio.zig` should hide.)
|
||||
|
||||
## M15 — interrupts for PCI devices ✅ done (MSI)
|
||||
|
||||
@@ -302,8 +335,10 @@ barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide
|
||||
rather than an out-struct. The rest of this section is the original design note.*
|
||||
|
||||
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception —
|
||||
`addBars` (then in the kernel's ACPI discovery; BAR decode has since moved to the
|
||||
ring-3 pci-bus driver, `system/drivers/pci-bus/pci-bus.zig`) records `.memory` and
|
||||
`.io_port` BARs and never an `.irq`; there is no `_PRT` parsing anywhere in the tree.
|
||||
The HPET is the one exception —
|
||||
it advertises its own interrupt routing in its own registers, a privilege no ordinary
|
||||
device has.
|
||||
|
||||
@@ -363,6 +398,6 @@ from hand-rolling `*volatile` and getting ARM wrong.
|
||||
## See also
|
||||
|
||||
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||
- [discovery.md](../os-development/discovery.md) / [acpi.md](../os-development/acpi.md) — where the device table comes from.
|
||||
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
- [resilience.md](../os-development/resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
@@ -4,13 +4,13 @@ In a monolithic kernel a driver is a function call away from everything: it runs
|
||||
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||
can crash without taking the kernel with it, and — the point of this document — it
|
||||
can be restarted ([resilience](resilience.md)).
|
||||
can be restarted ([resilience](../os-development/resilience.md)).
|
||||
|
||||
That leaves three questions the kernel has to answer, because a process can't answer
|
||||
them for itself:
|
||||
|
||||
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||
([discovery](discovery.md), [acpi](acpi.md)).
|
||||
([discovery](../os-development/discovery.md), [acpi](../os-development/acpi.md)).
|
||||
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||
device's physical MMIO window into your address space, and from then on it's plain
|
||||
memory. No syscall per register access.
|
||||
@@ -31,8 +31,8 @@ kernel ──spawns──► init (PID 1) ──spawns──► device-manag
|
||||
| | |
|
||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||
publishes the starts the system /system/devices, matches each device
|
||||
initial-ramdisk services (vfs, the to a driver, and system_spawn's it
|
||||
so user space can device-manager). Its
|
||||
initial-ramdisk services (device-manager, to a driver, and system_spawn's it
|
||||
so user space can fat, logger, ...). Its
|
||||
system_spawn from it list is init policy.
|
||||
```
|
||||
|
||||
@@ -42,15 +42,22 @@ in the initial-ramdisk as a fresh ring-3 process — `name` becoming its argv[0]
|
||||
the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Everything else is a user-space decision:
|
||||
|
||||
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||
supervisor**. It spawns the system services danos brings up at boot — today `vfs` and
|
||||
the `device-manager` — from a small list. Drivers are deliberately *not* its job.
|
||||
supervisor**. It spawns the system services danos brings up at boot — today `input`,
|
||||
the `device-manager`, `fat`, `display`, `display-demo`, and the `logger` — from a
|
||||
small list. Drivers are deliberately *not* its job. (An earlier draft listed a `vfs`
|
||||
service here; that service is retired — the router moved into the kernel as
|
||||
`fs_resolve`.)
|
||||
- **device-manager** ([system/services/device-manager](system/services/device-manager/device-manager.zig))
|
||||
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||
its probe path, entirely from ring 3:
|
||||
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||
ACPI/PCI ([discovery](discovery.md)).
|
||||
2. **Match** — for each device it looks up a driver by `DeviceClass`. The match policy
|
||||
is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads
|
||||
ACPI/PCI ([discovery](../os-development/discovery.md)).
|
||||
2. **Match** — for each device it looks up a driver. The match policy is code, a few
|
||||
small per-bus tables: from the boot snapshot only the PCI host bridge matches
|
||||
(→ `pci-bus`); everything else arrives later as bus reports and matches on
|
||||
identity — `pciDriverForIdentity` (xHCI → `usb-xhci-bus`, virtio-gpu →
|
||||
`virtio-gpu`), `hidDriverFor` (PNP0303/PNP0F13 → `ps2-bus`), and
|
||||
`usbDriverForIdentity` (USB keyboard, mouse, storage). A fuller system reads
|
||||
what each driver *binds* (a manifest under `/system/drivers`, or the driver
|
||||
describing its own match).
|
||||
3. **Spawn** — `system_spawn(driver_name, arguments)` starts the matched driver (the
|
||||
@@ -68,7 +75,7 @@ capability yet.
|
||||
## The capability: claim before touch
|
||||
|
||||
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
(`library/device/model/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
|
||||
| # | Call | Meaning |
|
||||
|---|------|---------|
|
||||
@@ -84,7 +91,7 @@ names a device by id and a resource by index. That indirection is the entire sec
|
||||
model. If `mmio_map` took a physical address, any process could map the kernel's
|
||||
memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and
|
||||
silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and
|
||||
the same check at the top of `sysMmioMap`):
|
||||
the same check at the top of `systemMmioMap`):
|
||||
|
||||
- `devices_broker.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||
- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq`
|
||||
@@ -95,11 +102,15 @@ The claim is the capability. Everything else follows from it.
|
||||
## Registers: `mmio_map`
|
||||
|
||||
`mmio_map` walks the caller's page tables and installs the device's physical frames
|
||||
with `present | user | writable | nx | pcd | pwt`
|
||||
(`arch/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits are load-bearing:
|
||||
with `present | user | writable | nx | device_grant` plus a cache mode
|
||||
(`system/kernel/architecture/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits
|
||||
are load-bearing:
|
||||
|
||||
- **`pcd | pwt`** — strong-uncacheable. A device register is not memory; a cached read
|
||||
would return a stale value and a write might never leave the CPU.
|
||||
- **`pcd | pwt`** — strong-uncacheable, the default cache mode. A device register is
|
||||
not memory; a cached read would return a stale value and a write might never leave
|
||||
the CPU. The one exception: a resource flagged write-combining
|
||||
(`resource_flag_write_combining` — today the kernel-seeded display framebuffer)
|
||||
gets the PAT bit instead, so pixel writes batch into bursts.
|
||||
- **`device_grant`** (bit 9, one of the PTE's available bits) — marks the leaf as MMIO
|
||||
rather than RAM, so `freeSubtree` skips `pmm.free` on it when the address space is
|
||||
destroyed. Without this, killing a driver would hand the HPET's registers back to
|
||||
@@ -143,7 +154,7 @@ interrupt fires exactly once, ever; call it before the device is quiet and you g
|
||||
interrupt storm. That single fact explains why `irq_bind` and `irq_ack` are two
|
||||
syscalls and not one.
|
||||
|
||||
This is also why `interruptDispatch` (`arch/x86_64/idt.zig`) no longer issues the EOI
|
||||
This is also why `interruptDispatch` (`system/kernel/architecture/x86_64/idt.zig`) no longer issues the EOI
|
||||
itself. It used to, before running the handler — correct for the LAPIC timer, and
|
||||
impossible for a routed device line. Each handler now owns its EOI, because only the
|
||||
handler knows which discipline its source needs.
|
||||
@@ -234,10 +245,10 @@ static capability.
|
||||
A device that *contains other devices* — a PCI bridge, a USB hub, or the HPET's block
|
||||
of comparators — needs a driver that enumerates it and tells the kernel what it found.
|
||||
That's `device_register`, and it makes the device table a tree rather than a list
|
||||
(`DeviceDesc.parent`).
|
||||
(`DeviceDescriptor.parent`).
|
||||
|
||||
```zig
|
||||
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||
var child = std.mem.zeroes(dev.DeviceDescriptor);
|
||||
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = .{ .kind = memory, .start = bus_base + 0x100, .len = 0x20 };
|
||||
@@ -248,15 +259,17 @@ The child is left **unclaimed**, which is the whole point: another process claim
|
||||
`mmio_map`s it, and sees only that 0x20-byte window.
|
||||
|
||||
The rule the kernel enforces is **containment**: every resource of a child must lie
|
||||
inside a resource of the same kind on its parent. Ranges must nest; an IRQ must match
|
||||
exactly. This isn't bureaucracy — a `DeviceDesc` is a licence to map physical memory, so
|
||||
without containment `device_register` would be a syscall for mapping any page you like. A
|
||||
bus driver may only ever subdivide what it already owns.
|
||||
inside a resource of the same kind on its parent. Ranges must nest; a child's IRQ —
|
||||
still exactly one line — must fall within the parent's IRQ range (a length-1 parent
|
||||
range is the old exact-match rule). This isn't bureaucracy — a `DeviceDescriptor` is a
|
||||
licence to map physical memory, so without containment `device_register` would be a
|
||||
syscall for mapping any page you like. A bus driver may only ever subdivide what it
|
||||
already owns.
|
||||
|
||||
A device with **no resources** is legal and common. A USB device is reached through its
|
||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||
drivers and host controller drivers fit together.
|
||||
@@ -277,8 +290,13 @@ Several things this list used to warn about are now available (see
|
||||
[driver-model.md](driver-model.md)): **port I/O** (`io_read`/`io_write`, claim-gated by
|
||||
the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so a PS/2 or
|
||||
16550 driver goes through these), **DMA memory** (`dma_alloc`: contiguous, pinned,
|
||||
uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||
`mb`/`rmb`/`wmb`). What remains:
|
||||
uncacheable, physical address exposed), **memory barriers** (`library/device/mmio`'s
|
||||
`memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier`, imported as the `mmio` module), **fault isolation** (a ring-3 fault kills only the faulting
|
||||
process — `killCurrentProcess` — and the machine keeps running,
|
||||
[resilience](../os-development/resilience.md)), and **reclaim + restart on death** (every path out of a
|
||||
process releases its claims and IRQ/MSI bindings — `releaseAllOwnedBy`,
|
||||
`irq.releaseOwner` — and the device manager respawns the driver with backoff,
|
||||
[device-manager.md](device-manager.md)). What remains:
|
||||
|
||||
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||
@@ -289,8 +307,9 @@ uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||
are programmed, so `device_claim` on a DMA-capable device is still effectively
|
||||
equivalent to granting ring 0. This is the largest gap between the design's promise and
|
||||
what it delivers; enforcement lands with the first DMA driver.
|
||||
- **No `dev_release`.** A claim is never dropped (only IRQ/MSI bindings are, on exit), so
|
||||
a device stays owned for the life of its driver — which blocks restart.
|
||||
- **No voluntary `dev_release`.** A *live* driver can't drop a claim — only exit
|
||||
releases it (any path out of a process runs `releaseAllOwnedBy`) — so handing a
|
||||
device between running drivers still means exiting.
|
||||
- **One endpoint per GSI**, so shared legacy PCI INTx lines can't be split between two
|
||||
drivers. MSI/MSI-X — one vector per device, edge-triggered, unshared — is the real
|
||||
answer, and QEMU's HPET doesn't offer it (`Tn_FSB_INT_DEL_CAP = 0`).
|
||||
@@ -303,13 +322,6 @@ uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||
the ISR until `irq_ack`, so at most one badge is ever outstanding. Bind nine devices
|
||||
to one endpoint, though, and a dropped badge leaves that line masked with nobody
|
||||
left to ack it.
|
||||
- **A faulting driver still kills the machine.** There is no per-process kill path: a
|
||||
ring-3 page fault halts the kernel, so `releaseIrqs` runs only on a voluntary
|
||||
`exit`. Fault isolation is the whole premise ([vision](vision.md)) and it is
|
||||
[not built yet](resilience.md).
|
||||
- **A dead driver's device is not reclaimed.** `releaseIrqs` unbinds and masks the
|
||||
line on exit, but the claim is never released — restart is
|
||||
[not built](resilience.md).
|
||||
- **On real hardware, the mask/EOI cycle may need a remote-IRR flush.** Masking a
|
||||
level-triggered redirection entry with remote-IRR set doesn't clear it on some
|
||||
chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the
|
||||
@@ -358,31 +370,31 @@ is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a PS/2
|
||||
driver possible). What's left is IOMMU *enforcement* (per-device domains — it waits on
|
||||
the first DMA driver to protect and test against) and these smaller items:
|
||||
|
||||
- **Releasing a claim.** There is no `dev_release`, and `devices_broker` never drops a claim on
|
||||
exit — only IRQ bindings are released. A dead driver's device stays owned forever,
|
||||
which blocks restart.
|
||||
- **Unregistering children.** `device_register` only appends. A USB device that is
|
||||
unplugged cannot be removed, and a bus driver in a loop can exhaust the 64-entry
|
||||
table.
|
||||
- **Restart.** A supervisor that *spawns* drivers now exists — the device-manager starts
|
||||
them with `system_spawn` — but a supervisor that *restarts* them does not. A driver that
|
||||
dies should release its claim, have its device quiesced, and be respawned; today nothing
|
||||
notices the death. Some pieces (`releaseIrqs`, `device_grant` teardown, the claim table)
|
||||
exist, and `dev_release` (below) is the missing mechanism; the restart policy is the
|
||||
resilience track ([resilience.md](resilience.md)).
|
||||
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||
a woken driver waits for the next scheduling point.
|
||||
- **Releasing a claim** — half done. The kernel now drops *all* of a dead driver's
|
||||
claims on every path out of a process (`releaseAllOwnedBy`, called from process
|
||||
teardown), which unblocked restart. A voluntary `dev_release` for a live driver
|
||||
still doesn't exist.
|
||||
- **Unregistering children** — half done. Hot-remove works at the manager layer:
|
||||
the xHCI bus reports `child_removed` on unplug and the device manager prunes its
|
||||
tree. The kernel's own device table is still append-only, so a bus driver in a
|
||||
loop can still exhaust the 64-entry table.
|
||||
- **Restart** — done. The device manager notices a driver's death, reads its exit
|
||||
reason, prunes the children it reported, and respawns it with exponential
|
||||
backoff — with a crash-loop cap that marks a repeat offender `failed` instead
|
||||
of respawning forever ([device-manager.md](device-manager.md)).
|
||||
- **Interrupt priority / threaded IRQ latency** — still open. `notifyFromIsr`
|
||||
enqueues the woken driver but doesn't preempt (`wakeLocked` deliberately leaves
|
||||
that to the caller), so a woken driver waits for the next scheduling point.
|
||||
|
||||
## The driver contract (M17–M18)
|
||||
|
||||
Claiming and mapping is half of being a danos driver; the other half is the
|
||||
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||
|
||||
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||
- Build on `service.run` — one replyWait loop folding protocol
|
||||
requests, signals, and notifications into callbacks. The harness answers the
|
||||
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
([process-lifecycle.md](../os-development/process-lifecycle.md)).
|
||||
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||
driver reports what it discovers with `child_added`
|
||||
@@ -5,14 +5,14 @@ window server, a logger. None of them owns the hardware, and the driver should n
|
||||
who is listening. So between the drivers and the listeners sits the **input service**
|
||||
(`system/services/input/`): drivers **publish** events to it, programs **subscribe**, and
|
||||
it fans each event out to every interested subscriber. It is an ordinary ring-3 process
|
||||
reached over IPC, like the [VFS server](../system/services/vfs/vfs.zig) — no kernel knows
|
||||
reached over IPC, like the [FAT server](../../system/services/fat/fat.zig) — no kernel knows
|
||||
what a key is.
|
||||
|
||||
## One service, several device classes
|
||||
|
||||
The service carries three device classes today — **keyboard**, **mouse**, and
|
||||
**joystick/gamepad** — and is built to take more
|
||||
([protocol.zig](../system/services/input/protocol.zig)). Each class has its own typed
|
||||
([protocol.zig](../../library/protocol/input/input-protocol.zig)). Each class has its own typed
|
||||
event:
|
||||
|
||||
- `KeyEvent` — `key_down`/`key_up` (physical make/break) and `key_press` (a character was
|
||||
@@ -43,10 +43,13 @@ consequences decide the whole design:
|
||||
|
||||
2. **A synchronous push can hang the whole service.** If the service delivered with
|
||||
`ipc_call`, it would block until each subscriber replied. `ipc_call` has no timeout, and
|
||||
the kernel does **not** wake a caller parked on a *dead* peer's endpoint (it only fails a
|
||||
peer that was mid-reply — see [process.zig](../system/kernel/process.zig)
|
||||
`releaseTaskResourcesLocked`). One subscriber that exits mid-delivery would wedge input
|
||||
for everyone. That is the opposite of the resilience the microkernel is for.
|
||||
a subscriber's endpoint is an *unregistered* capability the kernel's death path cannot
|
||||
reach (since display v2's V6, `killOwnedEndpointsLocked` in
|
||||
[ipc-synchronous.zig](../../system/kernel/ipc-synchronous.zig) marks a dead owner's
|
||||
*registered* endpoints dead and wakes parked callers with `-EPEER` — but unregistered
|
||||
ones just drop with the task's handle table). One subscriber that exits mid-delivery
|
||||
would wedge input for everyone. That is the opposite of the resilience the microkernel
|
||||
is for.
|
||||
|
||||
The fix is the asynchronous send that [ipc.md](ipc.md) had already earmarked as future
|
||||
work ("asynchronous / buffered send … for notifications between servers"):
|
||||
@@ -63,7 +66,7 @@ the badge (distinguishing it from a bare IRQ/child-exit notification), the sende
|
||||
in the low bits, and the payload in the receive buffer, with no reply owed. The queue holds
|
||||
16 messages per endpoint; a full queue **drops the oldest**, because a buffered message is
|
||||
discrete data, not a coalescing "level" like an interrupt. See
|
||||
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||
[ipc-synchronous.zig](../../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||
the `replyWait` receive loop).
|
||||
|
||||
This is the async counterpart of `ipc_call`, and the input service is its first consumer.
|
||||
@@ -86,7 +89,7 @@ This is the async counterpart of `ipc_call`, and the input service is its first
|
||||
- A **subscriber** calls `input.subscribe(mask)` — or a typed helper: `subscribeKeyboard()`,
|
||||
`subscribeMouse()`, `subscribeJoystick()` (one class, `next()` returns the decoded event),
|
||||
or `subscribeAll()` (every class, `next()` returns a tagged `InputEvent`)
|
||||
([library/runtime/input.zig](../library/runtime/input.zig)). It creates its own endpoint
|
||||
([library/client/input/input.zig](../../library/client/input/input.zig)). It creates its own endpoint
|
||||
and hands it to the service as a **capability** (M13 capability passing — the input
|
||||
service is that feature's first real user), along with its `device_mask`. Then it loops on
|
||||
`next()`, a `replyWait` on that endpoint returning each pushed event.
|
||||
@@ -95,7 +98,7 @@ This is the async counterpart of `ipc_call`, and the input service is its first
|
||||
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||
subscriber.
|
||||
- The **service** ([input.zig](../system/services/input/input.zig)) keeps a small subscriber
|
||||
- The **service** ([input.zig](../../system/services/input/input.zig)) keeps a small subscriber
|
||||
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||
@@ -110,18 +113,18 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
|
||||
- **The keyboard is real.** The `ps2-bus` driver owns PNP0303, which carries *both* the
|
||||
0x60/0x64 ports and IRQ1, so reading the hardware lives in the bus, not in
|
||||
[keyboard.zig](../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||
[keyboard.zig](../../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||
interrupt, drains port 0x60, routing every byte by the status register's
|
||||
auxiliary-output bit to whichever child driver **attached** for that device (an
|
||||
`AttachRequest` to the well-known `ps2_bus` service, carrying the child's endpoint as a
|
||||
capability; the bytes then arrive as asynchronous `ForwardedByte` messages, so the IRQ
|
||||
path never blocks on a child). The keyboard driver decodes the stream — scancode **set 2**,
|
||||
what the keyboard sends with the 8042's legacy translation off, decoded by
|
||||
[scancode.zig](../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||
[scancode.zig](../../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||
make/break, typematic-repeat, and modifier tracking (host-tested under `zig build test`) —
|
||||
and publishes real `key_down`/`key_press`/`key_up` events.
|
||||
- **Keycode → character** is wired in: the keyboard driver fills a `key_press` event's
|
||||
`character` through [`library/xkeyboard-config`](../library/xkeyboard-config/README.md)
|
||||
`character` through [`library/xkeyboard-config`](../../library/xkeyboard-config/README.md)
|
||||
(`xkb.map(layout, keycode, mods)` → keysym + Unicode character), synthesizing the ASCII
|
||||
control characters for Enter/Tab/Backspace/Escape, whose keysyms map to no Unicode. The
|
||||
layout defaults to `us`; the bus can pass another as the driver's argv[2] — the seam for
|
||||
@@ -129,9 +132,9 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
- **The mouse is real too.** IRQ12 is enumerated on the auxiliary device's own ACPI node
|
||||
(PNP0F13), so the bus claims that node alongside the controller and routes both IRQs to
|
||||
its one endpoint, acking whichever line the notification's badge names.
|
||||
[mouse.zig](../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||
[mouse.zig](../../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||
assembles the forwarded bytes with
|
||||
[mouse-packet.zig](../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||
[mouse-packet.zig](../../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||
packets: sync/overflow handling, nine-bit movement, screen-convention `dy` — host-tested
|
||||
under `zig build test`) into `button_down`/`button_up` transitions and `motion` events.
|
||||
**Follow-up:** the IntelliMouse magic-knock for a scroll wheel (four-byte packets) and
|
||||
@@ -147,7 +150,7 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
## Verifying it
|
||||
|
||||
The `input` case (`python3 test/qemu_test.py input`, in
|
||||
[tests.zig](../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||
[tests.zig](../../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||
service, the synthetic source (which cycles keyboard, mouse, and joystick events), and a
|
||||
subscriber that took all three classes. It passes only when the subscriber heartbeats
|
||||
`input-test: ok` — proof that an event travelled source → service → subscriber over IPC,
|
||||
@@ -157,5 +160,5 @@ serial line names the class received, so the log shows all three arriving on one
|
||||
## See also
|
||||
|
||||
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||
- [syscall.md](syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [syscall.md](../os-development/syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||
@@ -1,7 +1,7 @@
|
||||
# IPC: message-passing channels
|
||||
|
||||
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
||||
services run isolated in their own address spaces ([vision](vision.md)), they can't
|
||||
services run isolated in their own address spaces ([vision](../vision.md)), they can't
|
||||
just call each other — a request becomes a **message**. In a microkernel, whatever
|
||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||
concern, not an afterthought.
|
||||
@@ -18,14 +18,14 @@ There are two layers, built a milestone apart:
|
||||
|
||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](scheduling.md).
|
||||
scheduler's [wait queues](../os-development/scheduling.md).
|
||||
|
||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||
ring buffer, a count, and two wait queues:
|
||||
|
||||
- **`send(msg)`** — if the channel is full, block on the *not-full* queue; otherwise
|
||||
write the message, bump the count, and wake a waiting receiver.
|
||||
- **`recv()`** — if the channel is empty, block on the *not-empty* queue; otherwise
|
||||
- **`receive()`** — if the channel is empty, block on the *not-empty* queue; otherwise
|
||||
take a message, drop the count, and wake a waiting sender.
|
||||
|
||||
Neither side busy-waits: a full channel parks the sender, an empty one parks the
|
||||
@@ -36,17 +36,20 @@ Two details make it correct:
|
||||
- **Recheck in a loop.** A woken task re-tests the condition (`while (full) wait`)
|
||||
rather than assuming the slot is still available — another waiter may have taken
|
||||
it first. This is the standard guard against spurious or racing wakeups.
|
||||
- **One critical section.** `send`/`recv` run under `saveInterrupts` /
|
||||
`restoreInterrupts` (the composable form, see [scheduling.md](scheduling.md)), so
|
||||
checking the condition and committing the block/enqueue happen atomically with
|
||||
respect to the timer preempting mid-operation. `waitLocked` / `wakeLocked` are the
|
||||
variants that assume the caller already holds that critical section.
|
||||
- **One critical section.** `send`/`receive` run under the [big kernel
|
||||
lock](../os-development/smp.md) (`sync.enter` / `sync.leave`), which disables interrupts on this
|
||||
core *and* takes the kernel's one spinlock — since SMP, the interrupt flag alone
|
||||
is not atomicity, because `cli` on one core does nothing to another. So checking
|
||||
the condition and committing the block/enqueue happen atomically both with respect
|
||||
to the timer preempting mid-operation and to the other side running on another
|
||||
CPU. `waitLocked` / `wakeLocked` are the variants that assume the caller already
|
||||
holds that critical section.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `ipc` test (see [testing.md](testing.md)) runs a producer and a consumer passing
|
||||
The `ipc` test (see [testing.md](../testing.md)) runs a producer and a consumer passing
|
||||
**100 messages through a 4-slot channel**. The small buffer means the channel goes
|
||||
full and empty over and over, so both the blocking-send and blocking-recv paths are
|
||||
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||
|
||||
@@ -87,13 +90,15 @@ elsewhere is not lost.
|
||||
This is what makes a user-space driver possible at all, and it's the subject of
|
||||
[drivers.md](drivers.md).
|
||||
|
||||
## What's next (not done here)
|
||||
## What's next (partly done since)
|
||||
|
||||
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
||||
low-priority server doesn't suffer unbounded priority inversion.
|
||||
- **Handle transfer.** A server can't hand a client a handle to a third endpoint, so
|
||||
every capability is either well-known (the registry) or inherited — there's no way
|
||||
to delegate one.
|
||||
- **Priority inheritance** through IPC — still open: a high-priority client
|
||||
blocked on a low-priority server suffers unbounded priority inversion.
|
||||
- **Handle transfer.** *Landed as cap-passing (M13)*: `ipc_call` and
|
||||
`ipc_reply_wait` carry an optional capability alongside the bytes (`send_cap`),
|
||||
copying an endpoint or shared-memory handle into the peer's table. First user:
|
||||
[input](input.md) subscribers register by handing over their own endpoint, and
|
||||
class drivers get a private channel to one device.
|
||||
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
||||
non-blocking post to an endpoint's bounded payload queue, delivered through
|
||||
@@ -101,24 +106,27 @@ This is what makes a user-space driver possible at all, and it's the subject of
|
||||
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
||||
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
||||
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
||||
lock; a bulk transfer wants shared pages, not a copy.
|
||||
- **A bounded reply** — half landed. The copy is still 256 bytes
|
||||
(`MESSAGE_MAXIMUM`) under the big kernel lock, but bulk transfer got its shared
|
||||
pages: `shared_memory_create`/`map`/`physical`, the region handle delegated as
|
||||
a capability (above). virtio-gpu's scanout surface is the first user
|
||||
([display-v2.md](display-v2.md)).
|
||||
|
||||
## Lifecycle conventions over IPC (M17)
|
||||
|
||||
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||
Three conventions from [process-lifecycle.md](../os-development/process-lifecycle.md) ride the
|
||||
notification mechanism:
|
||||
|
||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||
`signal_bind` (`process.bindSignals`): badge = the signal bit plus the
|
||||
coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
||||
never questions; no payload, no reply.
|
||||
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||
- **One-shot timers** (`timer_bind`, `time.timerOnce`) land as a
|
||||
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||
serving, instead of blocking in sleep.
|
||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||
answered with a zero-length reply by the service harness itself
|
||||
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||
(`service.run`). No protocol's requests start at length zero, so the
|
||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||
protocol message.
|
||||
@@ -0,0 +1,114 @@
|
||||
# USB hubs (M22)
|
||||
|
||||
A hub is USB **bus infrastructure**, not an application peripheral, so hub
|
||||
topology is handled **inside the `usb-xhci-bus` driver** — the process that owns
|
||||
the controller's device slots and contexts. A device behind a hub is not reached
|
||||
by any hub-specific software path: it is reached by the **controller**,
|
||||
programmed with a *route string* in its slot context. Route strings and slot
|
||||
contexts are xHCI hardware concepts that only exist inside the controller driver,
|
||||
so that is where hub handling belongs. Class drivers (HID, storage) stay separate
|
||||
and unaware — the hub is transparent to them; a keyboard behind a hub reaches the
|
||||
same `usb-hid-keyboard` driver as one on a root port.
|
||||
|
||||
This is a deliberate scoping choice, not a microkernel compromise: the USB *bus*
|
||||
driver handles USB *bus* topology. The alternative — a separate `usb-hub`
|
||||
class-driver process plus a cross-process enumeration protocol — would only
|
||||
shuttle the bus's own topology state (slot ids, route strings, TT linkage) out to
|
||||
another process and back, since the hub driver cannot build a slot context
|
||||
itself.
|
||||
|
||||
## The compound-hub reality
|
||||
|
||||
A USB 3.0 hub is physically **two hubs** sharing each connector: a SuperSpeed hub
|
||||
and a USB 2.0 companion hub, enumerated as **separate devices on separate root
|
||||
ports**. A full- or low-speed device plugged into a USB 3.0 hub attaches to the
|
||||
**USB 2.0 companion**, not the SuperSpeed hub. So supporting full-speed devices
|
||||
(keyboards, mice) behind a hub means driving the USB 2.0 companion and handling
|
||||
**transaction translators** — there is no SuperSpeed-only shortcut that reaches a
|
||||
full-speed keyboard.
|
||||
|
||||
## Slot-context fields for a downstream device
|
||||
|
||||
`buildAddressInputContext` fills the Slot Context from the device record: a
|
||||
root-port device carries route 0 and its own root-hub port. A downstream device
|
||||
additionally carries:
|
||||
|
||||
- **Route String** (Slot Context dword 0, bits 19:0) — 5 tiers × 4 bits, each
|
||||
tier the downstream hub-port number. Composed as
|
||||
`route = (parent_route << 4) | hub_port`, capped at the xHCI 5-tier max.
|
||||
- **Root Hub Port Number** (dword 1, bits 23:16) — the *root* port the whole hub
|
||||
chain hangs off, inherited from the parent hub (not the hub's own port number).
|
||||
- **Speed** (dword 0, bits 23:20) — read from the hub's downstream port status
|
||||
after reset, not assumed.
|
||||
- **Parent Hub Slot ID** (dword 2, bits 7:0) + **Parent Port Number** (dword 2,
|
||||
bits 13:8) — the **transaction translator**: set when a full/low-speed device
|
||||
sits behind a high-speed hub, so the controller routes split transactions
|
||||
through that hub's TT. For a multi-TT hub, **MTT** (Slot Context dword 0 bit
|
||||
25) is set and the TT port is the device's own hub port.
|
||||
|
||||
## Detection: the status-change interrupt endpoint
|
||||
|
||||
A hub has one interrupt IN endpoint that returns a **port-status-change bitmap**
|
||||
(bit N set = port N changed). The bus arms an interrupt transfer on it (reusing
|
||||
the controller's existing interrupt-endpoint machinery, but serviced
|
||||
**in-process** — no class-driver subscription IPC), and on each report:
|
||||
|
||||
1. For each changed port, `GET_STATUS` (hub class request) reads the port's
|
||||
connect/enable/reset state and speed, and `CLEAR_FEATURE(C_PORT_*)`
|
||||
acknowledges the change.
|
||||
2. On a **connect**: `SET_FEATURE(PORT_RESET)`, wait for reset-complete via a
|
||||
later status-change report, read the enabled speed, then `setupDevice` with
|
||||
the composed route string / root port / TT fields, `enumerate`, and register
|
||||
the interfaces — exactly the existing path, recursing if the new device is
|
||||
itself a hub.
|
||||
3. On a **disconnect**: tear down the downstream device (report each interface
|
||||
`ChildRemoved`, Disable Slot) — the B3 teardown path, keyed by the device's
|
||||
route rather than a root port.
|
||||
|
||||
## Hub setup (once, when the hub enumerates)
|
||||
|
||||
When the bus scan (or a hot-plug bring-up) finds a device of class 9:
|
||||
|
||||
1. Read the **hub descriptor** (class GET_DESCRIPTOR, type 0x2A for a USB 3.0
|
||||
hub / 0x29 for USB 2.0) → downstream port count, characteristics.
|
||||
2. For a USB 3.0 hub, `SET_FEATURE(BH_PORT_RESET)` semantics and the depth
|
||||
(`SET_HUB_DEPTH`) so the hub knows its tier for route-string forwarding.
|
||||
3. `SET_FEATURE(PORT_POWER)` each downstream port.
|
||||
4. Configure the hub's slot as a hub: **Hub** bit (Slot Context dword 0 bit 26),
|
||||
**Number of Ports** (dword 1, bits 31:24), **TT Think Time** and **MTT** for a
|
||||
USB 2.0 multi-TT hub — via an Evaluate/Configure Endpoint on the hub's slot.
|
||||
5. Arm the status-change interrupt endpoint.
|
||||
|
||||
## Testing
|
||||
|
||||
QEMU's `usb-hub` is a USB 2.0 single-TT hub. A **static boot topology** on a
|
||||
dedicated second controller (`-device qemu-xhci,id=xhci2 -device
|
||||
usb-hub,bus=xhci2.0,port=1 -device usb-kbd,bus=xhci2.0,port=1.1` — isolated from
|
||||
the boot controller's auto-assigned devices, whose ports the hub would collide
|
||||
with) presents the downstream device connected from the start, so the bus reads
|
||||
it on the first status-change report — exercising the full path (hub setup, TT slot context,
|
||||
downstream enumerate, class-driver bind) without needing a hot-plug event. A new
|
||||
`usb-hub` QEMU case asserts the hub enumerates, the downstream keyboard
|
||||
enumerates behind it, and `usb-hid-keyboard` binds.
|
||||
|
||||
Real-hardware validation (the user's SuperSpeed Genesys hub + full-speed
|
||||
keyboard/mouse on its USB 2.0 companion) is flagged separately — the compound
|
||||
USB 3.0 hub path is not modelled by QEMU's USB 2.0 hub.
|
||||
|
||||
## Milestones (all complete)
|
||||
|
||||
- **B4a** ✓ — hub recognition + setup: detect class 9 in the scan, read the hub
|
||||
descriptor, configure the slot as a hub, power downstream ports, log the
|
||||
topology.
|
||||
- **B4b** ✓ — downstream enumeration: the in-process status-change subscription,
|
||||
port reset, Address Device with route string + root port + TT fields,
|
||||
enumerate + register. A full-speed keyboard behind a USB2 hub binds
|
||||
`usb-hid-keyboard` in QEMU.
|
||||
- **B4c** ✓ — disconnect teardown (recursive: a hub takes its subtree with it)
|
||||
and hub-behind-hub recursion (route strings compose across tiers). QEMU's hub
|
||||
*does* raise downstream status changes, so both connect and disconnect are
|
||||
harness-tested (`usb-hub`, `usb-hub-nested`, `usb-hub-unplug`).
|
||||
|
||||
Real-hardware validation of the user's SuperSpeed Genesys hub with full-speed
|
||||
devices on its USB 2.0 companion remains pending — QEMU's USB 2.0 hub does not
|
||||
model the compound USB 3.0 hub.
|
||||
+23
-16
@@ -1,6 +1,6 @@
|
||||
# DanOS Filesystem Hierarchy Standard (DFHS)
|
||||
|
||||
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. This is provided by virtual file system driver (VFS).
|
||||
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. Root path resolution is provided by the kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted filesystem servers serve the subtrees they own.
|
||||
|
||||
## Directory structure
|
||||
|
||||
@@ -18,8 +18,10 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o
|
||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||
| /system/services | system-service binaries — init, the FAT server, and other user-mode servers (e.g. /system/services/init, /system/services/fat) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /test | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like /system, and its layout likewise mirrors the source tree (the repo's test/ directory). Present on development and test images; a volume without it still boots. |
|
||||
| /test/system/services | test-fixture binaries (e.g. /test/system/services/vfs-test, /test/system/services/thread-test) — the same path in the repo source tree and on the boot volume |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||
@@ -47,11 +49,15 @@ addressed by device id. `/dev` is the much smaller set of devices that have a dr
|
||||
willing to serve them, addressed by name.
|
||||
|
||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
IPC endpoint, and subsequent reads and writes are calls against it. This is what
|
||||
`system/services/vfs/vfs.zig` reserves for M10 and what the `Stat.kind` field is for; **none of it is
|
||||
implemented today.** The current VFS is a flat, in-memory ramfs of eight nodes, with no
|
||||
directories at all and `kind` hardcoded to zero. The three sections below describe the
|
||||
([drivers.md](../device-driver-development/drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
IPC endpoint, and subsequent reads and writes are calls against it. Resolve-to-endpoint
|
||||
is exactly what the kernel's `fs_resolve` already does for any mounted backend, and
|
||||
`FileStatus.kind` is the field that marks a device node; **what is not implemented today
|
||||
is `/dev` itself** — no service mounts it. (The flat eight-node ramfs this section once
|
||||
described is retired: the kernel-resident VFS root in `system/kernel/vfs.zig` serves a
|
||||
read-only initrd mount per top-level tree — `/system`, and `/test` on images that carry
|
||||
the fixtures — with real directories and node kinds, and filesystem
|
||||
backends such as the FAT server mount the rest.) The three sections below describe the
|
||||
intended shape, and are honest about which parts the kernel can already support.
|
||||
|
||||
### Character devices
|
||||
@@ -84,9 +90,9 @@ A block driver is now **writable, but not yet memory-safe.** Every storage contr
|
||||
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||
physical address disclosed — and **`/lib/mmio`**'s barriers order the descriptor writes
|
||||
physical address disclosed — and **`/lib/device/mmio`**'s barriers order the descriptor writes
|
||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||
can be written today (the M14/M15 work in [driver-model.md](driver-model.md); the earlier
|
||||
can be written today (the M14/M15 work in [driver-model.md](../device-driver-development/driver-model.md); the earlier
|
||||
"cannot host a block driver at all" is no longer true).
|
||||
|
||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||
@@ -107,15 +113,16 @@ yielding unpredictable bytes.
|
||||
|
||||
These are the only `/dev` entries danos can implement immediately, and they are the
|
||||
sensible place to start, because they are exactly the entries that need no driver
|
||||
process, no `device_claim`, no MMIO grant and no interrupt. The VFS server answers them
|
||||
out of its own address space — `null` and `zero` are a few lines each in
|
||||
`system/services/vfs/vfs.zig`'s `read` and `write` handlers. Doing so forces the two pieces of
|
||||
structure that every later device node depends on and that the flat ramfs currently
|
||||
lacks: a directory, so that `/dev/null` is a path rather than a name; and a populated
|
||||
`Stat.kind`, so that a caller can tell a character device from a regular file.
|
||||
process, no `device_claim`, no MMIO grant and no interrupt. A future pseudo-device
|
||||
service would answer them out of its own address space — `null` and `zero` are a few
|
||||
lines each in its `read` and `write` handlers — and mount itself at `/dev` the way the
|
||||
FAT server mounts `/mnt/usb`. The two pieces of structure every later device node
|
||||
depends on (and that the flat ramfs of the time lacked) exist now: directories, so that
|
||||
`/dev/null` is a path rather than a name; and a populated `FileStatus.kind`, so that a
|
||||
caller can tell a character device from a regular file.
|
||||
|
||||
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
||||
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
||||
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
||||
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
||||
it until it is a real one.
|
||||
it until it is a real one.
|
||||
@@ -1,20 +1,25 @@
|
||||
# The VFS wire protocol
|
||||
|
||||
> **Status:** built and spoken today between `runtime.fs` (the client) and the
|
||||
> VFS server (`system/services/vfs`), with mounted backends (the FAT server)
|
||||
> speaking the same protocol behind the router. The Zig source of truth is
|
||||
> `system/services/vfs/protocol.zig` (the `vfs-protocol` module), whose unit
|
||||
> tests pin the sizes and values below. This page is the **language-neutral
|
||||
> wire specification** of that contract — what a Rust or C client implements
|
||||
> ([vdso.md](vdso.md) explains why the IPC protocols, not the syscall
|
||||
> numbers, are danos's public ABI).
|
||||
> **Status:** built and spoken today between `file_system` (the client) and the
|
||||
> filesystem BACKENDS (the FAT server). The mount router lives in the
|
||||
> **kernel** (`system/kernel/vfs.zig`): `fs_resolve` routes a path and either
|
||||
> serves it directly (the read-only /system initrd mount, via `fs_node`) or
|
||||
> redirects the caller to the owning backend's endpoint plus the rewritten
|
||||
> mount-relative path — after which the client speaks THIS protocol to the
|
||||
> backend, unchanged. The Zig source of truth is `library/protocol/vfs/vfs-protocol.zig`
|
||||
> (the `vfs-protocol` module), whose unit test pins a sample of the sizes
|
||||
> and values below. This page is the **language-neutral wire specification**
|
||||
> of that contract — what a Rust or C client implements ([vdso.md](../os-development/vdso.md)
|
||||
> explains why the IPC protocols, not the syscall numbers, are danos's
|
||||
> public ABI).
|
||||
|
||||
## Transport
|
||||
|
||||
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
|
||||
docs/ipc.md): the client sends one message and blocks; the server replies
|
||||
with one message. The endpoint is found by well-known service id
|
||||
(`ipc_lookup`, service id **1** = vfs).
|
||||
with one message. The endpoint comes from the kernel's `fs_resolve` — which
|
||||
also hands back the path rewritten relative to the mount — not from a
|
||||
registry lookup. (Service id 1, the old userspace router, is retired.)
|
||||
|
||||
- A message is at most **256 bytes** (`message_maximum`).
|
||||
- A request is a fixed 32-byte **Request** header followed by an inline
|
||||
@@ -26,8 +31,11 @@ with one message. The endpoint is found by well-known service id
|
||||
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||
(`extern struct`), offsets given below so nothing need be inferred.
|
||||
|
||||
The kernel never parses any of this — it only moves the bytes
|
||||
(docs/syscall.md); files are entirely a user-space affair.
|
||||
The kernel resolves NAMES (the mount table) but never parses these
|
||||
messages — it moves the bytes; file state is entirely the backend's affair.
|
||||
With clients holding backend node ids directly, a backend records each open
|
||||
handle's owner and sweeps a dead client's handles via the published process
|
||||
exit events.
|
||||
|
||||
## Request header — 32 bytes
|
||||
|
||||
@@ -50,16 +58,20 @@ The kernel never parses any of this — it only moves the bytes
|
||||
| 16 | 4 | `len` | reply payload length in bytes |
|
||||
| 20 | 4 | — | padding |
|
||||
|
||||
On failure the router replies `status = -1`; a mounted backend's negative
|
||||
status is forwarded to the client verbatim. A richer errno vocabulary is
|
||||
future work — clients must treat *any* negative status as failure, not match
|
||||
on -1.
|
||||
On failure the backend replies `status = -1`, and that reply reaches the
|
||||
client directly — there is no party between them on the wire. (Kernel-served
|
||||
paths produce no wire replies at all: `fs_resolve`/`fs_node` failures are
|
||||
syscall register statuses.) A richer errno vocabulary is future work —
|
||||
clients must treat *any* negative status as failure, not match on -1.
|
||||
|
||||
## Operations
|
||||
|
||||
Values are append-only and never renumbered (the same evolution rule every
|
||||
danos protocol follows); an unrecognised operation gets a `status = -1`
|
||||
reply.
|
||||
danos protocol follows). Send only values from this table: the shipped server
|
||||
decodes the operation into an exhaustive enum, so an out-of-range value is
|
||||
not answered with a `status = -1` reply — it trips a safety check in safe
|
||||
builds and is undefined otherwise. (The `-1` replies cover recognised but
|
||||
refused operations, such as `mount` sent to a backend.)
|
||||
|
||||
| value | operation | request payload | reply |
|
||||
|------:|-----------|-----------------|-------|
|
||||
@@ -77,10 +89,12 @@ reply.
|
||||
|
||||
Notes per operation:
|
||||
|
||||
- **open** — paths are absolute (`/mnt/usb/notes.txt`) or bare names
|
||||
(`greeting`); bare names resolve in the VFS's flat ramfs, absolute paths
|
||||
route through the mount table (below). The returned `node` is an id in the
|
||||
*router's* open table; clients never see a backend's own ids.
|
||||
- **open** — the path is the mount-relative path `fs_resolve` handed back
|
||||
(absolute-shaped: `/notes.txt` under fat's `/mnt/usb` mount). Bare names
|
||||
(`greeting`) resolve nowhere — the flat ramfs is retired, and `fs_resolve`
|
||||
refuses non-absolute paths. The returned `node` is the *backend's* own
|
||||
open-node id: with the router in the kernel there is no forwarding table,
|
||||
and clients hold backend ids directly (see *Lifetimes and trust*).
|
||||
- **read / write** — a single exchange moves at most 224 bytes
|
||||
(`maximum_payload`); the client loops, advancing `offset` by the returned
|
||||
`len`, until done (read) or the slice is written (write). A `write` reply
|
||||
@@ -90,15 +104,20 @@ Notes per operation:
|
||||
position. Each call returns exactly one entry; the client increments the
|
||||
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
|
||||
have been opened with the `directory` flag.
|
||||
- **mount** — the one operation that passes a **capability**: the caller
|
||||
(a filesystem server, e.g. FAT) sends its own request endpoint as the
|
||||
`ipc_call` capability argument, and the router forwards everything under
|
||||
the mount point to it — speaking this same protocol, with paths rewritten
|
||||
relative to the mount. Prefixes match at path boundaries only
|
||||
(`/mnt/usb` never captures `/mnt/usbextra`); the longest matching prefix
|
||||
wins.
|
||||
- **rename** — same-directory rename only (the router requires old and new to
|
||||
resolve under one mount).
|
||||
- **mount / unmount** — RETIRED from the wire: mounting is the `fs_mount`
|
||||
syscall now (a filesystem server passes its endpoint handle; possession is
|
||||
the capability, exactly the trust of the old cap-passing op). The op
|
||||
numbers stay reserved. Mount-prefix semantics are unchanged: prefixes
|
||||
match at path boundaries only (`/mnt/usb` never captures `/mnt/usbextra`),
|
||||
the longest matching prefix wins, and an optional backend-side rewrite
|
||||
prefix maps a mount into the backend's namespace (fat serves `/mnt/usb`
|
||||
from its volume root and `/var` from its `/var` subtree).
|
||||
- **rename** — same-directory rename only: the backend compares the old and
|
||||
new parent paths and refuses a mismatch. The client (`file_system`) refuses
|
||||
earlier when the two paths resolve to different backend endpoints, but that
|
||||
check is coarser than "one mount" — one endpoint can serve several mounts
|
||||
(fat serves `/mnt/usb` and `/var`), so a cross-mount rename reaches the
|
||||
backend and fails on its same-directory check.
|
||||
|
||||
## Open flags
|
||||
|
||||
@@ -148,11 +167,12 @@ table can grow.
|
||||
|
||||
## Lifetimes and trust
|
||||
|
||||
Open-node ids live in the server. A client that dies without closing leaks
|
||||
nothing permanently: the VFS subscribes to the kernel's published process-exit
|
||||
events (docs/process-lifecycle.md) and releases a dead client's handles,
|
||||
closing forwarded backend nodes best-effort. Ids are plain integers, not
|
||||
capabilities — the VFS trusts its callers with each other's ids today, which
|
||||
Open-node ids live in the backend. A client that dies without closing leaks
|
||||
nothing permanently: the backend (the FAT server) subscribes to the kernel's
|
||||
published process-exit events (docs/process-lifecycle.md) and releases a dead
|
||||
client's handles. The kernel VFS root needs no sweep at all — its node tokens
|
||||
are permanent for a boot and carry no open state. Ids are plain integers, not
|
||||
capabilities — a backend trusts its callers with each other's ids today, which
|
||||
is acceptable while every client is part of the system image and worth
|
||||
revisiting (per-client id namespaces) before third-party binaries arrive.
|
||||
|
||||
@@ -161,8 +181,10 @@ revisiting (per-client id namespaces) before third-party binaries arrive.
|
||||
What a non-Zig implementation may rely on, and what it must not:
|
||||
|
||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||
**append-only and frozen once shipped** — the unit tests in `protocol.zig`
|
||||
pin them exactly so a refactor can't silently move them.
|
||||
**append-only and frozen once shipped**. The unit test in
|
||||
`library/protocol/vfs/vfs-protocol.zig` pins a sample of them (the `DirectoryEntry`
|
||||
size, `NodeKind` 0–1, `Operation` values 0, 4 and 5); this page is the
|
||||
full record of the frozen values.
|
||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||
-104
@@ -1,104 +0,0 @@
|
||||
# Logging: the diagnostic log vs. the display
|
||||
|
||||
danos separates two things that are easy to conflate: the **diagnostic log** — the
|
||||
machine-readable stream of *what the kernel is doing* — and the **display**, the
|
||||
framebuffer surface the OS draws on. They are different concerns with different
|
||||
lifetimes, so they're different code paths.
|
||||
|
||||
The guiding rule: **output is a diagnostic convenience, never a correctness
|
||||
dependency.** The kernel must boot and run correctly with *zero* output channels —
|
||||
no serial, no screen. Logging that can take the kernel down isn't robust; it's a
|
||||
liability. This is the same [resilience](resilience.md) posture the rest of the
|
||||
kernel follows.
|
||||
|
||||
## The log is multi-sink
|
||||
|
||||
`system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||
registered **sinks**, each best-effort and self-guarding:
|
||||
|
||||
```zig
|
||||
log.addSink(arch.serialWrite); // the serial UART
|
||||
if (arch.debugconPresent()) log.addSink(arch.debugconWrite); // 0xE9 debug console
|
||||
// later: log.addSink(fs.logWrite); // a file on a ramdisk / USB / SSD
|
||||
log.write("…"); log.print("x={d}\n", .{x});
|
||||
```
|
||||
|
||||
Properties that matter:
|
||||
|
||||
- **No allocation.** The sink table is a fixed array, so the log works before the
|
||||
heap is up and inside a panic.
|
||||
- **Best-effort.** A sink whose device is absent is a no-op (e.g. writing to a
|
||||
missing UART just goes nowhere — the TX-wait is bounded so it can't hang). A
|
||||
message reaches whatever channels exist; if none do, the kernel runs on, silent.
|
||||
- **Order-independent.** Every registered sink gets every message. Adding the file
|
||||
logger later is one `addSink` call and **zero** changes to call sites.
|
||||
|
||||
## The framebuffer is *not* a log sink
|
||||
|
||||
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
||||
Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
||||
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
||||
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
||||
|
||||
Instead the two paths are explicit:
|
||||
|
||||
```
|
||||
verbose diagnostics ──► log ──► serial, debugcon, (file later)
|
||||
user status / panics ──► status() ──► log (above) + framebuffer (if present)
|
||||
```
|
||||
|
||||
A handful of user-facing lines (`kernel initialised`, a panic) go through
|
||||
`main.zig`'s `status()` / `statusPrint()`, which write to the log **and** paint the
|
||||
framebuffer when one is present. Everything else uses `log.*` and never touches the
|
||||
screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
||||
|
||||
## Optional framebuffer (headless machines)
|
||||
|
||||
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||
serial-less machine boots and runs correctly — it just goes quiet.
|
||||
|
||||
## Last-resort channels (no text output at all)
|
||||
|
||||
Two signals bypass the sink list, because they must survive even a total
|
||||
output-channel failure:
|
||||
|
||||
- **`log.checkpoint(code)`** — a one-byte **POST code** to I/O port `0x80` (a POST
|
||||
card or BMC shows it). `main.zig` emits one at each boot milestone (`cp_paging`,
|
||||
`cp_heap`, …) and on a fault/panic, so "where did it hang?" is answerable with no
|
||||
text output whatsoever. Writing `0x80` is universally safe.
|
||||
- **`log.recordPanic(msg)`** — stamps the panic message into a fixed record
|
||||
(`log.panic_record`, with a `magic` written last). A post-mortem — an attached
|
||||
debugger, a RAM dump, or a future file/pstore reader — recovers *what killed it*
|
||||
even though nothing was on screen.
|
||||
|
||||
The panic and CPU-exception handlers fan out to every sink, emit a POST code, and
|
||||
drop the breadcrumb — they never assume a console.
|
||||
|
||||
## The 0xE9 debug console
|
||||
|
||||
Port `0xE9` is the Bochs/QEMU debug console. It's detected safely: the port returns
|
||||
`0xE9` when read if present, and `0xFF` on real hardware, so `debugconPresent()`
|
||||
only enables the sink when it's really there. Under QEMU it's captured with
|
||||
`-debugcon file:…`, giving CI a log channel independent of `-serial`.
|
||||
|
||||
## The robustness spectrum
|
||||
|
||||
The result handles every combination — framebuffer only, serial only, both, or
|
||||
**neither**. With no channels at all the kernel still boots and runs; port-`0x80`
|
||||
checkpoints track progress and the panic breadcrumb captures failures. *Runs blind
|
||||
but correct* is the goal, not *always has output*.
|
||||
|
||||
## Related
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the display surface itself (pitch, format), the
|
||||
thing that becomes a graphics device driver.
|
||||
- [efi.md](efi.md) — where the loader captures (or, headless, doesn't capture) the
|
||||
framebuffer before `ExitBootServices`.
|
||||
- [device-interrupts.md](device-interrupts.md) — the serial UART bring-up the log's
|
||||
primary sink rides on.
|
||||
- [resilience.md](resilience.md) — why "never let a missing peripheral take the
|
||||
kernel down" is a core design stance.
|
||||
@@ -1,9 +0,0 @@
|
||||
# OS Developer Guide
|
||||
|
||||
This document is for those who need to understand the architectural decisions behind the OS.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The os was initially written in zig because it has excellent support for EFI. With zig, we could forgo using a third party bootloader, reducing the time to boot up the kernel. Following the "Zen of Zig", helped to produce the most readable codebase for an operating system ever created. So those, new to OS development could quickly get up to speed.
|
||||
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
# OS Development
|
||||
|
||||
This document explains the architectural decisions behind the operating system.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The OS is written in Zig because it has excellent EFI support, so the OS boots quickly without a third-party bootloader.
|
||||
|
||||
Zig comes batteries included for systems work — cross-compilation, a build system, and a test runner are all part of the toolchain. Building with `-Doptimize=ReleaseSafe` keeps runtime safety checks on in the shipped kernel, which removes entire classes of bugs. The built-in test suite, combined with a QEMU integration harness, means every feature is proven to work, before it is shipped.
|
||||
|
||||
The codebase of the OS prioritizes readability. The aim is a codebase where someone new to OS development can find their way around without a guide.
|
||||
|
||||
## A microkernel?
|
||||
|
||||
The kernel is a thin layer: it schedules processes and manages memory. Everything else — drivers, file systems, the display — runs in user space as separate, isolated processes.
|
||||
|
||||
The payoff is resilience. When a driver crashes, it doesn't take the OS down with it; it gets restarted. That makes this an ideal environment for *developing* an operating system, because a buggy driver is an ordinary bug: patch it, restart the service, and keep going.
|
||||
|
||||
There is a security benefit too. Processes are isolated and talk over Inter-Process Communication (IPC) channels, so compromising one service doesn't hand an attacker the whole machine. Vulnerabilities tend to stay contained in the process they started in.
|
||||
|
||||
Other operating systems choose to pack all of these duties into one binary as a Monolithic kernel, mostly for performance: a function call inside the kernel is faster than passing a message between isolated processes. That cost is real — an IPC round-trip is a few microseconds where a function call is nanoseconds — but it is also workload-shaped. Compute-bound programs don't notice it at all. For bulk data like file contents and pixels, the design moves data through shared memory and DMA so it is copied once, the same as a monolithic kernel; only small control messages cross the IPC boundary. What remains is the per-message cost on chatty paths, and the scheduler and memory management are designed to keep that small.
|
||||
|
||||
## Private ABI
|
||||
|
||||
The syscall layer is private. The numbers and structures in `abi.zig` are an internal detail shared between the kernel and the system's own libraries, and they are free to change between builds.
|
||||
|
||||
The public boundary sits one level up: the [vDSO](vdso.md) that programs call into, and the documented IPC protocols such as the [VFS protocol](../file-system-development/vfs-protocol.md). Programs that stick to those interfaces keep working while the kernel rearranges itself underneath. This is the opposite of the Linux approach, where raw syscall numbers are frozen forever; here, stability is promised at the library and protocol level, and nowhere below it.
|
||||
|
||||
## Steal the best bits and dump the legacy
|
||||
|
||||
The OS is Unix-like, but selectively. It borrows the ideas that have aged well — everything is a file, small services composed over clean interfaces — and skips the parts of POSIX that have caused decades of headaches.
|
||||
|
||||
Some concrete choices:
|
||||
|
||||
- **`spawn`, not `fork`.** Creating a process starts a fresh program and returns the child's id. There is no clone-the-whole-address-space-then-immediately-throw-it-away dance, and none of the subtle state-inheritance bugs that come with it.
|
||||
- **Time is a syscall.** The kernel owns the clock and timers directly. There is no time daemon to keep alive and no ambiguity about where the truth lives.
|
||||
- **Lifecycle events arrive as messages.** A supervisor learns that a child exited through an IPC message on an endpoint it already owns — delivered like any other message, not as an interrupt that can fire between any two instructions.
|
||||
|
||||
The test for keeping an idea is simple: does it still pull its weight, or is it only there because it was there in 1979?
|
||||
@@ -5,7 +5,7 @@ PCIe config window, the timer, the power registers — in a set of **system
|
||||
description tables** (SDTs). But before it can read any of them, danos has to *find*
|
||||
them, and they aren't at a fixed address. Getting there is a short chain of pointers,
|
||||
and this note explains it — in particular the question it's easy to trip on: **how
|
||||
does the [platform / device module](arch.md) know where the RSDT is?**
|
||||
does the [platform / device module](architecture.md) know where the RSDT is?**
|
||||
|
||||
Short answer: it doesn't receive the RSDT. The firmware hands over the **RSDP**, and
|
||||
the RSDT's address is a field *inside* the RSDP. The platform follows that pointer.
|
||||
@@ -16,13 +16,13 @@ the RSDT's address is a field *inside* the RSDP. The platform follows that point
|
||||
UEFI configuration table
|
||||
│ the loader reads the RSDP's physical address
|
||||
▼
|
||||
BootInfo.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||
│ the kernel forwards the whole BootInfo
|
||||
BootInformation.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||
│ the kernel forwards the whole BootInformation
|
||||
▼
|
||||
platform.discover(boot_info, …) system/devices/platform.zig
|
||||
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
||||
platform.discover(boot_information, …) system/kernel/platform.zig
|
||||
│ reads boot_information.acpi_rsdp, hands it to the ACPI backend
|
||||
▼
|
||||
acpi.discover(rsdp_phys, …) system/devices/acpi.zig
|
||||
acpi.discover(rsdp_phys, …) system/kernel/acpi.zig
|
||||
│ dereferences the RSDP, reads the pointer it contains
|
||||
▼
|
||||
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
||||
@@ -40,11 +40,12 @@ still up. `acpiRootSystemDescriptorPointer()` in `boot/efi.zig` walks the UEFI
|
||||
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
||||
the [memory map](memory-map.md).
|
||||
|
||||
## Step 2 — the handoff: a physical address in `BootInfo`
|
||||
## Step 2 — the handoff: a physical address in `BootInformation`
|
||||
|
||||
The loader can't just call the device module: the bootloader binary and the kernel
|
||||
binary are compiled separately, and **the loader isn't linked against the `platform`
|
||||
module at all** (it imports only the `boot-handoff` contract). So instead of a call, it
|
||||
module at all** (it imports only the `boot-handoff` contract and the
|
||||
`initial-ramdisk` module). So instead of a call, it
|
||||
deposits a value in the handoff struct:
|
||||
|
||||
```zig
|
||||
@@ -56,13 +57,14 @@ Two things about what crosses the boundary:
|
||||
|
||||
- **It's a *physical* address, not a Zig pointer.** The loader and kernel don't share
|
||||
an address space at the moment of the jump, so a raw `u64` physical address is the
|
||||
only thing that survives the handoff. `BootInfo.acpi_rsdp` is `0` when the firmware
|
||||
only thing that survives the handoff. `BootInformation.acpi_rsdp` is `0` when the firmware
|
||||
exposed no ACPI (e.g. a future device-tree machine, which would fill a different
|
||||
field instead — the kernel never learns which firmware booted it).
|
||||
- **The kernel can dereference it because it identity-maps ACPI memory.** The RSDP
|
||||
lives in ACPI-reclaim memory, which [paging.zig](paging.md) identity-maps along with
|
||||
the rest of RAM, so by the time discovery runs `@ptrFromInt(rsdp_phys)` is a valid
|
||||
pointer.
|
||||
- **The kernel can dereference it through the physmap.** The RSDP lives in
|
||||
ACPI-reclaim memory, which [paging.zig](paging.md) maps — along with the rest of
|
||||
RAM — into the higher-half **physmap** (there is no identity mapping; the low half
|
||||
belongs to user space). So by the time discovery runs,
|
||||
`@ptrFromInt(physicalToVirtual(rsdp_phys))` is a valid pointer.
|
||||
|
||||
This is the concrete form of the "capture the description pointer" step sketched in
|
||||
[discovery.md](discovery.md) — a plain `acpi_rsdp: u64` rather than a tagged handle,
|
||||
@@ -75,13 +77,13 @@ and then reads the root-table pointer *out of it*. Which pointer depends on the
|
||||
version, because the RSDP carries **both**:
|
||||
|
||||
```zig
|
||||
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(rsdp_phys);
|
||||
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(physicalToVirtual(rsdp_phys));
|
||||
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
|
||||
if (!checksumOk(@ptrFromInt(rsdp_phys), 20)) return error.BadRsdpChecksum;
|
||||
if (!checksumOk(@ptrFromInt(physicalToVirtual(rsdp_phys)), 20)) return error.BadRsdpChecksum;
|
||||
|
||||
if (rsdp.revision >= 2) {
|
||||
// ACPI 2.0+: use the 64-bit XSDT pointer (the 32-bit RSDT is deprecated)
|
||||
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(rsdp_phys);
|
||||
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(physicalToVirtual(rsdp_phys));
|
||||
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, …);
|
||||
} else {
|
||||
// ACPI 1.0: use the 32-bit RSDT pointer
|
||||
@@ -120,13 +122,15 @@ namespace and the port grant.
|
||||
|
||||
**The kernel hands the service what it needs and no more.** Reading PM1 event
|
||||
blocks and GPE blocks requires the FADT, which the kernel already parses for its
|
||||
own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one
|
||||
own power register map (feeding reboot), the PM timer, and the SCI line — the
|
||||
kernel itself has no S5/poweroff path. Rather than re-parse, the kernel appends the **FADT as one
|
||||
more memory resource** on the `acpi-tables` node; the service tells it apart
|
||||
from the AML blob resources by signature — the FADT keeps its intact `"FACP"`
|
||||
header, while the blob resources are header-stripped bytecode that starts with
|
||||
no signature. The kernel's own FADT parse is untouched; the service reads the
|
||||
PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control*
|
||||
for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself**
|
||||
PM1 *event* blocks (which the kernel never parsed — it extracts only the PM1
|
||||
*control* register, and it is the service, not the kernel, that writes it for
|
||||
`\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself**
|
||||
arrives as the node's one `len == 1` irq resource (distinct from the broad
|
||||
`[0, 256)` window that covers children's legacy lines), which is how the service
|
||||
finds the line to `irq_bind`.
|
||||
@@ -141,8 +145,9 @@ some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI:
|
||||
service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler
|
||||
method, drains the **Notify** queue that method produced, maps each notified
|
||||
device to an event (battery, AC, lid, or a generic `notify` with its code),
|
||||
and clears the status bit. A missing handler method is clear-and-log, not an
|
||||
error. Making GPEs work required teaching the interpreter one opcode it never
|
||||
and clears the status bit. A missing handler method is not an error: the
|
||||
status bit is cleared and the event silently dropped. Making GPEs work
|
||||
required teaching the interpreter one opcode it never
|
||||
handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained
|
||||
per evaluation; everything else a handler needs (field access, control flow,
|
||||
method calls) was already proven by the ring-3 `_STA`/`_CRS` work.
|
||||
@@ -152,8 +157,9 @@ so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML
|
||||
with a `Notify` inside a method body, run under `zig build test`. The QEMU
|
||||
`power-button` scenario proves the fixed-event path end to end: a QMP
|
||||
`system_powerdown` injects a real ACPI power-button press, and the service's SCI
|
||||
handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries
|
||||
are interface-complete but validated on real hardware later.
|
||||
handler must log it. Battery/AC/lid mapping is interface-complete but validated
|
||||
on real hardware later; the embedded controller's `_Qxx` queries are out of
|
||||
scope.
|
||||
|
||||
The service surface these events are *published on* — subscription, the event
|
||||
vocabulary, and orderly shutdown — is the power service, [power.md](power.md).
|
||||
@@ -168,5 +174,5 @@ vocabulary, and orderly shutdown — is the power service, [power.md](power.md).
|
||||
ACPI enumeration and events moved to the ring-3 acpi service.
|
||||
- [power.md](power.md) — the domain-named power service the ACPI event side publishes
|
||||
to (button, lid, battery) and its orderly-shutdown path into S5.
|
||||
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
||||
- [architecture.md](architecture.md) — why the kernel reaches the device code through a `platform`
|
||||
module and never names ACPI directly.
|
||||
@@ -0,0 +1,109 @@
|
||||
# Architecture split
|
||||
|
||||
danos targets x86_64 today, but is meant to grow onto other systems later — a
|
||||
Raspberry Pi, say, which is AArch64 and has no UEFI. To keep that possible without
|
||||
a rewrite, CPU-specific kernel code lives behind a boundary: the generic kernel
|
||||
never names an architecture, and each architecture plugs in behind it.
|
||||
|
||||
## The seam is a build-time module named `architecture`
|
||||
|
||||
The mechanism is deliberately boring — no vtables, no function-pointer tables, no
|
||||
runtime dispatch. `build.zig` exposes one architecture's code as a module called
|
||||
`architecture`:
|
||||
|
||||
```zig
|
||||
const architecture_module = b.addModule("architecture", .{
|
||||
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||
});
|
||||
```
|
||||
|
||||
and the generic kernel imports it by that name:
|
||||
|
||||
```zig
|
||||
const architecture = @import("architecture");
|
||||
// ...
|
||||
architecture.halt(); // never says "x86_64"
|
||||
```
|
||||
|
||||
Adding a second architecture is then a build-time choice: create
|
||||
`system/kernel/architecture/aarch64/`, and point the `architecture` module at it when the target CPU is
|
||||
AArch64. `kernel.zig` and `console.zig` don't change. **That compiler-checked module
|
||||
boundary _is_ the architecture interface** — when a new architecture is missing a function
|
||||
the generic kernel calls, the build fails and names exactly what's missing.
|
||||
|
||||
## What's arch-specific vs generic
|
||||
|
||||
The split follows a simple test: does it name a CPU instruction, a hardware
|
||||
register, or a memory-management structure? If so, it's arch-specific.
|
||||
|
||||
| Arch-specific — `system/kernel/architecture/x86_64/` | Generic — kernel core |
|
||||
|---|---|
|
||||
| `cpu.zig`: CPU state, trap-frame accessors, paging, SMP | `console.zig` — pure pixel math, framebuffer drawing |
|
||||
| `gdt.zig`, `idt.zig`, `tss.zig` — descriptor tables | `kernel.zig` — kernel orchestration, scheduler, IPC |
|
||||
| `paging.zig` — page-table setup and management | `process.zig` — process lifecycle, address spaces |
|
||||
| `apic.zig`, `ioapic.zig` — interrupt controllers | `scheduler.zig` — task scheduling and context switch |
|
||||
| `serial.zig`, `io.zig` — UART, I/O primitives | `vfs.zig` — filesystem abstraction |
|
||||
| `isr.s`, `smp.zig` — exceptions, AP bring-up, context switch | `irq.zig`, `ipc*.zig` — interrupt dispatch, messaging |
|
||||
| `linker.ld` — kernel link layout, load address | |
|
||||
|
||||
Notice the framebuffer console is *generic*: it just writes pixels into whatever
|
||||
framebuffer it's handed, so it needs no per-arch version. Most of the kernel
|
||||
should end up on the generic side; the architecture module stays small.
|
||||
|
||||
## Two axes, kept separate
|
||||
|
||||
There are really two independent questions, and it's worth not conflating them:
|
||||
|
||||
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
||||
`system/kernel/architecture/<cpu>/`.
|
||||
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
||||
*separately*, because loaders are their own binaries. `boot/efi.zig` builds
|
||||
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
||||
separate loader at all — the firmware jumps straight into the kernel with a
|
||||
device-tree pointer, so that entry work would live in the AArch64 architecture code.
|
||||
Either path converges on the same neutral [`BootInformation`](memory-map.md).
|
||||
|
||||
## Current x86_64 contents
|
||||
|
||||
- **`system/kernel/architecture/x86_64/cpu.zig`** — the `architecture` module root. Exposes the trap-frame
|
||||
`CpuState` and accessors, `init()` (bring up the descriptor tables), `enablePaging()`,
|
||||
`enterUser()`/`userExit()` for ring-0 ↔ ring-3 transitions, address-space management, and SMP
|
||||
entry points (see [halting.md](halting.md), [interrupts.md](interrupts.md), [paging.md](paging.md),
|
||||
[scheduling.md](scheduling.md)).
|
||||
- **`system/kernel/architecture/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
||||
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables and address-space
|
||||
management (see [paging.md](paging.md)).
|
||||
- **`system/kernel/architecture/x86_64/apic.zig`** / **`ioapic.zig`** — the Local APIC, its timer,
|
||||
and the I/O APIC for device interrupts (see [device-interrupts.md](../device-driver-development/device-interrupts.md)).
|
||||
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
machine-readable log channel, see [testing.md](../testing.md)) and the shared port-I/O + MSR primitives.
|
||||
- **`system/kernel/architecture/x86_64/smp.zig`** / **`per-cpu.zig`** — application-processor bring-up
|
||||
and per-CPU state (GS base, system-call entry point, see [scheduling.md](scheduling.md)).
|
||||
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||
helpers, ring-0 ↔ ring-3 transitions, and the context switch — real assembly, since Zig inline asm can't
|
||||
express them (see [scheduling.md](scheduling.md)).
|
||||
- **`system/kernel/architecture/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||
address, one PT_LOAD per permission set).
|
||||
|
||||
The kernel entry point `_start` lives in the architecture-specific `isr.s` (x86_64 here).
|
||||
On x86_64 it sets up the kernel stack in BSS and jumps to `kmain()` in `kernel.zig`.
|
||||
This is already per-architecture — an AArch64 port would have its own `isr.s` entry
|
||||
that parses the device-tree pointer from a register and jumps to the same `kmain()`.
|
||||
The entry interface is minimal and emerges naturally from the [boot-handoff](memory-map.md)
|
||||
contract both share.
|
||||
|
||||
## The discipline
|
||||
|
||||
The thing that makes this help rather than hurt: **only extract what's provably
|
||||
architecture-specific, and let the interface emerge with the second
|
||||
implementation.** With a single architecture you're guessing at the seam, and a
|
||||
wrong guess encoded as elaborate abstraction is expensive to undo. So:
|
||||
|
||||
- Move code into `architecture/` only when it genuinely names CPU-specific machinery.
|
||||
- Grow the `architecture` surface one function at a time, as steps need it.
|
||||
- Don't pre-design the interrupt or paging interfaces before writing them.
|
||||
|
||||
Directory hygiene is cheap and reversible; premature abstraction is neither. When
|
||||
architecture #2 lands and something doesn't fit, reshaping a few hundred lines is nothing —
|
||||
unwinding an abstraction empire is not.
|
||||
@@ -5,7 +5,7 @@ though — the Pis span **two different CPU architectures** (32-bit `arm` and 64
|
||||
`aarch64`) and (stock) a different boot protocol from x86-64's UEFI. **danos targets
|
||||
`aarch64` only** (see the decision below); the `arm`/`aarch64` distinction still
|
||||
matters for understanding why. This page maps the landscape so the
|
||||
[arch split](arch.md) and build system can be planned for it.
|
||||
[architecture split](architecture.md) and build system can be planned for it.
|
||||
|
||||
## `arm` vs `aarch64` — 32-bit vs 64-bit
|
||||
|
||||
@@ -18,7 +18,7 @@ matters for understanding why. This page maps the landscape so the
|
||||
new ISA.
|
||||
|
||||
They are as different from each other as either is from x86-64: separate registers,
|
||||
page-table formats, and calling conventions. Each needs its own `system/kernel/arch/<name>/`.
|
||||
page-table formats, and calling conventions. Each needs its own `system/kernel/architecture/<name>/`.
|
||||
|
||||
## The Raspberry Pi models
|
||||
|
||||
@@ -39,7 +39,7 @@ page-table formats, and calling conventions. Each needs its own `system/kernel/a
|
||||
|
||||
## Booting: UEFI is not x86-only
|
||||
|
||||
The boot protocol is a **separate axis** from the CPU (see [arch.md](arch.md)):
|
||||
The boot protocol is a **separate axis** from the CPU (see [architecture.md](architecture.md)):
|
||||
|
||||
- **UEFI** exists for ARM too — ARM servers require it (SBSA/SBBR), QEMU boots it
|
||||
with **AAVMF** (the AArch64 build of the same EDK2 firmware as x86's OVMF), and
|
||||
@@ -57,9 +57,9 @@ the DTB/ACPI tells you what devices exist.
|
||||
|
||||
## What danos needs, layer by layer
|
||||
|
||||
- **One CPU arch module: `system/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||
- **One CPU arch module: `system/kernel/architecture/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||
providing the same `arch` interface as x86_64: `halt`, context switch,
|
||||
interrupt/exception vectors, page tables, a UART, a timer. No `system/kernel/arch/arm/` is
|
||||
interrupt/exception vectors, page tables, a UART, a timer. No `system/kernel/architecture/arm/` is
|
||||
planned (see the decision above), so there's a single ARM backend to write.
|
||||
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
||||
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
||||
@@ -67,11 +67,16 @@ the DTB/ACPI tells you what devices exist.
|
||||
from a different source. This is where keeping boot-protocol knowledge on the
|
||||
loader side (as we did for the UEFI memory-map classification) pays off.
|
||||
- **The UEFI loader mostly carries over.** `boot/efi.zig` is largely
|
||||
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
||||
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
||||
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
||||
can reuse it — which makes **aarch64-UEFI the easiest second target**, easier than
|
||||
the device-tree Pi.
|
||||
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its truly
|
||||
x86-specific bits are the ELF machine check (`.X86_64`), the SysV calling
|
||||
convention for the kernel jump, the `hlt` park on failure — and, the substantial
|
||||
one, the bootstrap page tables: `buildBootstrapTables` builds x86-64 4-level
|
||||
tables (PML4/PDPT/PD index shifts, x86 PTE bits, 2 MiB leaves) and hands the
|
||||
kernel a CR3. Page-table formats are per-architecture (see above), so an
|
||||
`aarch64` loader keeps the protocol code but rewrites that builder in the
|
||||
aarch64 translation-table format. Even so, an `aarch64`-UEFI target (QEMU
|
||||
`virt` + AAVMF) reuses most of it — which makes **aarch64-UEFI the easiest
|
||||
second target**, easier than the device-tree Pi.
|
||||
|
||||
## Pi hardware quirks (for when we port)
|
||||
|
||||
@@ -82,7 +87,7 @@ The Pi is not a "standard" ARM platform — expect Broadcom-specific peripherals
|
||||
Pi 5. Everything below is an offset from it.
|
||||
- **UART**: a **PL011** (at base + `0x20_1000`) plus a mini-UART; on some boards the
|
||||
PL011 is wired to Bluetooth, so which one is the console varies. This is the
|
||||
`aarch64`/`arm` equivalent of our x86 [COM1 serial](testing.md).
|
||||
`aarch64`/`arm` equivalent of our x86 [COM1 serial](../testing.md).
|
||||
- **Interrupt controller**: *not* a standard ARM GIC on the older parts — the Zero W
|
||||
and Pi 3 use Broadcom's own ARMCTRL controller (Pi 3 adds a per-core "local"
|
||||
controller for timers/mailboxes). The **Pi 4 and 5 do have a GIC-400**. So the
|
||||
@@ -114,7 +119,7 @@ Two routes, mirroring how we test x86-64 with OVMF:
|
||||
|
||||
## Related
|
||||
|
||||
- [arch.md](arch.md) — the arch-module boundary these targets plug into, and the
|
||||
- [architecture.md](architecture.md) — the arch-module boundary these targets plug into, and the
|
||||
CPU-arch vs boot-protocol "two axes".
|
||||
- [efi.md](efi.md) — the UEFI loader that carries over to aarch64-UEFI.
|
||||
- [vision.md](vision.md) — why isolated, portable-across-architectures is the goal.
|
||||
- [vision.md](../vision.md) — why isolated, portable-across-architectures is the goal.
|
||||
@@ -5,21 +5,25 @@ MMIO addresses, IRQ numbers, CPU count, and interrupt controller it can't just
|
||||
assume. On x86 that description comes from **ACPI** tables; on ARM from a **device
|
||||
tree** (DTB). This note is a design plan, not built yet: *when* danos should tackle
|
||||
it, and *how* to keep it architecture-agnostic — the same discipline the
|
||||
[memory map](memory-map.md) and [arch split](arch.md) already follow.
|
||||
[memory map](memory-map.md) and [architecture split](architecture.md) already follow.
|
||||
|
||||
## What the kernel assumes today
|
||||
## What the kernel assumed when this plan was written
|
||||
|
||||
Right now danos discovers almost nothing — it coasts on legacy PC fixtures that are
|
||||
guaranteed to exist under QEMU + UEFI:
|
||||
At the time danos discovered almost nothing — it coasted on legacy PC fixtures that
|
||||
are guaranteed to exist under QEMU + UEFI:
|
||||
|
||||
- `arch/x86_64/apic.zig` assumes the **Local APIC** at the default `0xFEE0_0000` and
|
||||
calibrates its timer against the **PIT** (the legacy 8254).
|
||||
- `arch/x86_64/serial.zig` hardcodes **COM1** at I/O port `0x3F8`.
|
||||
- `system/kernel/architecture/x86_64/apic.zig` assumed the **Local APIC** at the
|
||||
default `0xFEE0_0000` and calibrated its timer against the **PIT** (the legacy
|
||||
8254). (Today the PIT is the *last-resort* reference: calibration prefers the
|
||||
CPUID-reported TSC frequency, then the HPET, then the ACPI PM timer.)
|
||||
- `system/kernel/architecture/x86_64/serial.zig` hardcoded **COM1** at I/O port
|
||||
`0x3F8`. (Today `0x3F8` is only the default: the kernel loopback-probes the UART
|
||||
and parses ACPI's SPCR table to target the firmware's actual debug port.)
|
||||
- The framebuffer and memory map come from **UEFI** — that *is* discovery, just done
|
||||
by the firmware and handed over, not read from ACPI.
|
||||
|
||||
This works only because PC-compatible hardware promises those legacy pieces exist at
|
||||
those addresses. It is a crutch, and it does not travel.
|
||||
This worked only because PC-compatible hardware promises those legacy pieces exist at
|
||||
those addresses. It was a crutch, and it did not travel.
|
||||
|
||||
## The forcing functions: when to build it
|
||||
|
||||
@@ -32,7 +36,7 @@ Two things drive the need, and they set the timing:
|
||||
**aarch64 it's required to boot at all**. The [aarch64 port](arm.md) is what
|
||||
forces the issue.
|
||||
|
||||
2. **Isolated user-space drivers need it.** In the [microkernel vision](vision.md),
|
||||
2. **Isolated user-space drivers need it.** In the [microkernel vision](../vision.md),
|
||||
drivers live in user space — but something has to enumerate the hardware and hand
|
||||
each driver its MMIO regions and IRQs. That enumeration *is* device discovery. So
|
||||
discovery is a prerequisite for real drivers, **not** for user mode itself.
|
||||
@@ -53,8 +57,8 @@ it isn't really agnostic.
|
||||
|
||||
## The one cheap step to take sooner
|
||||
|
||||
Have the **loader capture the description pointer** into `BootInfo` — a neutral
|
||||
handle, no parsing:
|
||||
Have the **loader capture the description pointer** into `BootInformation` — a
|
||||
neutral handle, no parsing:
|
||||
|
||||
```zig
|
||||
pub const HardwareInfo = extern struct {
|
||||
@@ -128,7 +132,7 @@ when*:
|
||||
- **User-space enumeration: a device-manager server.** Everything else — PCI devices,
|
||||
peripherals — is parsed (or queried from the kernel's parse) by a privileged
|
||||
user-space server that hands each driver process its MMIO regions and IRQ rights
|
||||
over [IPC](ipc.md). Combined with **interrupts-as-messages** (an IRQ delivered to a
|
||||
over [IPC](../device-driver-development/ipc.md). Combined with **interrupts-as-messages** (an IRQ delivered to a
|
||||
driver as a message on a channel — a natural extension of the wait queues and
|
||||
channels already built), that's what makes drivers genuinely isolated.
|
||||
|
||||
@@ -140,13 +144,13 @@ slice is unavoidably in-kernel.
|
||||
On ARMv8 the generic timer exposes its frequency directly via the `CNTFRQ` register —
|
||||
no calibration needed. That's cleaner than the x86 side, where we measure the LAPIC
|
||||
and TSC against the PIT because nothing tells us their frequency (see
|
||||
[device-interrupts.md](device-interrupts.md)). Discovery on ARM hands you more for
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md)). Discovery on ARM hands you more for
|
||||
free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
|
||||
## Suggested ordering
|
||||
|
||||
1. **Now (cheap):** plumb the neutral `HardwareInfo` pointer through the loader into
|
||||
`BootInfo`. No parser yet.
|
||||
`BootInformation`. No parser yet.
|
||||
2. **Next milestone unchanged:** user mode + address-space isolation — needs no
|
||||
discovery.
|
||||
3. **With the aarch64 port:** build the agnostic discovery layer, **DTB first** (the
|
||||
@@ -162,18 +166,18 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
- [arm.md](arm.md) — the aarch64 target that forces genuine discovery (DTB, GIC).
|
||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam,
|
||||
and the note about grabbing the RSDP before `ExitBootServices`.
|
||||
- [device-interrupts.md](device-interrupts.md) — the LAPIC/timer bring-up that
|
||||
- [device-interrupts.md](../device-driver-development/device-interrupts.md) — the LAPIC/timer bring-up that
|
||||
discovery will eventually feed (IOAPIC, real IRQ routing).
|
||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
- [ipc.md](../device-driver-development/ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
will ride on.
|
||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||
- [vision.md](../vision.md) — why drivers belong in isolated user space at all.
|
||||
|
||||
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||
|
||||
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||
([device-manager.md](../device-driver-development/device-manager.md)): it claims the bridge, repeats the
|
||||
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||
@@ -182,22 +186,27 @@ the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||
|
||||
The kernel no longer folds the AML namespace's Device objects into the device
|
||||
tree. It still parses the *static* tables (MADT for SMP, HPET for the tick, MCFG
|
||||
for the host bridge, FADT) and still builds the AML namespace — but only to read
|
||||
the `\\_S5` sleep type for poweroff. Device discovery is the ring-3 **acpi
|
||||
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||
for the host bridge, FADT); at this point it also still built the AML namespace —
|
||||
but only to read the `\\_S5` sleep type for poweroff. (That remnant is gone too:
|
||||
the kernel now runs no AML at all — soft-off belongs to the acpi service, and the
|
||||
kernel keeps only the AML-free reboot path.) Device discovery is the ring-3 **acpi
|
||||
service** ([device-manager.md](../device-driver-development/device-manager.md)): it claims the `acpi-tables`
|
||||
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||
and registers + reports each `_HID` device — the device manager matches drivers
|
||||
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||
entirely in user space; the kernel seeds only the host bridge and the
|
||||
acpi-tables node.
|
||||
entirely in user space, anchored on two kernel-seeded nodes: the host bridge and
|
||||
the acpi-tables node. (The kernel's static-table parse also seeds the processor,
|
||||
interrupt-controller, and HPET timer nodes, and it publishes the boot
|
||||
framebuffer as a claimable display node — but no *enumeration* happens in
|
||||
ring 0.)
|
||||
|
||||
## Discovery is a swappable process per firmware (M19–M20)
|
||||
|
||||
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||
to do it before the second architecture rather than after. Everything at and
|
||||
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||
above the [device-manager](../device-driver-development/device-manager.md) protocol — descriptors,
|
||||
containment, reports, matching, supervision — is generic and may never become
|
||||
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||
isolated as **one swappable process per firmware**:
|
||||
@@ -231,11 +240,14 @@ Two consequences of neutrality bind on later work:
|
||||
|
||||
Two supporting decisions keep the kernel's remaining slice honest:
|
||||
|
||||
- **The AML interpreter is a shared build module**, compiled into both the
|
||||
kernel and the acpi service — one source, two builds, no fork. The kernel
|
||||
links it for the `\_S5` poweroff evaluation, the service links it for
|
||||
everything else, and the `acpi-parse` test asserts the two produce the same
|
||||
device count across the ring-3 move.
|
||||
- **The AML interpreter is a single build module**
|
||||
(`library/device/acpi/aml/aml.zig`) — one source, no fork. During the ring-3 move
|
||||
it was compiled into both the kernel (which linked it just for the `\_S5`
|
||||
poweroff evaluation) and the acpi service, with the `acpi-parse` test
|
||||
asserting the two produce the same device count. Since soft-off followed
|
||||
discovery out of the kernel, only the acpi service links the module — the
|
||||
kernel runs no AML — and the test now asserts a device-count *floor* for the
|
||||
ring-3 parse instead, there being no kernel count left to equal.
|
||||
- **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||
PCI functions carry BAR resources, and `device_register` containment demands
|
||||
the bridge own windows that cover them. Those apertures are derived
|
||||
@@ -23,14 +23,20 @@ Partition (ESP)** and running a file at a well-known fallback path:
|
||||
EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
```
|
||||
|
||||
The boot volume is the **FHS-shaped `zig-out`** itself (see the repository-layout note
|
||||
in [README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) to `zig-out/EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `zig-out/system/kernel`, init at
|
||||
`zig-out/system/services/init`, the initial-ramdisk at `zig-out/boot/`. The
|
||||
`run-x86-64` step points QEMU at OVMF (UEFI firmware for virtual machines) and presents
|
||||
`zig-out` to the guest as a FAT drive. The firmware finds `BOOTX64.efi` and runs it —
|
||||
that's our `main()`, which then loads the kernel and init from their FHS paths.
|
||||
The boot volume is **FHS-shaped** (see the repository-layout note in
|
||||
[README.md](../README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) at `EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `system/kernel`, init at
|
||||
`system/services/init`, the pre-packed boot capsule at `boot/system.img`
|
||||
([system-image.md](system-image.md)).
|
||||
`zig-out` mirrors that tree, but what a machine actually boots is the
|
||||
self-contained FAT32 image `tools/make-fat-image.py` builds from the same files
|
||||
(`danos-usb.img`). The `run-x86-64` step points QEMU at OVMF (UEFI firmware for
|
||||
virtual machines) and attaches that image (its serial-logging twin, built the
|
||||
same way) as a USB mass-storage device on the xHCI bus — the guest never sees
|
||||
`zig-out`. The firmware finds `BOOTX64.efi` on the image and runs it — that's
|
||||
our `main()`, which then loads the kernel and the system binaries from their
|
||||
FHS paths.
|
||||
|
||||
## Boot services: the firmware's API
|
||||
|
||||
@@ -56,7 +62,16 @@ The comment in `boot()` says exactly this:
|
||||
|
||||
## What our loader actually does
|
||||
|
||||
`boot()` runs four steps in order:
|
||||
The four milestones below are the spine of `boot()`. Along the way it also
|
||||
captures the **ACPI RSDP** from the UEFI configuration table (while boot
|
||||
services are still up), loads the system binaries into an in-RAM
|
||||
**initial ramdisk** (`loadSystemTree` — normally a single read of the pre-packed
|
||||
`boot\system.img` capsule, which already *is* the ramdisk wire format; it falls
|
||||
back to opening each manifest-listed path, and walks the `/system` and `/test`
|
||||
trees only as a last resort for hand-assembled sticks — the capsule's format, builder, and
|
||||
fallback chain are documented in [system-image.md](system-image.md). Best-effort either way — a kernel-only
|
||||
volume still boots), and builds the **bootstrap page tables** the kernel starts
|
||||
life on (`buildBootstrapTables`), all before the jump:
|
||||
|
||||
### 1. Query the framebuffer (`queryFramebuffer`)
|
||||
|
||||
@@ -91,16 +106,20 @@ All of this *must* happen now, because after exit there's no GOP to ask. (See
|
||||
may return short, so we loop.)
|
||||
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
||||
walk the program headers. For every `PT_LOAD` segment we:
|
||||
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
||||
- reserve the exact physical pages it asks to be loaded at (`p_paddr`) via
|
||||
`allocatePages`,
|
||||
- `@memcpy` the file-backed bytes to that address,
|
||||
- `@memset` the `.bss` tail (the part where `p_memsz > p_filesz`) to zero.
|
||||
|
||||
The kernel is linked to load at physical `0x100000` (1 MiB) — set by
|
||||
`exe.image_base` in `build.zig` and the linker script. UEFI identity-maps memory,
|
||||
so the physical address the ELF asks for is the address it actually runs at. If a
|
||||
segment's `p_paddr` collided with firmware-reserved memory, `allocatePages` would
|
||||
fail and we'd need to move `image_base`.
|
||||
The kernel is linked to *run* in the higher half (virtual base
|
||||
`0xFFFFFFFF80000000`; `exe.image_base` in `build.zig` is the *virtual* address
|
||||
`0xFFFFFFFF80100000`) but is *loaded* low: the linker script's `AT()` clauses
|
||||
give every segment a low physical load address (`p_paddr`, with `.text` at
|
||||
`0x100000`, 1 MiB), which is what the loader allocates and copies into. The
|
||||
bootstrap page tables built before the jump map the high link addresses onto
|
||||
those low physical pages. If a segment's `p_paddr` collided with
|
||||
firmware-reserved memory, `allocatePages` would fail and we'd need to move the
|
||||
load addresses.
|
||||
|
||||
`loadElf` returns `e_entry`, the kernel's entry-point address.
|
||||
|
||||
@@ -124,13 +143,12 @@ entirely ours.
|
||||
### 4. Jump to the kernel
|
||||
|
||||
```zig
|
||||
const kernel: *const fn (*const BootInfo) callconv(boot_handoff.kernel_abi) noreturn =
|
||||
@ptrFromInt(entry);
|
||||
kernel(&boot_info);
|
||||
handoff(cr3, entry, &boot_information);
|
||||
```
|
||||
|
||||
We cast the entry address to a function pointer and call it, passing a pointer to
|
||||
the `BootInfo` we filled in. This never returns.
|
||||
`handoff` is a single inline-asm block — `cli`, load the bootstrap page tables'
|
||||
`cr3`, place the `boot_information` pointer in RDI, then `callq *entry` — so
|
||||
nothing runs between the CR3 load and the jump. This never returns.
|
||||
|
||||
## The ABI subtlety: RCX vs RDI
|
||||
|
||||
@@ -138,14 +156,15 @@ There's a deliberate detail worth calling out. A UEFI binary is compiled with th
|
||||
**Microsoft x64** calling convention (first argument in register **RCX**). Our
|
||||
kernel is freestanding and uses the **SysV AMD64** convention (first argument in
|
||||
**RDI**). If we let each side use its target's default, the loader would place
|
||||
`boot_info` in RCX while the kernel looked for it in RDI — and the kernel would
|
||||
read garbage.
|
||||
`boot_information` in RCX while the kernel looked for it in RDI — and the kernel
|
||||
would read garbage.
|
||||
|
||||
So both sides pin the convention explicitly to SysV via the shared
|
||||
`boot_handoff.kernel_abi` (defined in `system/boot-handoff.zig`). The loader's
|
||||
function-pointer type and the kernel's `_start` both reference it, so the pointer lands
|
||||
in the register the kernel expects. This is the whole reason `kernel_abi` lives in the
|
||||
shared `boot-handoff` module: it's a contract both binaries must agree on. See
|
||||
So the convention is pinned explicitly to SysV via the shared
|
||||
`boot_handoff.kernel_abi` (defined in `system/boot-handoff.zig`). The kernel's
|
||||
`kmainEntry` declares `callconv(kernel_abi)`; the loader honours the same contract
|
||||
by loading RDI by hand in `handoff`'s inline asm rather than trusting its own
|
||||
Microsoft-x64 default. `kernel_abi` lives in the shared `boot-handoff` module
|
||||
because it's a contract both binaries must agree on. See
|
||||
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
||||
|
||||
## The handoff contract
|
||||
@@ -154,15 +173,18 @@ The loader and kernel are two *separate* binaries built for two different target
|
||||
so everything they exchange must have an identically-defined memory layout. That's
|
||||
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
(`library/device/model/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
|
||||
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
||||
framebuffer; this is where future handoff data like the memory map will go).
|
||||
- `Framebuffer`, `PixelFormat`, `kernel_abi` — the shared field layouts and the
|
||||
calling convention.
|
||||
- `BootInformation` — the top-level struct passed to the kernel: the
|
||||
framebuffer, the memory map, the kernel's own `PT_LOAD` segments
|
||||
(`kernel_segments` + `kernel_segment_count`, so the kernel can re-map itself
|
||||
with correct permissions), the ACPI RSDP address, and the initial-ramdisk
|
||||
base/length.
|
||||
- `Framebuffer`, `PixelFormat`, `MemoryMap`/`MemoryRegion`, `KernelSegment`,
|
||||
`kernel_abi` — the shared field layouts and the calling convention.
|
||||
|
||||
Both are `extern struct`, giving them a stable, C-compatible layout so the bytes
|
||||
the loader writes are the bytes the kernel reads.
|
||||
The structs are `extern struct`, giving them a stable, C-compatible layout so the
|
||||
bytes the loader writes are the bytes the kernel reads.
|
||||
|
||||
## The whole flow at a glance
|
||||
|
||||
@@ -170,15 +192,20 @@ the loader writes are the bytes the kernel reads.
|
||||
power on
|
||||
-> UEFI firmware initialises hardware
|
||||
-> finds EFI/BOOT/BOOTX64.efi on the FHS volume, runs it (our efi.zig main)
|
||||
-> grab boot services
|
||||
-> grab boot services (+ the ACPI RSDP from the configuration table)
|
||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||
-> loadKernel (read system/kernel ELF, load PT_LOAD segments to 0x100000)
|
||||
-> loadKernel (read system/kernel ELF, load PT_LOAD segments low, .text at 0x100000)
|
||||
-> loadSystemTree (read the boot\system.img capsule as the in-RAM initial ramdisk;
|
||||
fallbacks: manifest-listed paths, then a /system + /test tree walk)
|
||||
-> buildBootstrapTables (identity + physmap + higher-half kernel mappings)
|
||||
-> exitBootServices (retry until the memory-map key holds)
|
||||
-> jump to e_entry, boot_info pointer in RDI
|
||||
-> kernel _start (system/kernel/kernel.zig: framebuffer console, then halt)
|
||||
-> handoff: load bootstrap CR3, jump to e_entry, boot_information pointer in RDI
|
||||
-> kernel _start (architecture/x86_64/isr.s: switch to a kernel-owned stack,
|
||||
call kmainEntry -> paging, heap, device discovery,
|
||||
scheduler, SMP, user space)
|
||||
```
|
||||
|
||||
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
||||
disappear.** `boot/efi.zig` is the thin bridge that collects those gifts into a
|
||||
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
||||
on our own.
|
||||
`BootInformation`, tears down the firmware, and jumps into the kernel — after
|
||||
which we're on our own.
|
||||
@@ -8,9 +8,9 @@ natural unit because that's the granularity the CPU's paging hardware maps — a
|
||||
it is the primitive everything above it stands on: page tables, the kernel heap,
|
||||
per-process memory all ultimately ask the frame allocator for pages.
|
||||
|
||||
It's **generic kernel code**: it operates on the neutral `system.MemoryRegion`
|
||||
It's **generic kernel code**: it operates on the neutral `boot_handoff.MemoryRegion`
|
||||
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
||||
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
||||
page. (Contrast [architecture.md](architecture.md), which is where CPU-specific code lives.)
|
||||
|
||||
## Why a bitmap
|
||||
|
||||
@@ -38,16 +38,18 @@ and a `next_hint` marking where the next allocation scan should start.
|
||||
|
||||
### init(map) — building it from the memory map
|
||||
|
||||
1. **Size it.** Find the highest address across all `usable` regions;
|
||||
`total_frames = highest / page_size`. Reserved and MMIO spans above that
|
||||
(remember the ~12 GiB of MMIO from [memory-map.md](memory-map.md)) sit *outside*
|
||||
the bitmap and are simply never allocatable.
|
||||
1. **Size it.** Find the highest address across all RAM regions — every kind
|
||||
*except* `mmio` — so reserved and ACPI spans sit *inside* the bitmap, marked
|
||||
used but trackable (e.g. so the boot buffers can be freed later);
|
||||
`total_frames = highest / page_size`. Only MMIO (device address space —
|
||||
remember the ~12 GiB of it from [memory-map.md](memory-map.md)) sits outside
|
||||
the bitmap and is simply never allocatable.
|
||||
2. **Place it (the bootstrap).** The bitmap needs storage before an allocator
|
||||
exists — a chicken-and-egg. Solution: pick the first `usable` region big enough
|
||||
to hold the bitmap and put it there, addressing it directly as a pointer. That
|
||||
last part relies on the firmware's **identity mapping** still being in effect
|
||||
(physical address == virtual address), which holds until the kernel installs
|
||||
its own page tables.
|
||||
to hold the bitmap and put it there, addressing it through the **physmap**
|
||||
(`boot_handoff.physicalToVirtual`). The loader's bootstrap page tables already
|
||||
provide the physmap and the kernel's own tables keep it, so the pointer stays
|
||||
valid across the paging switch.
|
||||
3. **Mark, then free.** Set the whole bitmap to `used` (`0xff`), then walk the
|
||||
`usable` regions clearing their bits. Doing it in that direction means every
|
||||
gap, reserved span, and hole is unallocatable *by default* — we only ever hand
|
||||
@@ -73,10 +75,13 @@ free, or one out of range) are ignored rather than corrupting the used count.
|
||||
|
||||
- **Generic walk.** Because `MemoryRegion` is danos's own type, the map is a plain
|
||||
slice — none of the variable descriptor-stride from the raw UEFI map.
|
||||
- **Identity mapping assumption.** Placing the bitmap by physical address only
|
||||
works while the firmware's identity map is live. When danos sets up its own
|
||||
paging, the bitmap (and any other physical pointer) will need an explicit
|
||||
mapping. This is a deliberate, documented dependency of this stage.
|
||||
- **Physmap addressing.** The bitmap (and the region array) is reached through
|
||||
the physmap via `boot_handoff.physicalToVirtual`, which both the loader's
|
||||
bootstrap tables and the kernel's own tables provide — no remapping is needed
|
||||
when danos switches to its own paging. The one invariant: the bitmap must sit
|
||||
under the bootstrap physmap's reach (4 GiB), which holds because the placement
|
||||
scan (step 2 above) runs from the lowest usable region up and takes the first
|
||||
one big enough — on the supported configurations that lands well under 4 GiB.
|
||||
- **Frame 0 is reserved** so `0` stays a safe "none" sentinel — and the bitmap is
|
||||
never placed there. (An early bug did exactly that: a `usable` region at physical
|
||||
address 0 collided with a `0`-means-not-found sentinel and tripped a panic. The
|
||||
@@ -90,12 +95,15 @@ free, or one out of range) are ignored rather than corrupting the used count.
|
||||
`kmain` brings the allocator up and self-tests it. Booted in QEMU with 128 MiB:
|
||||
|
||||
```
|
||||
danos: frame allocator online
|
||||
free frames: 19751 (77 MiB) <- matches the map's 77 MiB usable
|
||||
alloc x3 : 0x2000 0x3000 0x4000 <- frame 0 reserved, bitmap at 0x1000, so allocs start at 0x2000
|
||||
after free : 19751 frames free <- three freed, count restored
|
||||
/system/kernel: frame allocator online
|
||||
free frames: 30520 (119 MiB) <- matches the map's 119 MiB usable
|
||||
alloc x3 : 0x3000 0x4000 0x5000 <- frame 0 reserved, bitmap at 0x1000, AP trampoline at 0x2000
|
||||
after free : 30520 frames free <- three freed, count restored
|
||||
```
|
||||
|
||||
(The AP-trampoline page is claimed with `allocBelow` right after `init`, before
|
||||
the demo allocations — hence they start at 0x3000.)
|
||||
|
||||
The `free frames` MiB agreeing with the memory map's `usable RAM`, the three
|
||||
distinct consecutive addresses, and the count returning to its start after freeing
|
||||
are the three signals that init, alloc and free are all correct.
|
||||
@@ -103,7 +111,7 @@ are the three signals that init, alloc and free are all correct.
|
||||
## Boot-services memory comes pre-reclaimed
|
||||
|
||||
The UEFI boot-services memory (~44 MiB) is defunct and free once
|
||||
`ExitBootServices` runs, taking usable RAM from ~76 MiB up to ~121 MiB. The frame
|
||||
`ExitBootServices` runs, taking usable RAM from ~76 MiB up to ~119 MiB. The frame
|
||||
allocator does **nothing special** to get it: the loader already classified it as
|
||||
`usable` (see [memory-map.md](memory-map.md)), so it's just part of the `usable`
|
||||
regions `init` frees. Keeping that boot-protocol knowledge on the loader side is
|
||||
@@ -114,11 +122,13 @@ leaves the single region containing it `reserved`, so `init` won't hand it out.
|
||||
later step will move task 0 onto a kernel-owned stack, freeing that last ~1 MiB
|
||||
region too (and giving user mode the clean stack it wants).
|
||||
|
||||
## What's next (not done here)
|
||||
## What's next (partly done since)
|
||||
|
||||
- **Contiguous allocation** — scan for N consecutive free bits — for callers that
|
||||
need physically adjacent frames.
|
||||
- **A kernel stack for task 0**, so the boot stack's region can be freed too (and
|
||||
for the clean stack user mode wants).
|
||||
- **Freeing the `reserved` `loader_data`** (the boot-time map buffers) once the
|
||||
kernel is done reading the memory map.
|
||||
- **Contiguous allocation** — done: `allocContiguous` scans for a run of clear
|
||||
bits, with an optional physical ceiling for DMA (`dma_alloc` is its user), and
|
||||
`allocBelow` serves the SMP trampoline.
|
||||
- **A kernel stack for task 0** — still open: the boot processor's idle task runs
|
||||
on the boot stack to this day, so that region can't be freed.
|
||||
- **Freeing the `reserved` `loader_data`** (the boot-time map buffers) — still
|
||||
open: the bitmap deliberately tracks those frames so they *can* be freed, but
|
||||
nothing frees them yet.
|
||||
@@ -89,11 +89,13 @@ it.
|
||||
|
||||
## Two subtleties worth noting
|
||||
|
||||
1. **`width` vs `pitch` in the loops.** In `fillRow`/`copyRow` we iterate `x`
|
||||
up to `self.fb.width` — the *visible* count — but jump between rows with
|
||||
1. **`width` vs `pitch` in the loops.** In `fillRow` we iterate `x` up to
|
||||
`self.fb.width` — the *visible* count — but jump between rows with
|
||||
`pitch`. That's the correct pairing: touch only real pixels, but skip the
|
||||
full stride (including padding) to reach the next row. We never write into
|
||||
the padding, which is right.
|
||||
the padding, which is right. (A `copyRow` used to sit alongside it; it's
|
||||
gone — `scroll` was since rewritten as a writes-only screen clear, because
|
||||
reading VRAM back is uncached-slow on real hardware.)
|
||||
|
||||
2. **`volatile`.** The pointer is `volatile` because this memory is special —
|
||||
it's watched by the display hardware. `volatile` tells the compiler *"don't
|
||||
@@ -16,8 +16,8 @@ safely, until the machine is reset or powered off.
|
||||
## The core of it: `hlt`
|
||||
|
||||
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
||||
the arch module, `system/kernel/architecture/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
||||
generic kernel calls it as `arch.halt()`:
|
||||
the arch module, `system/kernel/architecture/x86_64/cpu.zig` (see [architecture.md](architecture.md)), and the
|
||||
generic kernel calls it as `architecture.halt()`:
|
||||
|
||||
```zig
|
||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
||||
@@ -56,10 +56,10 @@ door: every time an interrupt wakes the core, the loop immediately runs `hlt`
|
||||
again and it goes back to sleep. The net effect is a permanent halt that still
|
||||
sleeps between the interrupts it can't prevent.
|
||||
|
||||
(At this stage danos hasn't set up an interrupt descriptor table, so most
|
||||
interrupts aren't even something we handle — but non-maskable interrupts and
|
||||
system-management interrupts can still wake a halted core regardless. The loop
|
||||
makes us robust to all of them.)
|
||||
(danos handles plenty of interrupts through its interrupt descriptor table —
|
||||
the timer tick waking a halted idle core is exactly how scheduling works — and
|
||||
non-maskable and system-management interrupts can wake a halted core regardless
|
||||
of what we handle. The loop makes the halt robust to all of them.)
|
||||
|
||||
## The `asm volatile` part
|
||||
|
||||
@@ -79,32 +79,40 @@ treats the call:
|
||||
|
||||
- Code *after* a `noreturn` call is unreachable, so the compiler needn't emit a
|
||||
return sequence, and won't warn about "missing return value" in the callers.
|
||||
- It lets `kmain` and `_start` themselves be `noreturn`, which is the honest
|
||||
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
||||
jumps back out.
|
||||
- It lets `kmain` and the exported entry shim `kmainEntry` themselves be
|
||||
`noreturn`, which is the honest signature for a kernel entry point — the
|
||||
bootloader jumps in and nothing ever jumps back out.
|
||||
|
||||
You can see the chain in `system/kernel/kernel.zig`: `_start` is `noreturn`, it calls
|
||||
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
||||
`noreturn`. The "never returns" property is threaded all the way down.
|
||||
You can see the chain in the code: `_start` (an assembly stub in
|
||||
`system/kernel/architecture/x86_64/isr.s`) installs a kernel-owned stack and calls
|
||||
`kmainEntry` in `system/kernel/kernel.zig`, which is `noreturn`; it calls `kmain`,
|
||||
also `noreturn`, which ends by calling `architecture.halt()`, again `noreturn`.
|
||||
The "never returns" property is threaded all the way down.
|
||||
|
||||
## Where danos halts
|
||||
|
||||
There are three halt sites, and they're all the same idea:
|
||||
|
||||
1. **Normal end of kernel work** — `kmain` prints its status, then calls
|
||||
`arch.halt()`:
|
||||
1. **The BSP's idle loop** — `kmain` no longer runs out of work: it spawns
|
||||
`/system/services/init` as PID 1, drops itself to priority 0, and ends as the
|
||||
bootstrap core's idle task — still by calling `architecture.halt()`:
|
||||
|
||||
```zig
|
||||
con.write("\nkernel initialised; nothing left to do, halting.\n");
|
||||
arch.halt();
|
||||
scheduler.setPriority(0);
|
||||
status("\n/system/kernel: kernel idle; user space is running.\n");
|
||||
architecture.halt();
|
||||
```
|
||||
|
||||
There's genuinely nothing more to do yet, so the kernel parks itself.
|
||||
The timer keeps preempting the idle context into init and whatever else is
|
||||
ready; between those interrupts, the halt loop is exactly the low-power park
|
||||
described above.
|
||||
|
||||
2. **Kernel panic** — the freestanding panic handler has no OS to report to, so
|
||||
it prints the message in red (if the console is up) and halts via the same
|
||||
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
||||
than limping on with corrupted state — is the safe response.
|
||||
it prints the message to the diagnostic log and the on-screen console —
|
||||
forcing the console back on even if a display service had it suppressed — and
|
||||
halts via the same `architecture.halt()`. A panic is unrecoverable here, so
|
||||
stopping the machine — rather than limping on with corrupted state — is the
|
||||
safe response.
|
||||
|
||||
3. **Bootloader failure** — in `boot/efi.zig`, if `boot()` fails *before* handing
|
||||
off to the kernel, `main` logs the error and parks the machine with the same
|
||||
@@ -112,14 +120,14 @@ There are three halt sites, and they're all the same idea:
|
||||
|
||||
```zig
|
||||
boot() catch |err| {
|
||||
log("\r\ndanos: boot failed: ");
|
||||
log("\r\nEFI: boot failed: ");
|
||||
logBytes(@errorName(err));
|
||||
log("\r\n");
|
||||
while (true) asm volatile ("hlt");
|
||||
};
|
||||
```
|
||||
|
||||
(Here it's an inline loop rather than `arch.halt()` because that lives in the
|
||||
(Here it's an inline loop rather than `architecture.halt()` because that lives in the
|
||||
kernel's arch module, and the loader is a separate binary from the kernel.)
|
||||
|
||||
## Summary
|
||||
@@ -132,5 +140,5 @@ There are three halt sites, and they're all the same idea:
|
||||
otherwise wake the core and let execution continue.
|
||||
- **`asm volatile`** emits the instruction and forbids the compiler from removing
|
||||
it; **`noreturn`** encodes "control never comes back" into the type system.
|
||||
- danos halts on normal completion, on a kernel panic, and on a bootloader error
|
||||
— the same "stop safely and stay stopped" in all three.
|
||||
- danos halts in the kernel's idle loop, on a kernel panic, and on a bootloader
|
||||
error — the same "park the core safely" in all three.
|
||||
@@ -8,7 +8,7 @@ the thing that unlocks dynamic data structures — lists, hash maps, driver stat
|
||||
eventually a process table.
|
||||
|
||||
It's generic kernel code (`system/kernel/heap.zig`): the allocator logic is
|
||||
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
||||
architecture-neutral, using `architecture.mapPage` and the frame allocator underneath.
|
||||
|
||||
## A growable free-list allocator
|
||||
|
||||
@@ -30,9 +30,10 @@ Allocations are 16-byte aligned; larger alignments aren't supported yet (the
|
||||
## Growing on demand
|
||||
|
||||
The heap lives in the **higher half** of the address space (virtual base
|
||||
`0xFFFF_8000_0000_0000`) — unmapped, well clear of the identity-mapped low half,
|
||||
and leaving the low half free for a future user address space. (That base is
|
||||
x86_64-canonical; another architecture would pick its own.)
|
||||
`0xFFFF_8000_0000_0000`) — unmapped, well clear of the low half, which belongs
|
||||
to user space (unmapped in the kernel's own tables; per-process user address
|
||||
spaces now map into it). (That base is x86_64-canonical; another architecture
|
||||
would pick its own.)
|
||||
|
||||
When the free list can't satisfy a request, `grow` extends the mapped region: it
|
||||
pulls fresh frames from the [frame allocator](frame-allocator.md) and `map`s each
|
||||
@@ -50,7 +51,7 @@ rest — works directly on the kernel heap, no bespoke containers required.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `heap` test (see [testing.md](testing.md)) exercises the allocator end to end:
|
||||
The `heap` test (see [testing.md](../testing.md)) exercises the allocator end to end:
|
||||
|
||||
```
|
||||
[PASS] alloc 4096 bytes
|
||||
@@ -65,11 +66,13 @@ the proof that free and the free list actually work, not just alloc; "heap growt
|
||||
forces allocation past the initial page so `grow`/`map` runs; and the `ArrayList`
|
||||
check is the std-integration payoff.
|
||||
|
||||
## What's next (not done here)
|
||||
## What's next (largely still true)
|
||||
|
||||
- **Thread/interrupt safety.** The heap assumes a single caller — no lock yet.
|
||||
It's safe now (nothing allocates from interrupt handlers), but threads or an
|
||||
allocating IRQ handler will need a lock (or `cli` around the critical section).
|
||||
- **Larger alignments** than 16 (for page-aligned buffers, DMA regions).
|
||||
- **`resize`/`remap` in place**, so growing an `ArrayList` needn't always copy.
|
||||
- **Reclaiming empty tail pages** back to the frame allocator when the heap shrinks.
|
||||
- **Thread/interrupt safety** — overtaken by the big kernel lock: SMP arrived
|
||||
with a single kernel lock taken at every kernel entry, which serializes all
|
||||
heap access. The heap still has no lock of its own, and needs none unless the
|
||||
big lock is ever split.
|
||||
- **Larger alignments** than 16 — still unsupported; page-aligned and DMA
|
||||
buffers come straight from the frame allocator instead.
|
||||
- **`resize`/`remap` in place** — still not done; growing an `ArrayList` copies.
|
||||
- **Reclaiming empty tail pages** — still not done; the heap only ever grows.
|
||||
@@ -8,9 +8,11 @@ which on real hardware and in QEMU means a silent reset. Debugging by spontaneou
|
||||
reboot is miserable.
|
||||
|
||||
This is the machinery that catches those faults and prints what happened instead.
|
||||
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
||||
`system/kernel/architecture/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
||||
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
||||
It's all x86_64-specific, so it lives behind the [architecture](architecture.md) boundary in
|
||||
`system/kernel/architecture/x86_64/`. Only the 32 CPU-defined exception vectors were wired up at
|
||||
this stage; device interrupts (timer, keyboard, via the APIC) came later, on the
|
||||
same IDT — today it installs 48 gates (vectors 0-47) plus the ring-3 syscall gate
|
||||
at vector 128.
|
||||
|
||||
## First the GDT
|
||||
|
||||
@@ -20,9 +22,11 @@ IDT gate names a code-segment *selector* that must resolve in the current GDT. T
|
||||
firmware left a GDT in place, but we don't control it, so we install our own with
|
||||
known selectors: `0x08` kernel code, `0x10` kernel data.
|
||||
|
||||
`system/kernel/architecture/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
||||
plus code and data — where the only bits that matter in long mode are the access
|
||||
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
||||
`system/kernel/architecture/x86_64/gdt.zig` held three flat descriptors at this point — a required
|
||||
null entry, plus code and data — where the only bits that matter in long mode are
|
||||
the access byte and the code segment's long-mode (`L`) flag. (The table has since
|
||||
grown to seven entries: ring-3 user data and user code descriptors arrived with
|
||||
user mode, and the TSS descriptor — below — spans two slots.) Loading it (`gdt_flush` in
|
||||
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
||||
registers take a plain `mov`, but **CS can't** — so we reload it with a far
|
||||
return, pushing the new selector and a return address and letting `lretq` pop them
|
||||
@@ -34,7 +38,8 @@ The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
||||
entry is a 16-byte *gate* holding the handler's address (split across three
|
||||
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
||||
means present, ring 0, 64-bit interrupt gate. `system/kernel/architecture/x86_64/idt.zig` builds the
|
||||
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
||||
table, points the first 32 vectors at their stubs (since grown to gates 0-47,
|
||||
plus the ring-3 syscall gate at vector 128), and loads it with `lidt`
|
||||
(`idt_flush`).
|
||||
|
||||
## The TSS and the double-fault stack
|
||||
@@ -52,9 +57,11 @@ third time and **triple-fault** — an instant reset. So the #DF gate is pointed
|
||||
**IST1**, a small dedicated stack (`system/kernel/architecture/x86_64/tss.zig`) that's always valid.
|
||||
|
||||
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
||||
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
||||
descriptor in the GDT (`gdt.setTssFor`), and load it into the task register with
|
||||
`ltr`. The TSS descriptor is a 16-byte system descriptor spanning two GDT slots,
|
||||
which is why the GDT grew from three entries to five.
|
||||
which is why the GDT grew from three entries to five (and later to seven, when
|
||||
user mode slotted the ring-3 user data/user code descriptors between the kernel
|
||||
data entry and the TSS pair).
|
||||
|
||||
## The stubs and the trap frame
|
||||
|
||||
@@ -74,12 +81,12 @@ push order in `isr.s`; the two must stay in sync.
|
||||
The stubs and table-loads are real assembly rather than Zig inline asm because
|
||||
they need things inline asm on this toolchain can't express: cross-symbol
|
||||
`jmp`/`call` (a stub jumping to `isr_common`, which calls the exported
|
||||
`exceptionHandler`), and the `lgdt`/`lidt` memory operands (which LLVM rejects
|
||||
`interruptDispatch`), and the `lgdt`/`lidt` memory operands (which LLVM rejects
|
||||
inline). `build.zig` adds `isr.s` to the arch module.
|
||||
|
||||
## Reporting a fault
|
||||
|
||||
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
||||
`isr_common` calls `interruptDispatch`, which forwards to a swappable `on_fault`
|
||||
hook. The generic kernel installs a reporter (`onException` in `kernel.zig`) that
|
||||
prints the exception name and vector, the error code, the faulting RIP
|
||||
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
||||
@@ -122,14 +129,23 @@ fault** — reported cleanly (`double fault (vector 8)`) rather than triple-faul
|
||||
into a reset, which only works because #DF ran on IST1. That's the proof the
|
||||
TSS/IST is wired up: the handler survived a completely broken stack.
|
||||
|
||||
## What's next (not done here)
|
||||
## What's next (since done)
|
||||
|
||||
- **The IO-APIC and the keyboard**: the timer (a local-APIC device interrupt) is
|
||||
covered in [device-interrupts.md](device-interrupts.md); external devices like
|
||||
the keyboard also need the IO-APIC to route their lines onto vectors.
|
||||
- **SSE state**: the stubs save general registers but not the vector registers, so
|
||||
a returning interrupt whose handler uses SSE will need that added. Fine for now,
|
||||
since our handlers don't.
|
||||
Both items originally deferred here have landed:
|
||||
|
||||
- **The IO-APIC**: [ioapic.zig](../../system/kernel/architecture/x86_64/ioapic.zig)
|
||||
routes external device lines onto vectors — discovered via ACPI's MADT, every
|
||||
input masked at init, lines unmasked one at a time as user-space drivers bind
|
||||
them (see [device-interrupts.md](../device-driver-development/device-interrupts.md)). The keyboard followed
|
||||
exactly as predicted: the PS/2 bus driver (`system/drivers/ps2-bus/`) claims
|
||||
the 8042 controller and binds its IRQ 1 (and the aux mouse's IRQ 12) through
|
||||
this routing. USB HID keyboards arrive over xHCI instead, which interrupts via
|
||||
MSI, and the HPET's GSI routing exercises the same path.
|
||||
- **SSE state**: `isr_common` (and the syscall entry) now `fxsave`/`fxrstor` the
|
||||
full SSE/x87 register file around dispatch. This stopped being optional the
|
||||
moment kernel code touched XMM — a 16-byte struct copy is a `movdqu` — and its
|
||||
absence was the root cause of a long-lived corruption Heisenbug; the comments
|
||||
in `isr.s` tell the story.
|
||||
|
||||
With faults now debuggable, the paging work that follows — where a wrong
|
||||
page-table entry means an instant #PF — is far less painful.
|
||||
@@ -0,0 +1,100 @@
|
||||
# Logging
|
||||
|
||||
Output is a *diagnostic convenience, never a correctness dependency*: the kernel
|
||||
and every service must run correctly with zero output channels. On top of that
|
||||
rule, danos has **per-process logging** — every process's output is attributed
|
||||
by the kernel and lands in its own file on the flash volume, which is what makes
|
||||
a headless real machine (no serial port) debuggable. The display (the
|
||||
framebuffer surface) is a separate concern with one bootstrap exception: the
|
||||
kernel's framebuffer console (`system/kernel/console.zig`) joins the log sinks
|
||||
at boot, so the whole transcript shows on screen until the display service
|
||||
claims the framebuffer and silences it; after that only panic/fatal messages
|
||||
are mirrored to it explicitly (`system/kernel/kernel.zig`).
|
||||
|
||||
## The pipeline
|
||||
|
||||
```
|
||||
process std.log ──▶ debug_write(level) ──▶ tagged kernel ring ──▶ logger service ──▶ /var/log/<boot-stamp>/<binary-path>.log
|
||||
kernel log.print ─┘ │
|
||||
└▶ serial / 0xE9 sinks (QEMU, -Dserial)
|
||||
```
|
||||
|
||||
1. **Emit.** A program calls `std.log.info("mounted {s}", .{path})` — the
|
||||
runtime's `logFn` (installed for every binary by the root shim,
|
||||
`library/kernel/logging.zig`) formats one line and issues one `debug_write`
|
||||
carrying the level. The payload does NOT contain the process's name.
|
||||
`logging.write` remains as the raw/bring-up path (panics, test
|
||||
fixtures); raw bytes ride the same ring, attributed all the same.
|
||||
|
||||
2. **Stamp.** The kernel wraps every payload LINE in a record stamped with the
|
||||
sender's pid, task name (its binary path, e.g. `/system/services/fat`),
|
||||
level, a per-boot sequence number, and a monotonic timestamp
|
||||
(`system/kernel/log.zig` + `log-ring.zig`). Attribution is structural — a
|
||||
payload cannot forge another sender's tag, and an embedded newline just ends
|
||||
the record, so the forged "prefix" lands inside the forger's own next line.
|
||||
|
||||
3. **Retain.** The 512 KiB ring overwrites oldest-first; sequence gaps make any
|
||||
loss countable. `klog_read` (#32) copies stream bytes from a free-running
|
||||
offset; `klog_status` (#45) returns the cursors plus the wall-clock time of
|
||||
boot. The framing (`abi.KlogRecordHeader`) is 32 bytes + name + payload,
|
||||
8-byte aligned.
|
||||
|
||||
4. **Render.** Registered sinks (serial under `-Dserial`, the 0xE9 debug
|
||||
console, and the framebuffer console until the display service claims the
|
||||
screen) get a live transcript: kernel/raw output verbatim, leveled records
|
||||
as `<binary path>: message` — one composed write per line, under the log's
|
||||
own spinlock (never the big kernel lock; panic paths try-acquire with a
|
||||
bound and fall back to sinks-only). Sinks are best-effort and self-guarding;
|
||||
a serial-less machine just goes quiet.
|
||||
|
||||
5. **Persist.** The **logger service** (`system/services/logger`) drains the
|
||||
ring every 250 ms and demultiplexes records into one file per source under
|
||||
`/var/log/<boot-stamp>/`, e.g.
|
||||
|
||||
```
|
||||
/var/log/2026-07-21T150434Z/kernel.log
|
||||
/var/log/2026-07-21T150434Z/system/services/fat.log
|
||||
/var/log/2026-07-21T150434Z/system/drivers/usb-storage.log
|
||||
```
|
||||
|
||||
The boot stamp is the RTC anchor from `klog_status` (FAT-safe: no colons; a
|
||||
dead RTC yields the 1970 directory rather than no logs). Each line carries
|
||||
the record's monotonic timestamp and level. Storage is best-effort and late:
|
||||
the ring buffers a whole boot many times over, and the first successful
|
||||
`makePath` of the per-boot directory (also the readiness probe) triggers a
|
||||
full backlog write. Files close — which is the fat server's SCSI cache
|
||||
flush — after a ~2 s quiet period, bounding data-at-risk without per-record
|
||||
flush thrash. At shutdown init stops the logger FIRST (it is last in the
|
||||
boot order), so its final drain runs over a live storage chain.
|
||||
|
||||
## Why a ring in the kernel, not a logging server
|
||||
|
||||
The storage stack must be able to log. If the fat server wrote its own log file
|
||||
through the VFS it would rendezvous-deadlock on itself; if processes sent
|
||||
records to a logging server over IPC, early boot would need a buffer that is —
|
||||
a ring, one hop later. The kernel ring is that buffer, placed where every
|
||||
process (and the kernel itself) can reach it with one syscall, before any
|
||||
service exists. The logger service is a *reader*, not a hop.
|
||||
|
||||
Two disciplines keep it honest:
|
||||
|
||||
- the logger announces itself **once** — a periodic status line would feed the
|
||||
very stream it drains;
|
||||
- lost records surface as an explicit `-- N records lost --` line, computed
|
||||
from sequence gaps, never silently.
|
||||
|
||||
## Last-resort channels
|
||||
|
||||
Unchanged, and independent of the sink list so they survive a total output
|
||||
failure: `checkpoint` (a one-byte POST code on port 0x80) and `recordPanic`
|
||||
(a fixed breadcrumb record, `log.panic_record`, findable in a RAM dump; magic
|
||||
written last so a reader only trusts a complete record).
|
||||
|
||||
## Accepted gaps
|
||||
|
||||
- A write-spamming process can evict other processes' unread records from the
|
||||
ring (a per-process quota is future work); the loss is at least visible via
|
||||
sequence gaps in every affected file.
|
||||
- `/var/log` files have no privacy until the VFS grows permissions.
|
||||
- Records emitted after the logger's final shutdown drain reach serial and the
|
||||
ring but not the files.
|
||||
@@ -16,7 +16,7 @@ forward it to the kernel as-is. We don't, for two reasons:
|
||||
memory-type numbers and walk the array using UEFI's variable descriptor stride.
|
||||
That's UEFI vocabulary bleeding across the handoff — and danos wants to boot on
|
||||
systems that have no UEFI at all (a Raspberry Pi describes its memory with a
|
||||
*device tree* instead). See [arch.md](arch.md) for the same "keep the kernel
|
||||
*device tree* instead). See [architecture.md](architecture.md) for the same "keep the kernel
|
||||
platform-agnostic" principle applied to CPU code.
|
||||
2. **We already established the better pattern.** The loader doesn't hand the
|
||||
kernel a raw UEFI GOP either — [`queryFramebuffer`](gop.md) converts it to
|
||||
@@ -57,22 +57,27 @@ itself, `@sizeOf` is authoritative: the kernel walks a plain `[]MemoryRegion` wi
|
||||
no variable-stride subtlety (that stride problem is a UEFI-ism, and it stays in the
|
||||
loader).
|
||||
|
||||
`BootInfo` carries it alongside the framebuffer:
|
||||
`BootInformation` carries it alongside the framebuffer (trimmed here to the
|
||||
fields this page is about — the full struct has since grown the kernel's
|
||||
PT_LOAD segments, the ACPI RSDP, and the initial-ramdisk span):
|
||||
|
||||
```zig
|
||||
pub const BootInfo = extern struct {
|
||||
pub const BootInformation = extern struct {
|
||||
framebuffer: Framebuffer,
|
||||
memory_map: MemoryMap,
|
||||
// ...kernel_segments, acpi_rsdp, initial_ramdisk_base/len
|
||||
};
|
||||
```
|
||||
|
||||
## The loader side (UEFI)
|
||||
|
||||
Two functions in `boot/efi.zig`, called from `exitBootServices`:
|
||||
Two functions in `boot/efi.zig`: `exitBootServices` calls `convertMemoryMap`,
|
||||
which runs `classify` on each descriptor:
|
||||
|
||||
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
||||
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
||||
`acpi_reclaim_memory → acpi_tables`; `acpi_memory_nvs → acpi_nvs`; **everything
|
||||
`acpi_reclaim_memory → acpi_tables`; `acpi_memory_nvs → acpi_nvs`;
|
||||
`memory_mapped_io`/`memory_mapped_io_port_space → mmio`; **everything
|
||||
else → reserved** (the safe default). Our own `loader_data` — the kernel image and
|
||||
these buffers — falls into `reserved`.
|
||||
|
||||
@@ -120,18 +125,25 @@ for the conversion.)
|
||||
The kernel receives a plain array and reads it with zero UEFI knowledge:
|
||||
|
||||
```zig
|
||||
const mm = boot_info.memory_map;
|
||||
const regions = @as([*]const system.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
const mm = boot_information.memory_map;
|
||||
const regions = @as(
|
||||
[*]const boot_handoff.MemoryRegion,
|
||||
@ptrFromInt(boot_handoff.physicalToVirtual(mm.regions)),
|
||||
)[0..mm.len];
|
||||
for (regions) |r| {
|
||||
if (r.kind == .usable) usable_pages += r.pages;
|
||||
}
|
||||
```
|
||||
|
||||
(`mm.regions` is a physical address, so it's dereferenced through the physmap —
|
||||
`physicalToVirtual` — since the kernel no longer runs under the loader's
|
||||
identity map.)
|
||||
|
||||
`kmain` summarises the map to prove the handoff works. Booted in QEMU with
|
||||
128 MiB, it reports:
|
||||
|
||||
```
|
||||
danos: physical memory
|
||||
/system/kernel: physical memory
|
||||
total RAM : 0.12 GiB (127 MiB) - RAM the firmware reported
|
||||
usable : 121 MiB - free RAM (incl. reclaimed boot-services memory)
|
||||
reserved : 6 MiB - kernel image, boot stack, ACPI, runtime services
|
||||
@@ -158,11 +170,13 @@ never knows the difference.
|
||||
This page is plumbing plus classification only. The map's first consumer, the
|
||||
**physical frame allocator**, is built directly on the `usable` regions here — which
|
||||
already include the reclaimed boot-services memory the loader folded in (see
|
||||
[frame-allocator.md](frame-allocator.md)). Still to come:
|
||||
[frame-allocator.md](frame-allocator.md)). Of the two items once listed here, one is done:
|
||||
|
||||
- Freeing the `reserved` `loader_data` (these boot-time buffers) once the kernel is
|
||||
done reading the map.
|
||||
- Freeing the `reserved` `loader_data` (these boot-time buffers) once the kernel
|
||||
is done reading the map — still open: the frame allocator's bitmap tracks those
|
||||
frames so they can be freed, but nothing frees them yet.
|
||||
- Capturing the ACPI RSDP from the UEFI configuration table before exit (the same
|
||||
"grab it before ExitBootServices" pattern), for when ACPI parsing arrives.
|
||||
"grab it before ExitBootServices" pattern) — done: the loader stows it in the
|
||||
boot handoff, and ACPI parsing consumes it from there ([acpi.md](acpi.md)).
|
||||
|
||||
See the roadmap in [efi.md](efi.md) for where this sits in the boot flow.
|
||||
@@ -7,15 +7,19 @@ which live in memory we'd like to reclaim and don't control), switches CR3 onto
|
||||
them, and — crucially — maps with **real permissions**.
|
||||
|
||||
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
||||
behind the [arch](arch.md) boundary in `system/kernel/architecture/x86_64/paging.zig`.
|
||||
behind the [architecture](architecture.md) boundary in `system/kernel/architecture/x86_64/paging.zig`.
|
||||
|
||||
## The format
|
||||
|
||||
x86_64 uses **4 levels**: PML4 → PDPT → PD → PT, each a 512-entry table, with 9
|
||||
bits of the virtual address indexing each level and the low 12 bits the offset into
|
||||
the final 4 KiB page. Each entry holds a physical address plus flag bits —
|
||||
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
||||
pages: precise, and the extra table memory is negligible against available RAM.
|
||||
present, writable, and (bit 63) **no-execute**. danos maps nearly everything with
|
||||
4 KiB pages: precise, and the extra table memory is negligible against available
|
||||
RAM. (The physmap has since become the one exception: 2 MiB-aligned RAM there is
|
||||
mapped with **2 MiB huge pages** — a PS-bit leaf at the PD level — with 4 KiB
|
||||
pages filling the unaligned edges, so the table footprint scales sanely with big
|
||||
RAM. Kernel segments, heap, user space, and on-demand MMIO stay 4 KiB.)
|
||||
|
||||
## Higher half: the address-space layout
|
||||
|
||||
@@ -27,7 +31,7 @@ address to the low load address in its bootstrap tables and jumps in). The entir
|
||||
alongside a **physmap** — a straight window onto all of physical memory at
|
||||
`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a
|
||||
page-table frame, an ACPI table, a device register), it adds that constant:
|
||||
`system.physToVirt(phys)`. The layout constants live in `system/boot-handoff.zig`:
|
||||
`boot_handoff.physicalToVirtual(phys)`. The layout constants live in `system/boot-handoff.zig`:
|
||||
|
||||
| region | virtual base | PML4 slot |
|
||||
|--------|--------------|-----------|
|
||||
@@ -46,7 +50,7 @@ builds its own precise tables below and abandons them. Because both use the same
|
||||
The address space is built in four passes (`init`):
|
||||
|
||||
1. **All RAM in the physmap, RW + NX.** Every non-MMIO region from the
|
||||
[memory map](memory-map.md) is mapped at `physToVirt(phys)`, read-write and
|
||||
[memory map](memory-map.md) is mapped at `physicalToVirtual(phys)`, read-write and
|
||||
*non-executable*. There is **no low/identity mapping** — the low half is user
|
||||
space. (Frames the kernel touches while still building these tables are reached
|
||||
through the loader's bootstrap physmap, which covers the low 4 GiB; both the
|
||||
@@ -72,7 +76,7 @@ Blanket RW+NX is fine for data but wrong for the kernel's own code, which must b
|
||||
executable — and its code must *not* be writable (W^X: no page is both). We get the
|
||||
right permissions per region straight from the kernel ELF: the **loader already
|
||||
parses the program headers**, so `efi.zig` records each `PT_LOAD` segment's
|
||||
address, size and R/W/X flags into `BootInfo`. Pass 3 re-maps those ranges with
|
||||
address, size and R/W/X flags into `BootInformation`. Pass 3 re-maps those ranges with
|
||||
flags derived from the ELF flags:
|
||||
|
||||
| segment | ELF flags | mapped as |
|
||||
@@ -97,9 +101,10 @@ real memory — turning a whole class of silent bugs into an immediate, located
|
||||
## Switching on, and the on-demand API
|
||||
|
||||
Loading the PML4's physical address into **CR3** switches address spaces and
|
||||
flushes the TLB in one step. This works because the firmware's identity map is
|
||||
still active *while we build*, so freshly allocated table frames are reachable by
|
||||
physical address; afterwards they're covered by pass 1.
|
||||
flushes the TLB in one step. This works because *while we build* the kernel is
|
||||
still running on the loader's bootstrap tables, whose 4 GiB physmap makes freshly
|
||||
allocated table frames reachable at the same `physmap_base + phys` addresses;
|
||||
afterwards they're covered by pass 1.
|
||||
|
||||
`init` keeps the PML4 and the frame allocator around and exposes `map(virt, phys,
|
||||
writable)` / `unmap(virt)` (with `invlpg` TLB invalidation) — the primitive the
|
||||
@@ -107,7 +112,7 @@ kernel heap will build on to map pages on demand.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Four tests (see [testing.md](testing.md)) pin down the guarantees:
|
||||
Four tests (see [testing.md](../testing.md)) pin down the guarantees:
|
||||
|
||||
- **`vmm`** — map a fresh frame at an unused virtual address, write and read it
|
||||
back. Proves `map` works end to end.
|
||||
@@ -124,13 +129,17 @@ Four tests (see [testing.md](testing.md)) pin down the guarantees:
|
||||
> pointer to force a real hardware access. And `invlpg`, like `lgdt`, needs its
|
||||
> operand staged through a register in inline asm.
|
||||
|
||||
## What's next (not done here)
|
||||
## What's next (mostly done since)
|
||||
|
||||
- **A kernel heap** — the first real user of `map`, giving the kernel dynamic
|
||||
allocation. This is the natural next milestone.
|
||||
- **A higher-half kernel**: relink the kernel at a high virtual base so a future
|
||||
user address space can own the low half.
|
||||
- **Per-address-space tables** once there are user processes, and shared/copy-on-
|
||||
write mappings.
|
||||
- **Uncacheable MMIO**: the APIC/framebuffer pages are mapped writeback-cacheable;
|
||||
real hardware wants MMIO marked uncacheable.
|
||||
- **A kernel heap** — done, built on `map` exactly as anticipated
|
||||
([heap.md](heap.md)).
|
||||
- **A higher-half kernel** — done: the kernel is linked at
|
||||
`0xFFFFFFFF80000000` (`linker.ld`), loaded low and running high, and user
|
||||
processes own the low half.
|
||||
- **Per-address-space tables** — done: each user process gets its own root with
|
||||
the kernel half shared, and refcounted shared-memory mappings exist
|
||||
([ipc.md](../device-driver-development/ipc.md)). Copy-on-write remains unbuilt — nothing has needed it yet.
|
||||
- **Uncacheable MMIO** — half done: user-space device and DMA mappings are
|
||||
strong-uncacheable and the framebuffer is write-combining via the PAT, but the
|
||||
kernel's own `mapMmio` path is still writeback — the LAPIC included (see
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md)).
|
||||
@@ -7,7 +7,7 @@ them owns the hardware that reported the event, and the reporter should not know
|
||||
who is listening. So system power is a **service**: an event source **publishes**
|
||||
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||
privileged caller — init — can ask it to power the machine off. It is the same
|
||||
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||
publish/subscribe shape as the [input service](../device-driver-development/input.md), applied to power.
|
||||
|
||||
## Why a service, and why it is named for the domain, not the firmware
|
||||
|
||||
@@ -17,7 +17,7 @@ What subscribers want is not: *the lid closed* means the same thing regardless o
|
||||
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||
Subscribers call `ipc.lookup(.power)` and never learn which firmware they
|
||||
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||
preserve, carried one layer up into a running-system surface.
|
||||
|
||||
@@ -27,7 +27,7 @@ unchanged.
|
||||
|
||||
## The protocol
|
||||
|
||||
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||
The `power-protocol` module ([library/protocol/power/power-protocol.zig](../../library/protocol/power/power-protocol.zig))
|
||||
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||
fields. Three operations:
|
||||
|
||||
@@ -75,7 +75,7 @@ notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||
On a `power_button` event or a `terminate` signal, init:
|
||||
|
||||
1. logs that it is shutting down,
|
||||
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||
2. runs the standard stop sequence — `process.stop(child, deadline,
|
||||
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||
last (other services may flush through it), each child getting the
|
||||
*terminate → deadline → kill* escalation from
|
||||
@@ -83,16 +83,18 @@ On a `power_button` event or a `terminate` signal, init:
|
||||
3. requests `.power` `shutdown`.
|
||||
|
||||
The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the
|
||||
PM1 control register(s) from ring 3, mirroring the kernel's own
|
||||
`system/devices/power.zig` `sleepValue`. If the write returns instead of powering
|
||||
the machine off, it logs loudly so a test fails rather than hangs.
|
||||
PM1 control register(s) from ring 3, with the `SLP_TYP` values taken from its own
|
||||
AML parse of the `_S5` object — the kernel has no S5 path of its own. If the
|
||||
write returns instead of powering the machine off, it logs loudly so a test
|
||||
fails rather than hangs.
|
||||
|
||||
**No new system call was needed for S5.** The broad io_port grant on the
|
||||
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||
could physically already do; formalizing it as a protocol operation added a
|
||||
contract, not authority. The kernel keeps `power.zig` for its own test paths and
|
||||
panic-time poweroff, where no user space is available to ask.
|
||||
contract, not authority. The kernel keeps only **reboot** (`acpi.reboot` in
|
||||
`system/kernel/acpi.zig` — the FADT reset register plus the legacy fallbacks, which
|
||||
need no AML); it has no poweroff path at all — S5 is not a kernel operation.
|
||||
|
||||
## Verifying it
|
||||
|
||||
@@ -124,5 +126,5 @@ until laptop sleep), and thermal zones.
|
||||
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||
- [device-manager.md](../device-driver-development/device-manager.md) — the supervision model init mirrors for
|
||||
its own children.
|
||||
@@ -6,10 +6,10 @@ harness are all in — the interface below is as-built. The primitives underneat
|
||||
predate this design ([process-management.md](process-management.md):
|
||||
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||
life, and the stable `process` interface that carries it. Nothing here is
|
||||
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||
reload, and die the same way. The device manager is simply this design's first
|
||||
serious customer ([device-manager.md](device-manager.md)).
|
||||
serious customer ([device-manager.md](../device-driver-development/device-manager.md)).
|
||||
|
||||
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||
@@ -19,7 +19,7 @@ rule is danos's own and it is strict: plain words that communicate intent
|
||||
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||
a concept that already has one. Literal POSIX arrives later and lives elsewhere: the
|
||||
`std.os.danos` seam that makes danos a Zig target, and eventually a **musl-based C
|
||||
layer** on the same native surface (see [zig-self-hosting.md](zig-self-hosting.md)) —
|
||||
layer** on the same native surface (see [zig-self-hosting.md](../zig-self-hosting.md)) —
|
||||
musl's syscall surface retargeted at danos system calls and IPC protocols (files onto
|
||||
the VFS protocol, `sigaction`/`wait` onto this lifecycle, sockets onto whatever
|
||||
networking becomes). Ported programs see POSIX; the system underneath never does.
|
||||
@@ -55,10 +55,17 @@ pattern reused. Signals are the same pattern reused a third time.
|
||||
- **`process_signal(id, signal)`** — posts the signal as an asynchronous
|
||||
notification to the target's bound endpoint: badge = `notify_badge_bit |
|
||||
notify_signal_bit | pending signals`. Non-blocking for the sender, always.
|
||||
- **Pending signals coalesce** in a per-process bitmask until the target next waits
|
||||
— exactly like interrupt notifications, and exactly POSIX's own semantics for
|
||||
non-realtime signals (two pending SIGTERMs are one SIGTERM). The bitmask *is* the
|
||||
design: signals carry no payload. Anything with a payload is a protocol message.
|
||||
Signals address the *process*: `id` may name any member of a threaded process
|
||||
and resolves to its leader — whose endpoint the harness binds — with authority
|
||||
mirroring `process_kill` ([shared-fate-plan.md](shared-fate-plan.md)).
|
||||
- **Pending signals coalesce** in a per-process bitmask while the target has no
|
||||
signal endpoint bound, and the whole mask arrives as one notification at bind —
|
||||
POSIX's own semantics for non-realtime signals (two pending SIGTERMs are one
|
||||
SIGTERM). Once bound, each `process_signal` flushes the mask straight into the
|
||||
endpoint's notification ring, so a busy receiver drains separate posts as
|
||||
separate notifications — harmless, because the badge is a set of bits, never a
|
||||
count. The bitmask *is* the design: signals carry no payload. Anything with a
|
||||
payload is a protocol message.
|
||||
- **Authority**: the supervisor may signal its children — the same link that is
|
||||
already the kill authority. A process may signal itself. Anything broader waits
|
||||
for transferable process handles.
|
||||
@@ -85,7 +92,7 @@ defense below the table.
|
||||
| SIGUSR1, SIGUSR2 | signals `user_1`, `user_2` | service-defined |
|
||||
| SIGCHLD | **already exists** — the exit notification | the badge carries the child id, dodging the classic coalescing bug (Unix code must loop `waitpid`) |
|
||||
| SIGKILL | `process_kill` — kernel mechanism | its definition is "cannot be handled"; it was never really a signal |
|
||||
| SIGABRT | exit reason `abort` | `abort()` is synchronous self-termination, not an event |
|
||||
| SIGABRT | exit reason `aborted` | synchronous self-termination is an exit, not an event; recorded for any nonzero exit code |
|
||||
| SIGSEGV, SIGILL, SIGFPE | exit reasons, **never delivered** | see below |
|
||||
| SIGPIPE | **an error return**, not a signal | see below |
|
||||
| SIGSTOP, SIGTSTP, SIGTTIN, SIGTTOU, SIGCONT | deferred | job control needs terminals, sessions, and process groups; stop/continue is scheduler territory |
|
||||
@@ -135,18 +142,21 @@ deadline.
|
||||
|
||||
1. **Cleanup is the kernel's job.** A process can die with no warning — fault,
|
||||
kill, power. Correctness must never depend on a `terminate` handler running. On
|
||||
any death the kernel releases the address space, IPC handles, IRQ bindings, and
|
||||
owed replies (built), and must also release **device, I/O-port, and interrupt
|
||||
claims and MSI vectors** (the known gap in
|
||||
[process-management.md](process-management.md); increment 1). A signal handler is
|
||||
any death the kernel releases the address space, IPC handles, IRQ bindings,
|
||||
owed replies, and **device, I/O-port, and interrupt claims and MSI vectors** —
|
||||
the last of these was once the known gap in
|
||||
[process-management.md](process-management.md), closed by increment 1
|
||||
(`releaseTaskResourcesLocked`, on every death path). A signal handler is
|
||||
for *graceful* work — flushing, deregistering, saving — never for *necessary*
|
||||
work.
|
||||
2. **Kill is not a signal, and exit reasons are load-bearing.** The standard stop
|
||||
sequence is *terminate → deadline → `process_kill`*; the unhandleable kill stays
|
||||
a kernel mechanism. And a supervisor deciding whether to restart must know *how*
|
||||
the child died: clean exit (meant to — don't restart), fault (restart with
|
||||
backoff), killed (the supervisor did it). The exit notification today carries
|
||||
only the id; it grows a reason. Restart policy cannot be written without it.
|
||||
backoff), killed (the supervisor did it). The exit notification carries only the
|
||||
id; the reason is recorded before the notification posts and read with the
|
||||
supervisor-gated `process_exit_reason` query. Restart policy cannot be
|
||||
written without it.
|
||||
|
||||
## Who learns of a death
|
||||
|
||||
@@ -154,25 +164,27 @@ A death has three audiences, and conflating them is how systems end up with eith
|
||||
zombie state or privileged snooping:
|
||||
|
||||
1. **The supervisor** — gets the exit notification on the endpoint it gave at spawn
|
||||
(built), which grows the `ExitReason` (increment 2). The supervisor is the only
|
||||
(built), then reads the `ExitReason` with the `process_exit_reason` query
|
||||
(increment 2). The supervisor is the only
|
||||
audience that needs the *reason*, because it is the only one deciding whether to
|
||||
restart.
|
||||
2. **The peer owed a reply** — already built: a client that dies mid-request fails
|
||||
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||
the same way. This covers the *synchronous* case only.
|
||||
3. **The subscribers** — the new piece, and it is the input service's
|
||||
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||
service accumulates per-client state across many requests: the VFS holds a dead
|
||||
client's open file handles, the input service holds its subscriptions, a future
|
||||
network stack holds its sockets. None of these are the client's supervisor, and
|
||||
none learn anything from a failed reply if the client simply never calls again.
|
||||
publish/subscribe shape ([input.md](../device-driver-development/input.md)) applied to exits. A stateful
|
||||
service accumulates per-client state across many requests: a filesystem server
|
||||
(FAT today) holds a dead client's open file handles, the input service holds
|
||||
its subscriptions, a future network stack holds its sockets. None of these
|
||||
are the client's supervisor, and none learn anything from a failed reply if
|
||||
the client simply never calls again.
|
||||
So the kernel **publishes every exit** to whoever subscribed:
|
||||
`process_subscribe(endpoint)` adds a subscriber, and each death posts a
|
||||
notification to every subscriber (badge = `notify_exit_bit | process id` — the
|
||||
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||
subscriber filters for ids it holds state for and releases what the dead client
|
||||
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||
task id (`ipc.Received`), so the id a service has been keying client
|
||||
state by all along is the id the exit event carries.
|
||||
|
||||
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||
@@ -180,16 +192,17 @@ zombie state or privileged snooping:
|
||||
non-blocking coalescing notification as everything else — a dying process never
|
||||
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||
do not receive the exit reason — the VFS does not care *why* the client died.
|
||||
do not receive the exit reason — the filesystem server does not care *why*
|
||||
the client died.
|
||||
|
||||
This is the service-side mirror of iron rule 1: **a service must never depend on
|
||||
its clients cleaning up after themselves.** Handle release on client death is the
|
||||
service's job, triggered by the published exit event — never by a courtesy
|
||||
"closing now" message that a crashed client will never send.
|
||||
|
||||
## The stable interface: `runtime.process`
|
||||
## The stable interface: `process`
|
||||
|
||||
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||
`process` already owns what a process receives at birth (`Init`, the
|
||||
argv contract). It grows to own the other end of life.
|
||||
|
||||
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||
@@ -221,7 +234,6 @@ pub const Signal = enum(u5) {
|
||||
pub const SignalSet = struct {
|
||||
pending: u32,
|
||||
pub fn has(set: SignalSet, signal: Signal) bool { ... }
|
||||
pub fn iterate(set: SignalSet) Iterator { ... }
|
||||
};
|
||||
|
||||
/// Nominate `endpoint` as this process's signal endpoint (signal_bind). The
|
||||
@@ -231,14 +243,15 @@ pub fn bindSignals(endpoint: usize) bool { ... }
|
||||
|
||||
/// Decode a received badge into signals, or null if the badge is not a signal
|
||||
/// notification (mirrors ipc.Received.isChildExit).
|
||||
pub fn signalsFrom(badge: usize) ?SignalSet { ... }
|
||||
pub fn signalsFrom(badge: u64) ?SignalSet { ... }
|
||||
|
||||
/// Send `signal` to process `id`. Supervisor-gated, like kill; non-blocking.
|
||||
pub fn sendSignal(id: u32, signal: Signal) bool { ... }
|
||||
|
||||
/// The standard stop sequence: terminate, wait up to `deadline_ms` for the exit
|
||||
/// notification, then process_kill. The one call a supervisor needs.
|
||||
pub fn stop(id: u32, deadline_ms: u64) void { ... }
|
||||
/// notification on `exit_endpoint` (the endpoint the child was spawned with),
|
||||
/// then process_kill. The one call a supervisor needs.
|
||||
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void { ... }
|
||||
|
||||
/// Subscribe `endpoint` to published exit events (process_subscribe). Every
|
||||
/// process death posts an asynchronous notification: badge = notify_exit_bit |
|
||||
@@ -252,7 +265,7 @@ pub fn subscribeExits(endpoint: usize) bool { ... }
|
||||
/// it first, so the two never race). What restart policy reads. (Built in M17.2.)
|
||||
pub const ExitReason = enum(u8) {
|
||||
exited, // returned from main / clean exit
|
||||
aborted, // abort() — deliberate self-termination (SIGABRT's ghost; reserved)
|
||||
aborted, // deliberate failure exit — any nonzero exit code (SIGABRT's ghost)
|
||||
segmentation_fault, // SIGSEGV's ghost
|
||||
illegal_instruction, // SIGILL's ghost
|
||||
arithmetic_fault, // SIGFPE's ghost
|
||||
@@ -270,7 +283,7 @@ callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||
|
||||
### The service harness
|
||||
|
||||
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||
`service` owns the `replyWait` loop and folds every event source — signals,
|
||||
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||
@@ -299,13 +312,15 @@ get POSIX; danos-native programs never pay for it.
|
||||
kill a claiming driver, spawn it again, the claim succeeds.
|
||||
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||
publishes on every death), `runtime.process.subscribeExits`; the VFS becomes the
|
||||
first subscriber — releasing a dead client's handles is its proof test.
|
||||
publishes on every death), `process.subscribeExits`; the userspace VFS
|
||||
router was the first subscriber — releasing a dead client's handles was its
|
||||
proof test — and the FAT server inherited the role when the router moved into
|
||||
the kernel (clients now hold the filesystem server's node ids directly).
|
||||
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||
`runtime.process` grows the interface above; the service harness handles
|
||||
`process` grows the interface above; the service harness handles
|
||||
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||
|
||||
[device-manager.md](device-manager.md) builds directly on all four.
|
||||
[device-manager.md](../device-driver-development/device-manager.md) builds directly on all four.
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
@@ -12,13 +12,15 @@ made the file tree the whole interface (`echo kill > /proc/n/ctl`). Microkernels
|
||||
mostly abandon ambient PIDs: Minix and QNX route everything through a user-space
|
||||
process-manager server, and Fuchsia/seL4 control processes only through handles.
|
||||
|
||||
danos rules out `/proc` **as the primitive**: here a `/proc` would be served by
|
||||
the VFS server — a user process — which would put the VFS in the path of process
|
||||
control. If the VFS (or anything under it) hangs, nothing could be listed or
|
||||
killed, *including the hung VFS*. The control plane for processes must not
|
||||
depend on a process. So the primitives are kernel system calls; a read-only
|
||||
`/proc` rendering can be layered on later, and a POSIX-style process-manager
|
||||
server can be built *from* these primitives when one is needed.
|
||||
danos rules out `/proc` **as the primitive**: the path router lives in the
|
||||
kernel (`fs_resolve`), but what is mounted under a path is served by a
|
||||
user-process filesystem server (the way FAT serves `/mnt/usb`) — a `/proc`
|
||||
would be one more such server, which would put a user process in the path of
|
||||
process control. If that server (or anything under it) hangs, nothing could be
|
||||
listed or killed, *including the hung server*. The control plane for processes
|
||||
must not depend on a process. So the primitives are kernel system calls; a
|
||||
read-only `/proc` rendering can be layered on later, and a POSIX-style
|
||||
process-manager server can be built *from* these primitives when one is needed.
|
||||
|
||||
## The three primitives
|
||||
|
||||
@@ -55,9 +57,13 @@ dangle even if the supervisor dies first.
|
||||
|
||||
### `process_kill(id) -> 0 / -ESRCH / -EPERM`
|
||||
|
||||
Only the supervisor may kill; kernel tasks are not killable processes. Like a
|
||||
signal, delivery is prompt but asynchronous — 0 means the kill is accepted and
|
||||
irrevocable; the exit notification confirms completion.
|
||||
Only the supervisor may kill; kernel tasks are not killable processes. The kill
|
||||
is a **whole-process** kill ([shared-fate-plan.md](shared-fate-plan.md)): `id`
|
||||
may name any member of a threaded process — it resolves to the group's leader,
|
||||
authorization is checked against the *leader's* supervisor, and every thread
|
||||
dies. Like a signal, delivery is prompt but asynchronous — 0 means the kill is
|
||||
accepted and irrevocable; the exit notification (badged with the leader, posted
|
||||
once the last member is gone) confirms completion.
|
||||
|
||||
## How a kill lands (the kernel mechanics)
|
||||
|
||||
@@ -100,11 +106,15 @@ the architecture layer calls up into `tick`.
|
||||
(`releaseTaskResourcesLocked`), so a restarted driver can claim its hardware
|
||||
again — the cleanup half of [process-lifecycle.md](process-lifecycle.md)'s iron
|
||||
rule 1. The `claim-release` test proves the kill → release → re-claim cycle.
|
||||
- Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet).
|
||||
- ~~Kernel stacks of dead tasks are leaked~~ Closed (threading-plan M8): a task
|
||||
exiting on its own core queues on the core's reap list in `.reaping` state, and
|
||||
the next switch away (or tick) frees its kernel stack; one killed while off-CPU
|
||||
has its stack freed synchronously by the reap itself. Both paths are accounted
|
||||
by `live_stack_bytes`, which returns to baseline when no extra tasks are live.
|
||||
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||
records how every process ends — exited, a fault class, or killed — before it
|
||||
posts the exit notification, and the supervisor reads it with
|
||||
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||
`process_exit_reason` (`process.exitReason`). This is the input to
|
||||
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||
for the clean case can still ride alongside later.
|
||||
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
@@ -44,7 +44,7 @@ tables of contents, both pointing at the same embedded FAT image:
|
||||
same `BOOTX64.efi` off it.
|
||||
|
||||
Neither path involves the legacy BIOS boot-sector machinery: danos is
|
||||
UEFI-only ([system-requirements.md](system-requirements.md)), so the MBR holds
|
||||
UEFI-only ([system-requirements.md](../system-requirements.md)), so the MBR holds
|
||||
no boot code, just the partition entry, and the El Torito entry is EFI-class,
|
||||
not floppy emulation.
|
||||
|
||||
@@ -57,7 +57,7 @@ allocated from the front) either way. The USB path has no such cap.
|
||||
## The builder
|
||||
|
||||
`tools/make-iso-image.py` follows the house rule of
|
||||
[make-fat-image.py](../tools/make-fat-image.py): pure Python 3 standard
|
||||
[make-fat-image.py](../../tools/make-fat-image.py): pure Python 3 standard
|
||||
library, no external tools (no xorriso, mkisofs, or isohybrid), with a
|
||||
`--verify` mode the `check-iso-image` step runs — it checks that the MBR
|
||||
partition and the El Torito catalog agree on where the FAT image lives and
|
||||
@@ -5,7 +5,7 @@ isolation; fault → kill the process → keep the core (`onException`; the
|
||||
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||
killed, recorded before the notice posts); and the **restart policy itself**
|
||||
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||
([device-manager.md](../device-driver-development/device-manager.md)): the device manager supervises every
|
||||
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||
@@ -14,9 +14,9 @@ more of the system moved into restartable processes (the discovery migration,
|
||||
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||
the reason the [microkernel](../vision.md) shape was chosen, and it's a *separate* goal
|
||||
from [real-time](smp.md#does-the-right-choice-depend-on-real-time-vs-resilience) —
|
||||
one that's less pervasive to build (see [vision.md](vision.md)).
|
||||
one that's less pervasive to build (see [vision.md](../vision.md)).
|
||||
|
||||
## The idea: "let it crash" + supervision
|
||||
|
||||
@@ -44,7 +44,7 @@ down. **Keeping the kernel minimal is a resilience strategy, not just an aesthet
|
||||
## The building blocks
|
||||
|
||||
1. **Address-space isolation.** A fault in one component can't corrupt another or the
|
||||
kernel. This is the [user-mode milestone](vision.md) (ring 3, per-process page
|
||||
kernel. This is the [user-mode milestone](../vision.md) (ring 3, per-process page
|
||||
tables) — the shared prerequisite for *any* of this, and it's needed regardless.
|
||||
2. **Fault detection** — how the system notices a component is dead or sick:
|
||||
- **Crash**: a CPU fault in a user process (page fault, illegal instruction) traps
|
||||
@@ -81,7 +81,7 @@ Detecting and killing is the easy half. The genuinely tricky questions are about
|
||||
- **In-flight IPC**: messages sent to the dead component, or replies its clients are
|
||||
blocked waiting for. The channel has to break cleanly and unblock the waiters with
|
||||
an error rather than hang them forever (a design constraint that reaches back into
|
||||
[ipc.md](ipc.md) — channels need a "peer died" outcome).
|
||||
[ipc.md](../device-driver-development/ipc.md) — channels need a "peer died" outcome).
|
||||
- **Clients**: how does a client discover the service it was talking to is gone and
|
||||
has been replaced? Options: capability revocation makes stale handles fail; or a
|
||||
**name server** re-binds clients to the new instance; or clients retry through a
|
||||
@@ -137,7 +137,7 @@ Honest boundaries:
|
||||
|
||||
Resilience needs **structural** features (isolation + supervision + a resource
|
||||
model); real-time needs a **pervasive** timing invariant. They're separable, and
|
||||
resilience is the lighter commitment (see [smp.md](smp.md) and [vision.md](vision.md)).
|
||||
resilience is the lighter commitment (see [smp.md](smp.md) and [vision.md](../vision.md)).
|
||||
Note the overlap, though: **preemptive scheduling** and **priorities** — already
|
||||
built — serve resilience too (you can preempt and kill a misbehaving component, and
|
||||
run the supervisor at high priority). So danos keeps the useful *mechanisms* of the
|
||||
@@ -156,10 +156,10 @@ real-time work without owing anyone a timing *guarantee*.
|
||||
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the goals this serves (learning by doing; resilience over
|
||||
- [vision.md](../vision.md) — the goals this serves (learning by doing; resilience over
|
||||
hard real-time).
|
||||
- [scheduling.md](scheduling.md) — preemption, which makes runaway components killable.
|
||||
- [ipc.md](ipc.md) — channels that need a "peer died" outcome for clean restart.
|
||||
- [ipc.md](../device-driver-development/ipc.md) — channels that need a "peer died" outcome for clean restart.
|
||||
- [interrupts.md](interrupts.md) — fault reporting that user mode turns into "kill and
|
||||
restart" instead of "halt".
|
||||
- [smp.md](smp.md) — the real-time-vs-resilience fork, in the SMP context.
|
||||
@@ -3,11 +3,11 @@
|
||||
The scheduler turns danos from a linear "boot then halt" kernel into a **running
|
||||
multitasking system**. It's **fixed-priority preemptive**: the highest-priority
|
||||
ready task always runs, and tasks at the same priority take turns. That model is
|
||||
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
||||
chosen for [real-time](../vision.md) — it's predictable (you can reason about which
|
||||
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
||||
|
||||
The scheduler proper (`system/kernel/sched.zig`) is generic; the context switch and new-task
|
||||
stack setup are architecture-specific (`system/kernel/architecture/x86_64/`, see [arch](arch.md)).
|
||||
The scheduler proper (`system/kernel/scheduler.zig`) is generic; the context switch and new-task
|
||||
stack setup are architecture-specific (`system/kernel/architecture/x86_64/`, see [architecture](architecture.md)).
|
||||
|
||||
## Tasks
|
||||
|
||||
@@ -37,7 +37,7 @@ down a return address pointing at `task_trampoline` and zeroed callee-saved slot
|
||||
`schedule()` — pick the best task and switch — runs from two places:
|
||||
|
||||
- **`yield()`** — a task voluntarily gives up the CPU.
|
||||
- **`tick()`** — the 1000 Hz [timer](device-interrupts.md) preempts the running
|
||||
- **`tick()`** — the 1000 Hz [timer](../device-driver-development/device-interrupts.md) preempts the running
|
||||
task. This is what lets a task that never yields still share the CPU.
|
||||
|
||||
The subtlety in mixing them is the **interrupt flag (IF)**. The rule: `switch_context`
|
||||
@@ -97,7 +97,7 @@ marks the task blocked with a wake deadline and switches away. On every tick the
|
||||
timer wakes any task whose deadline has passed (a bounded scan, so it stays
|
||||
deterministic), which makes it ready again; the scheduler then runs it when its
|
||||
priority comes up. `sleep` measures its deadline on the [calibrated
|
||||
clock](device-interrupts.md), so it's real time.
|
||||
clock](../device-driver-development/device-interrupts.md), so it's real time.
|
||||
|
||||
When *every* task is blocked, something still has to run — so there's an **idle
|
||||
task** at the lowest priority that just `hlt`s until the next interrupt (see
|
||||
@@ -111,7 +111,7 @@ The other form of blocking is waiting for an **event** rather than a duration. A
|
||||
the caller on it, `wake(wq)` moves the highest-priority waiter back to ready
|
||||
(preempting if it now outranks the running task). A task links into a wait queue
|
||||
through the same field the ready queues use — it's in exactly one queue at a time.
|
||||
These are the primitives locks, semaphores and [IPC](ipc.md) are built on.
|
||||
These are the primitives locks, semaphores and [IPC](../device-driver-development/ipc.md) are built on.
|
||||
|
||||
Blocking safely needs **composable critical sections**. A blanket `cli`/`sti` pair
|
||||
doesn't nest: an IPC channel that `cli`s and then calls `wait` would have `wait`'s
|
||||
@@ -124,7 +124,7 @@ caller's state.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Three tests (see [testing.md](testing.md)) prove the guarantees:
|
||||
Three tests (see [testing.md](../testing.md)) prove the guarantees:
|
||||
|
||||
- **`sched`** spawns three tasks that busy-loop *without ever yielding*. They all
|
||||
make progress — which can only happen if the timer is **preempting** between them
|
||||
@@ -136,13 +136,14 @@ Three tests (see [testing.md](testing.md)) prove the guarantees:
|
||||
- **`event`** blocks a task on a wait queue; waking it (from another task) resumes
|
||||
it, and since it's higher priority it preempts immediately.
|
||||
|
||||
## What's next (not done here)
|
||||
## What's next (partly done since)
|
||||
|
||||
- **Priority inheritance.** Once tasks block on shared resources (locks, IPC),
|
||||
danos will need it to bound priority inversion — a [real-time](vision.md)
|
||||
requirement.
|
||||
- **Task exit / a reaper.** `exit` currently leaks the task's stack; nothing frees
|
||||
finished tasks' memory yet.
|
||||
- **Per-address-space tasks.** Today all tasks share the kernel address space. User
|
||||
processes will each get their own, switching page tables (CR3) on the context
|
||||
switch.
|
||||
- **Priority inheritance** — still open. Tasks now do block on shared resources
|
||||
(IPC rendezvous, the big kernel lock), and nothing yet bounds priority
|
||||
inversion — a [real-time](../vision.md) requirement.
|
||||
- **Task exit / a reaper** — done. A dying task goes on its core's reap list in a
|
||||
`.reaping` state; the timer tick drains the list, frees the stack back to the
|
||||
heap, and recycles the task-table slot.
|
||||
- **Per-address-space tasks** — done. User processes each own an address space,
|
||||
and the context switch reloads CR3 when the target's tables differ (see
|
||||
[paging.md](paging.md)).
|
||||
@@ -0,0 +1,310 @@
|
||||
# Shared fate: whole-process death (plan)
|
||||
|
||||
**Status: implemented 2026-07-22 (branch shared-fate), M1–M4 all landed; leader
|
||||
`thread_exit` → `-EPERM` as decided. One scope addition forced by M4: the
|
||||
per-task DMA/shared-memory arena cursors moved to the per-space object (the
|
||||
`shm-mapping-ref` test could not distinguish corruption-by-remap from
|
||||
corruption-by-free while sibling threads overlapped the arena) — the same move
|
||||
the mmap/MMIO cursors made in threading M7.**
|
||||
|
||||
[threading.md](threading.md) promises that a process dies *whole* — a fault in any
|
||||
thread, or a kill, takes down every thread. The kernel doesn't do that yet: every
|
||||
death path (`exit`, `thread_exit`, a ring-3 fault, `process_kill`) tears down
|
||||
exactly one `Task`, and the address-space refcount keeps the space alive for the
|
||||
siblings — so a faulting worker orphans its threads, which keep running in the
|
||||
possibly-corrupted address space (`process.zig` `killCurrentProcess`;
|
||||
`scheduler.zig` `exitUserLocked`). This plan closes that gap the way Linux, Windows,
|
||||
and Fuchsia all did: **the process is the unit of fate; only a voluntary
|
||||
`thread_exit` is per-thread.** The supervisor contract — one exit notification,
|
||||
then `process_exit_reason` — is deliberately unchanged.
|
||||
|
||||
## The contract
|
||||
|
||||
| Event | Who dies | Reason the supervisor reads (leader's record) |
|
||||
|---|---|---|
|
||||
| CPU fault in **any** thread (recoverable vector) | the whole group | the fault class (`segmentation_fault`, …) |
|
||||
| `process_kill` on **any member id** | the whole group | `killed` |
|
||||
| `exit(code)` from **any** thread | the whole group | `exited` (0) / `aborted` (≠0) |
|
||||
| `thread_exit` from a worker | that worker only | — (per-task record: `exited`) |
|
||||
| `thread_exit` from the **leader** | nobody — refused, `-EPERM` (decided, below) | — |
|
||||
| NMI, double fault, machine check | the core halts (unchanged) | — |
|
||||
|
||||
`exit` gaining group semantics is the `exit_group` lesson from Linux: the runtime's
|
||||
main-return path calls `exit`, and a process whose main returned must not leave
|
||||
workers running. `thread_exit` (what the worker trampoline calls) keeps today's
|
||||
per-thread behavior, refcount and all.
|
||||
|
||||
**The leader-`thread_exit` rule (decided: refuse).** The syscall is reachable from
|
||||
the leader even though the runtime never does it. Options weighed: (i) **refuse
|
||||
with `-EPERM`** — cheapest and honest; the group ends only through
|
||||
`exit`/fault/kill; (ii) escalate to `exit(0)` — Linux-flavored, but silently turns
|
||||
a (buggy) library call into process death; (iii) a Linux-style zombie leader whose
|
||||
slot survives until the group ends — the most faithful, and by far the most
|
||||
machinery. **(i) chosen at sign-off**; rows and tests below follow it.
|
||||
|
||||
## Group identity: a leader id
|
||||
|
||||
Nothing on `Task` names a process today — threads are tied to their process only by
|
||||
an equal `address_space`, their name is `"thread"`, and their `supervisor` is
|
||||
whichever *task* spawned them (possibly another worker), so supervision links form a
|
||||
chain, not a group. Scanning by `address_space` is also fragile during teardown,
|
||||
because both death paths zero it.
|
||||
|
||||
So: **add `leader: u32` to `Task`** — the Linux tgid, in danos clothes.
|
||||
`spawnProcessSupervised` sets `leader = own id`; `spawnThreadSupervised` copies the
|
||||
*caller's* leader; kernel tasks keep `leader = 0`, which is never followed. The
|
||||
leader id is exactly the id `system_spawn` returned to the supervisor, so the
|
||||
outside world already speaks it. Group membership = equal `leader`, where a *live
|
||||
member* means `state` ∈ {`.ready`, `.blocked`, `.running`} — the same filter
|
||||
`taskByIdLocked` applies; `.reaping` corpses are excluded. While the field is being
|
||||
introduced, add `leader` to `ProcessDescriptor` too (the ABI is private, so this is
|
||||
cheap now and lets `process_enumerate` consumers group threads).
|
||||
|
||||
`process_kill` re-derives its authority through the leader — with the existing
|
||||
guard order preserved: a kernel task (`address_space == 0`) is `-ESRCH` *before*
|
||||
any leader resolution (the kernel test asserts exactly that). Then: resolve the
|
||||
target, follow `target.leader`, require `leader.supervisor == caller`. The kill
|
||||
capability becomes per-*process*, aimed at any member id, and the odd
|
||||
today-behavior where a worker can be individually killed by its spawning thread
|
||||
disappears. (Audited: nothing in-tree kills a worker tid or relies on
|
||||
thread-supervisor kill semantics.)
|
||||
|
||||
## The group-dying latch
|
||||
|
||||
`AddressSpaceRef` — one per space, refcounted by its member tasks, recycled with a
|
||||
full struct re-init — gains the group-death state:
|
||||
|
||||
```zig
|
||||
dying: bool = false, // set by the first trigger; never cleared
|
||||
group_reason: abi.ExitReason, // what the leader's record will say
|
||||
exit_endpoint: ?*ipc.Endpoint, // the leader's counted ref, moved here
|
||||
leader: u32,
|
||||
```
|
||||
|
||||
The latch answers three attacks the red team confirmed against a latch-free
|
||||
design:
|
||||
|
||||
- **The spawn gate.** A member already *inside* `thread_spawn` on another core
|
||||
when the fan-out runs (it passed the syscall-entry `kill_pending` check, then
|
||||
spun on the BKL) would otherwise complete the spawn after the fan-out's lock
|
||||
hold ends — a fresh, uncondemned member that escapes the kill and, worse, holds
|
||||
a space reference that keeps the group-death hook from ever firing. Fix:
|
||||
`retainAddressSpace` (equivalently `spawnUserLocked`) **refuses a dying
|
||||
space**; the in-flight `thread_spawn` fails with `-ESRCH` under the same lock
|
||||
that would have created the member.
|
||||
- **Concurrent triggers.** A second member faulting (or exiting) on another core
|
||||
while the first fan-out runs must not re-run the fan-out, double-bump
|
||||
`fault_kill_count`, or re-stamp reasons. Every kill path checks the latch first:
|
||||
already dying → skip straight to `terminateCurrentLocked`, no stamp, no count.
|
||||
First trigger wins, deterministically. `exit_reason` and `fault_kill_count`
|
||||
writes move under the BKL as part of this.
|
||||
- **Notification ownership.** The leader's `exit_endpoint` is a counted
|
||||
birth-to-death reference dropped at notify time. The stamp pass **moves** that
|
||||
reference onto the `AddressSpaceRef` and nulls `Task.exit_endpoint` in the same
|
||||
hold, so the leader's own `releaseTaskResourcesLocked` sees null (no early
|
||||
notify, no double drop); the group-death hook notifies and drops exactly once.
|
||||
|
||||
## The fan-out: `killGroupLocked`
|
||||
|
||||
One new function in `process.zig`, running under a **single BKL hold** (built from
|
||||
the `*Locked` primitives — the lock is non-recursive, and `terminateCurrentLocked`
|
||||
never returns, which forces the shape):
|
||||
|
||||
```
|
||||
killGroupLocked(leader: u32, reason: ExitReason, trigger: ?*Task)
|
||||
0. Latch: AddressSpaceRef.dying = true, stash {reason, leader,
|
||||
leader's exit_endpoint (moved)}.
|
||||
1. Stamp pass: the LEADER's exit_reason = reason — the leader's
|
||||
record is the one the supervisor can read, so it carries the
|
||||
group reason even when the trigger is a worker. The trigger
|
||||
also keeps `reason` (its own record tells the truth); every
|
||||
other live member gets .killed. All members get kill_pending.
|
||||
Stamping precedes any teardown, because recordExitLocked
|
||||
snapshots the reason first thing.
|
||||
2. Reap pass, to fixpoint: reap every member in .ready or .blocked
|
||||
via reapTaskLocked, re-reading Task.state each iteration — a
|
||||
member's teardown can wake another member (-EPEER wakes, joiner
|
||||
wakes), flipping it .blocked → .ready behind the scan cursor.
|
||||
Terminates in ≤ one pass per member: the scrub calls in
|
||||
releaseTaskResourcesLocked (abandonSenderLocked,
|
||||
removeFromWaitQueueLocked, forgetIpcClientLocked,
|
||||
killOwnedEndpointsLocked) run before destroy, so no wake path
|
||||
holds a pointer to a reaped member.
|
||||
3. Members .running on other cores stay condemned (kill_pending);
|
||||
a condemned member dies at its next syscall entry, at its own
|
||||
core's next tick while in user mode, or — once it blocks or is
|
||||
preempted — at any core's next tick reap. There is no kill IPI.
|
||||
(The entry check reads kill_pending unlocked; benign on
|
||||
x86-TSO — a missed read is caught by the next delivery point —
|
||||
but make the field atomic when touching it.)
|
||||
4. If the current task is a member (fault, exit, in-group kill):
|
||||
terminateCurrentLocked, last, because it switches away and the
|
||||
reap paths free the kernel stack being stood on.
|
||||
If the caller is outside the group (supervisor kill): return.
|
||||
```
|
||||
|
||||
The invariants this preserves, each load-bearing today:
|
||||
|
||||
- **Only `.ready`/`.blocked` tasks are reaped synchronously.** A member running on
|
||||
another core can only be condemned — it tears itself down after switching CR3
|
||||
off the dying page tables (the stack it stands on is freed later by the reap
|
||||
list), and its address-space reference protects the page tables its CR3 still
|
||||
points at. Force-destroying the space under a running sibling is the one
|
||||
unrecoverable mistake available here.
|
||||
- **The refcount decides when the space dies.** Reaping N members drops N
|
||||
references; the last drop — possibly on a condemned sibling's core, a tick
|
||||
later — destroys the space. No path forces it.
|
||||
- **`fault_kill_count` bumps once per group**, not per member (`fault-recovery`
|
||||
asserts `== 1` exactly); the latch is what enforces this under racing faults.
|
||||
- **Per-tid resource sweeps stay per-tid.** Each member's
|
||||
`releaseTaskResourcesLocked` releases what *that tid* owns — claims, GSI/MSI
|
||||
bindings, registered endpoints, handles. That keying is correct under shared
|
||||
fate (and is today's hazard: a lone worker death already yanks its claims out
|
||||
from under live siblings). A worker that *does* carry an `exit_endpoint` (the
|
||||
ABI allows it; the runtime passes `no_cap`) keeps today's per-task posting at
|
||||
its own teardown — only the leader's notification moves.
|
||||
|
||||
## When is the group dead? The notification
|
||||
|
||||
Today each task posts its own exit notification as the *last* step of its release,
|
||||
so a supervisor observes a fully-released child. For a group that guarantee must
|
||||
hold for the **whole group**: if the leader's notification fires while a condemned
|
||||
sibling still runs on another core, the device manager can respawn the driver into
|
||||
a claim conflict with a not-yet-dead sibling.
|
||||
|
||||
The clean fix falls out of the refcount: **the group is dead exactly when the
|
||||
address space is destroyed.** `releaseAddressSpace`'s last-drop path calls a new
|
||||
`group_exit_hook` (the scheduler already calls up through hooks —
|
||||
`terminate_current_hook` — precisely to keep this layering), which:
|
||||
|
||||
1. **re-stamps the leader's exit record** with the stashed `group_reason` — the
|
||||
record is written (again) at group-death time, so "the reason is recorded
|
||||
before the notification posts" stays true and a supervisor can never be
|
||||
notified and then read `-ESRCH` because the burst evicted an old record;
|
||||
2. posts the leader's exit notification (and subscriber broadcast) from the
|
||||
stashed endpoint, and drops that reference — exactly once.
|
||||
|
||||
Both `releaseAddressSpace` call sites (`exitUserLocked`, `destroyTaskLocked`) run
|
||||
under the BKL, so the hook does too; its wakes are safe at both (verified). For a
|
||||
single-threaded process the behavior is *externally indistinguishable* from
|
||||
today — the order of notify vs. destroy inverts, but both sit inside one lock
|
||||
hold, so no other core can observe the space destroyed but the notification
|
||||
unposted, or vice versa. That sentence is the correctness argument; it is also the
|
||||
first invariant to re-examine if the BKL is ever split, along with
|
||||
`killGroupLocked`'s single-hold atomicity. (Hand-built spaces that were never
|
||||
retained take the immediate-destroy path and are out of the hook's scope.)
|
||||
|
||||
Workers' `exit_subscribers` broadcasts still fire per task — the FAT server's
|
||||
dead-client sweep is keyed by tid and needs those.
|
||||
|
||||
**Signals.** `signal_bind` is per-task and the service harness binds on the main
|
||||
thread, so signals address the leader in practice; that stays. During a group
|
||||
death, `process_signal` may return `0` (accepted by a condemned member — never
|
||||
delivered, every delivery point kills first) or `-ESRCH` (member already reaped);
|
||||
init's stop sequence already tolerates both, and its timer escalation to
|
||||
`process_kill` covers the gap. `process_signal` follows `process_kill`'s
|
||||
leader re-key for consistency.
|
||||
|
||||
## The shared-memory frame hazard
|
||||
|
||||
`dropSharedMemoryReference` frees a region's physical frames when the last *handle*
|
||||
reference drops, but mappings die only with the address space. If the last handle
|
||||
lived in a torn-down member while any task still has the region mapped, that task
|
||||
holds a live mapping onto freed frames — and the red team showed this is **not**
|
||||
group-specific: a plain `thread_exit` of the handle-holding thread, or a last-ref
|
||||
drop by a task *outside* the dying group during the condemned window, hits the same
|
||||
use-after-free.
|
||||
|
||||
So the fix is a property of the **object**, not the dropper: give
|
||||
`SharedMemoryObject` a per-*mapping* reference — `shared_memory_map` (and create's
|
||||
self-map) retains; each space's destruction releases. "Last reference" then means
|
||||
*no handles and no mappings*, both hazard paths collapse into the existing
|
||||
refcount, and no group-kill special case is needed at all.
|
||||
|
||||
## Deliberately unchanged
|
||||
|
||||
- Worker `thread_exit`: per-thread, full per-tid resource sweep, refcount drop.
|
||||
- The condemned-but-running window: a member on another core can finish its
|
||||
in-flight syscall and run user code for up to a tick before dying — identical to
|
||||
today's single-task `process_kill` semantics ("prompt but asynchronous, like a
|
||||
Unix signal"). A kill IPI would shrink it; it is not part of this plan.
|
||||
- `thread_join` returns 0 for a killed thread; joiners inside a dying group are
|
||||
woken and then reaped like any member.
|
||||
- The `.reaping` state, reap lists, and stack reaper.
|
||||
|
||||
## Accepted limits (documented, not fixed here)
|
||||
|
||||
- **Notify-ring overflow**: a group death posts one subscriber badge per member
|
||||
into 8-slot rings; a >8-member group can drop badges. Group size is bounded by
|
||||
the 48-task table; today's largest production group is 2 (display) and
|
||||
thread-test already reaches 5.
|
||||
- **Exit-record ring pressure**: one 64-entry ring, one record per member — made
|
||||
harmless for the supervisor by the hook's group-death re-stamp.
|
||||
- **Enumerate shows a partial group** mid-death: reaped members vanish at once,
|
||||
condemned members linger up to a tick (audited: no in-tree consumer
|
||||
misbehaves; the `leader` field in `ProcessDescriptor` lets future consumers
|
||||
group correctly).
|
||||
- **Pre-existing reap race, not widened**: a task preempted *mid-syscall* is
|
||||
`.ready` with `in_system_call = true`, and the tick's reap loop will reap it —
|
||||
an existing hazard the fan-out inherits but must not add new instances of.
|
||||
Filed to investigate separately.
|
||||
- **Per-task DMA/shm cursors** — *fixed during M4 after all*: the
|
||||
`shm-mapping-ref` test tripped the overlap (the sibling's churn regions mapped
|
||||
over the worker's region), so both cursors moved to the `AddressSpaceRef`
|
||||
like the mmap/MMIO cursors before them. The post-implementation review then
|
||||
found the other half: the shm/DMA page-table walks and their pmm/heap calls
|
||||
ran *outside* the big kernel lock — pre-existing, but fatal once siblings
|
||||
were invited to race them (and a plausible root for the long-standing
|
||||
intermittent AP ring-3 fault at the shm base). All three paths now follow
|
||||
the mmap discipline: metadata and allocation under one hold, the map itself
|
||||
per-page under brief holds.
|
||||
- **Mapping-record slots are never recycled**: 16 per space, one per
|
||||
`shared_memory_create`/`map`, freed only at space destruction (there is no
|
||||
shm unmap). A long-lived compositor that churns surfaces will hit the cap;
|
||||
the failure is a clean refused create, and slot recycling can ride whatever
|
||||
adds `shared_memory_unmap`.
|
||||
- **Two properties lack direct tests**: the spawn gate (an in-flight
|
||||
`thread_spawn` racing the fan-out — inherently nondeterministic to arrange;
|
||||
covered by code inspection and the `-ESRCH` path) and the `process_signal`
|
||||
leader re-key (exercised only implicitly by the signals case).
|
||||
|
||||
## Milestones
|
||||
|
||||
- **M1 — the leader id.** `Task.leader` (kernel tasks: 0, never followed), set on
|
||||
both spawn paths; `leader` added to `ProcessDescriptor`; `process_kill` and
|
||||
`process_signal` re-keyed (kernel-task `-ESRCH` guard *before* leader
|
||||
resolution). No fan-out yet. Existing tests must pass untouched.
|
||||
- **M2 — the latch + fan-out.** `AddressSpaceRef.dying` + stash;
|
||||
`retainAddressSpace` refuses dying spaces; `killGroupLocked`; wire the fault
|
||||
path, `exit`, and `process_kill` into it; leader `thread_exit` → `-EPERM`;
|
||||
`exit_reason`/`fault_kill_count` writes under the BKL; `kill_pending` made
|
||||
atomic. Group notification via the `group_exit_hook` re-stamp + post.
|
||||
- **M3 — shared-memory mapping refs.** `SharedMemoryObject` counts mappings;
|
||||
space destruction releases them; frames free only at zero handles *and* zero
|
||||
mappings.
|
||||
- **M4 — tests + docs.** New `-Dtest-case`s (all `smp: 4` where cross-core
|
||||
matters), driving `thread-test` with new argv modes:
|
||||
- `thread-fault-group`: a worker faults; assert both tasks gone from
|
||||
`enumerate`, `fault_kill_count == 1`, `process_exit_reason(leader) ==
|
||||
segmentation_fault`, address-space and stack-bytes counters return to base.
|
||||
- `kill-threaded-group`: `process_kill(leader)` with a worker spinning on
|
||||
another core; assert the worker dies by the deferred path, exactly one exit
|
||||
badge, delivered only after both members are dead, and
|
||||
`process_exit_reason(leader) == .killed`.
|
||||
- `kill-via-worker-tid`: `process_kill(worker)` kills the whole group;
|
||||
`-EPERM` for a non-supervisor aiming at the worker.
|
||||
- `racing-triggers`: two members fault/exit simultaneously on different cores;
|
||||
assert a deterministic leader reason and `fault_kill_count == 1`.
|
||||
- `exit-group`: a *worker* calls `exit(3)`; group dies, leader reason
|
||||
`.aborted`.
|
||||
- `leader-thread-exit`: asserts the chosen rule (`-EPERM`, workers unaffected).
|
||||
- `thread-exit-solo`: regression — worker `thread_exit` still leaves siblings
|
||||
running.
|
||||
- `group-claim-release`: a member claims a device; assert the claim is free and
|
||||
the leader notification arrives only after every member is dead.
|
||||
- `shm-mapping-ref`: last handle dropped by a dying thread; sibling's mapping
|
||||
stays valid until space death (M3 regression).
|
||||
Then update [threading.md](threading.md) (the shared-fate gap note),
|
||||
[process-lifecycle.md](process-lifecycle.md),
|
||||
[process-management.md](process-management.md), and
|
||||
[ipc.md](../device-driver-development/ipc.md)/[drivers.md](../device-driver-development/drivers.md) mentions.
|
||||
@@ -1,10 +1,11 @@
|
||||
# SMP: multiple cores, the microkernel way
|
||||
|
||||
A design/research note, not built yet. danos runs on **one core** today (see
|
||||
[scheduling.md](scheduling.md)); this maps how microkernels — especially the L4
|
||||
family and seL4 — handle **symmetric multiprocessing (SMP)**, so the eventual port
|
||||
has a plan and a reading list. It also flags where those choices depend on whether
|
||||
danos is chasing **real-time** or **resilience** (see the note at the end).
|
||||
A design/research note that predates the build — danos now runs on **multiple
|
||||
cores** by default (see [Implementation status](#implementation-status) and
|
||||
[scheduling.md](scheduling.md)). This maps how microkernels — especially the L4
|
||||
family and seL4 — handle **symmetric multiprocessing (SMP)**, the plan and
|
||||
reading list the port followed. It also flags where those choices depend on
|
||||
whether danos is chasing **real-time** or **resilience** (see the note at the end).
|
||||
|
||||
## First, the vocabulary
|
||||
|
||||
@@ -111,7 +112,7 @@ Yes — and this is the branch that matters for danos right now.
|
||||
|
||||
These pull in different directions, so **picking the primary goal comes before
|
||||
picking the SMP design.** (danos's founding assumption was real-time; that's under
|
||||
active reconsideration in favour of resilience — see [vision.md](vision.md).)
|
||||
active reconsideration in favour of resilience — see [vision.md](../vision.md).)
|
||||
|
||||
## What this would mean for danos
|
||||
|
||||
@@ -158,13 +159,17 @@ next lands.
|
||||
`ipc.zig` run every critical section under it. Uncontended on one core, so behaviour
|
||||
is identical to the old interrupt-flag model.
|
||||
- **Per-CPU state** — a `PerCpu` struct (running task, idle task, APIC id) per core,
|
||||
its pointer kept in the x86 **GS base** (`IA32_GS_BASE`; no `swapgs`, since there's
|
||||
no user mode yet). The old global `current` is now `thisCpu().current`. The ready
|
||||
its pointer kept in the x86 **GS base** (`IA32_GS_BASE`). No `swapgs` was needed at
|
||||
the time — there was no user mode yet; with ring 3 in place, every ring transition
|
||||
now swaps it against the user's own GS base under the `swapgs` discipline (see
|
||||
`system/kernel/architecture/x86_64/per-cpu.zig`), and each AP enables the fast
|
||||
system-call path (`initSystemCall`) for itself at bring-up. The old global
|
||||
`current` is now `thisCpu().current`. The ready
|
||||
queues stay **global** under the lock — work-conserving, so any idle core will pull
|
||||
the highest-priority ready task; per-core queues are a later optimisation.
|
||||
- **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||
- **AP wake to long mode** — `architecture.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
|
||||
real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
|
||||
real mode at a low page and runs the [trampoline](../../system/kernel/architecture/x86_64/trampoline.s)
|
||||
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
|
||||
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
|
||||
all four cores report `online`.
|
||||
@@ -203,8 +208,9 @@ next lands.
|
||||
[scheduling.md](scheduling.md#affinity-pinning-a-task-to-a-core)). The `affinity`
|
||||
test confirms a pinned task never migrates. This is the mechanism the fault-on-AP
|
||||
test rides on, and the *explicit-affinity* real-time-predictable model.
|
||||
- **Right-sized footprint** — the per-CPU ceiling (`system.max_cpus`, one constant
|
||||
shared by discovery, the scheduler, and the per-core GDT/TSS) is generous (128), but
|
||||
- **Right-sized footprint** — the per-CPU ceiling (`parameters.maximum_cpus`, one
|
||||
constant shared by discovery, the scheduler, and the per-core GDT/TSS) is generous
|
||||
(128), but
|
||||
the *large* per-core resources — the kernel and IST (double-fault) stacks — are
|
||||
**heap-allocated at bring-up**, only for cores that actually come online. Only the
|
||||
BSP's IST stack is static, because it must exist before the frame allocator does.
|
||||
@@ -217,7 +223,7 @@ next lands.
|
||||
life, but kept **inert between wakes**: zeroed and non-executable, armed (blob
|
||||
copied in, page made executable) only for the moment a core is actually climbing,
|
||||
then disarmed again. So there's never a dormant executable page, and a core can be
|
||||
(re)woken at any time — `arch.startSecondary` is one self-contained attempt (arm →
|
||||
(re)woken at any time — `architecture.startSecondary` is one self-contained attempt (arm →
|
||||
INIT–SIPI–SIPI → disarm), and its `INIT` resets a wedged core, so retrying just
|
||||
works. Boot retries a non-responding core up to three times; the same primitive is
|
||||
the groundwork a future **power manager** would drive to bring cores up (and,
|
||||
@@ -264,6 +270,6 @@ next lands.
|
||||
|
||||
- [scheduling.md](scheduling.md) — the single-core scheduler SMP would extend.
|
||||
- [discovery.md](discovery.md) — enumerating cores is a device-discovery problem.
|
||||
- [ipc.md](ipc.md) — the message passing cross-core coordination rides on.
|
||||
- [vision.md](vision.md) — the goals question (real-time vs resilience) this note
|
||||
- [ipc.md](../device-driver-development/ipc.md) — the message passing cross-core coordination rides on.
|
||||
- [vision.md](../vision.md) — the goals question (real-time vs resilience) this note
|
||||
keeps bumping into.
|
||||
@@ -1,15 +1,18 @@
|
||||
# System Calls
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
|
||||
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||
> **Status:** danos has real user processes. User programs enter the kernel
|
||||
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
|
||||
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||
> dispatcher); the `int 0x80` gate is kept alongside as a minimal test path.
|
||||
> The live table is `system/abi.zig` (private, renumberable — see
|
||||
> [vdso.md](vdso.md) for the public boundary): process lifecycle + threads,
|
||||
> memory (mmap/dma/shared-memory), synchronous + async IPC with capability passing,
|
||||
> device access, time, the tagged-log diagnostics (`debug_write` with a level,
|
||||
> `klog_read`/`klog_status`), and filesystem NAMING (`fs_resolve`/`fs_node`/
|
||||
> `fs_mount`/`fs_unmount` — the kernel VFS root routes paths and serves the
|
||||
> read-only /system initrd mount; file DATA stays with userspace filesystem
|
||||
> servers over the vfs-protocol, docs/vfs-protocol.md).
|
||||
|
||||
## The Mechanism of a Syscall
|
||||
|
||||
@@ -48,7 +51,7 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
||||
3. **`Yield()`/`Thread_Ctrl()`**
|
||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](../device-driver-development/input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
|
||||
* * * * *
|
||||
|
||||
@@ -89,4 +92,4 @@ Managing the Payload Challenge
|
||||
Because it is a microkernel, performance lives or dies by how fast your`IPC_Call`can move data from Client to Server. You have two minimal choices for handling the`message_buffer`pointer:[[1](https://anazimzada2020.medium.com/microkernel-architectural-pattern-5e4e9184170e)]
|
||||
|
||||
- **The Copy Method (Simplest to start):**Your kernel pauses the client, reads the data from the client's memory space, switches page tables to the server, and copies the data into the server's buffer.
|
||||
- **The Shared Memory Method (Fastest):**The kernel sets up a temporary, shared virtual memory page between the client and server. The client writes to it, calls`syscall`/`svc`, and the server reads it instantly without the kernel copying any bytes
|
||||
- **The Shared Memory Method (Fastest):**The kernel sets up a temporary, shared virtual memory page between the client and server. The client writes to it, calls`syscall`/`svc`, and the server reads it instantly without the kernel copying any bytes
|
||||
@@ -0,0 +1,131 @@
|
||||
# system.img — the boot capsule
|
||||
|
||||
## What it is
|
||||
|
||||
`boot/system.img` is the **boot capsule**: every bundled user binary — init, the
|
||||
services, the drivers, the test programs — packed into **one file** on the boot
|
||||
volume. It is not a filesystem image and it is not compressed; it is exactly the
|
||||
kernel's **initial-ramdisk wire format** (`system/initial-ramdisk.zig`, format
|
||||
v2), written to disk ahead of time. The EFI loader reads it in a single
|
||||
sequential pass and hands the bytes to the kernel unmodified.
|
||||
|
||||
The capsule is a *performance artifact*, not a source of truth. The boot
|
||||
volume's `/system` and `/test` file trees remain the canonical layout (see
|
||||
[danos-file-system-hierarchy-FSH.md](../file-system-development/danos-file-system-hierarchy-FSH.md));
|
||||
the capsule is a pre-baked snapshot of the same binaries, derived from the same
|
||||
build graph, so the running system is identical whether the loader read the
|
||||
capsule or walked the tree.
|
||||
|
||||
## Why it exists
|
||||
|
||||
Firmware file I/O has exactly one fast shape: **one open + one sequential
|
||||
read**. Everything else is a lottery. Loading the system per-file — dozens of
|
||||
opens, seeks, and short reads through the firmware's FAT driver — measured
|
||||
**minutes** on real hardware, against milliseconds in QEMU/OVMF. Packing the
|
||||
binaries into a single file turns the whole of user space into the shape
|
||||
firmware is good at.
|
||||
|
||||
Because the capsule already *is* the ramdisk wire format, the loader doesn't
|
||||
even repack it: `loadCapsule` (`boot/efi.zig`) validates the magic and passes
|
||||
the buffer straight through as `BootInformation.initial_ramdisk_base`/`len`.
|
||||
|
||||
## The format
|
||||
|
||||
The container is deliberately trivial — danos owns both producer and consumer,
|
||||
so it need be no fancier. Little-endian throughout:
|
||||
|
||||
```
|
||||
Header magic: u32 = "DNR2" (0x32524E44), count: u32
|
||||
Entry × count name: [64]u8 (NUL-padded FHS path), offset: u64, len: u64
|
||||
blobs... each entry's file bytes, at its offset within the image
|
||||
```
|
||||
|
||||
- **Names are full FHS paths** (`/system/services/init`), not basenames — that
|
||||
is what "v2" means. The 64-byte capacity matches `abi.maximum_process_name`,
|
||||
so a task named after its binary path is never truncated. Paths longer than
|
||||
63 bytes are a build error (`pack-system-image.py` rejects them).
|
||||
- **The v1 magic (`"DNRD"`, basename entries) is rejected**, not tolerated: a
|
||||
stale image should fail loudly at `Reader.init`, not misparse names.
|
||||
- `initial_ramdisk.Reader` is the one validated view over the bytes — magic
|
||||
check, table bounds, per-blob bounds — used by the kernel and shared with the
|
||||
loader. `Reader.find` resolves a binary by exact path first, then by unique
|
||||
basename, ASCII case-insensitively (the entries come from a FAT volume, whose
|
||||
name lookups are case-insensitive by definition).
|
||||
|
||||
## How it is built
|
||||
|
||||
`build.zig` maintains one `bundled` list — every user binary and its FHS home.
|
||||
Three artifacts are derived from that same list, in the same build graph, so
|
||||
they cannot drift apart:
|
||||
|
||||
1. **The tree**: each binary installed at its FHS path (`zig-out/system/...`
|
||||
and `zig-out/test/...`, mirrored onto the FAT boot volume by
|
||||
`tools/make-fat-image.py`).
|
||||
2. **The manifest** (`system/manifest`): the FHS path of every bundled binary,
|
||||
one per line — the loader's per-file fallback input.
|
||||
3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into
|
||||
the v2 container, installed at `zig-out/boot/system.img` and placed on the
|
||||
boot volume at `boot/system.img`.
|
||||
|
||||
Note what the capsule does *not* contain: the kernel (`system/kernel` is loaded
|
||||
separately by `loadKernel`, as an ELF) and the EFI loader itself. It is user
|
||||
space only.
|
||||
|
||||
## How it is loaded
|
||||
|
||||
`loadSystemTree` (`boot/efi.zig`) tries three strategies, most portable first —
|
||||
the running system cannot tell which one ran, because all three produce the
|
||||
same in-RAM ramdisk image:
|
||||
|
||||
1. **The capsule** — open `boot\system.img`, read it whole, check the magic,
|
||||
hand it over as-is. The normal path on any build-produced volume.
|
||||
2. **The manifest** — read `system\manifest` and open each listed path *by
|
||||
name*. FAT name lookup is case-insensitive and firmware-portable, unlike
|
||||
directory enumeration. The loader assembles the v2 image in RAM itself.
|
||||
3. **The tree walk** — enumerate `/system` and `/test` recursively (`/test`
|
||||
is optional: a stick without fixtures still boots). Last resort for
|
||||
hand-assembled sticks with neither file: some firmware FAT drivers return
|
||||
bare 8.3 names uppercase from enumeration, which is why this is the
|
||||
fallback and not the primary path.
|
||||
|
||||
All three are best-effort: a **kernel-only volume still boots** — the kernel
|
||||
just has no user binaries to spawn and reports the absence.
|
||||
|
||||
One operational consequence of the ordering: the capsule *shadows* the tree.
|
||||
If you hand-edit binaries on a stick that also carries a `boot/system.img`,
|
||||
your edits are invisible — the loader boots the capsule's snapshot. Delete
|
||||
`boot/system.img` from the volume to force the manifest/tree path.
|
||||
|
||||
## What the kernel does with it
|
||||
|
||||
The loader records the image's physical base and length in `BootInformation`;
|
||||
the kernel (`kernel.zig`) then publishes the same bytes twice, to two
|
||||
consumers:
|
||||
|
||||
- **The process layer** (`process.zig`): `system_spawn` looks binaries up in
|
||||
the ramdisk via `Reader.find` — exact FHS path, or unique basename for
|
||||
pre-path callers — and loads them as fresh ring-3 processes. The stored path
|
||||
becomes the task's name.
|
||||
- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as
|
||||
kernel-backed, read-only mounts — one per top-level tree named by the entry
|
||||
paths, so `/system` and, when the fixtures are bundled, `/test`. Directory
|
||||
nodes are derived from the entry paths (the unique parents), so the trees
|
||||
are listable and their files readable over the normal VFS protocol — the
|
||||
FHS boot tree every process sees comes straight out of the capsule bytes.
|
||||
|
||||
The image is never copied after the handoff and never mutated: the initrd is
|
||||
immutable, which is what makes the VFS's node serving lock-free.
|
||||
|
||||
## What it is not
|
||||
|
||||
- **Not `danos-usb.img`.** That is the 64 MiB FAT32 *boot volume* built by
|
||||
`tools/make-fat-image.py` — the thing a machine actually boots, which
|
||||
*contains* `boot/system.img` alongside the loader, kernel, manifest, and
|
||||
tree. See [efi.md](efi.md) and [release-iso.md](release-iso.md).
|
||||
- **Not a mountable filesystem.** No FAT, no block device, no driver — just a
|
||||
header, a table, and concatenated blobs, parsed by ~90 lines of
|
||||
`initial-ramdisk.zig`.
|
||||
- **Not required.** It is the fast path, with two slower equivalents behind
|
||||
it.
|
||||
- **Not a place where state lives.** It is regenerated on every build from the
|
||||
bundled binaries; nothing writes to it, at build time or runtime.
|
||||
@@ -56,12 +56,12 @@ danos's two binaries default to different conventions:
|
||||
Microsoft x64 (first argument → RCX).
|
||||
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
||||
|
||||
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
||||
When the loader jumps to the kernel passing the `BootInformation` pointer, both sides have
|
||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||
garbage. So both sides reference the same `boot_handoff.kernel_abi` (SysV): the loader's
|
||||
function-pointer type and the kernel's `_start` both carry
|
||||
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
`callconv(boot_handoff.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||
the handoff it governs.
|
||||
|
||||
@@ -90,9 +90,9 @@ process) instead of silently corrupting the image
|
||||
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||
(`library/runtime/start.zig`) hands the block to `rt_start`, which builds a
|
||||
`runtime.process.Init` from it and passes that to the program's `main`
|
||||
(`pub fn main(init: runtime.process.Init)`; a parameterless `main()` is also
|
||||
(`library/kernel/start.zig`) hands the block to `rt_start`, which builds a
|
||||
`process.Init` from it and passes that to the program's `main`
|
||||
(`pub fn main(init: process.Init)`; a parameterless `main()` is also
|
||||
accepted). A C runtime's `crt0` would walk
|
||||
the identical layout unmodified — that's the compatibility being bought. The
|
||||
`args` test proves the round trip.
|
||||
@@ -116,7 +116,7 @@ has to speak that same convention at the point of the call.
|
||||
|
||||
## A note on other architectures
|
||||
|
||||
This is x86-64-specific. An AArch64 port ([arch.md](arch.md)) has its own calling
|
||||
This is x86-64-specific. An AArch64 port ([architecture.md](architecture.md)) has its own calling
|
||||
convention (arguments in X0–X7, and so on) — a different ABI entirely. `kernel_abi`
|
||||
would be set per-architecture, but the *principle* is the same: the loader/entry
|
||||
boundary and the kernel must agree on how arguments are passed.
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
The ordered, checkpointable build-out for [threading.md](threading.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-v2-plan.md](display-v2-plan.md). Read threading.md first for the *why*.
|
||||
[display-v2-plan.md](../device-driver-development/display-v2-plan.md). Read threading.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
@@ -13,18 +13,18 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
`single_threaded = false`.
|
||||
- **Blocking is futex-backed, never spin-backed** — waiters park in the kernel so an
|
||||
idle core still halts ([halting.md](halting.md)).
|
||||
- **New syscalls are private**: extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shm_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
- **New syscalls are private**: extend [abi.zig](../../system/abi.zig) `SystemCall` after
|
||||
`shared_memory_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
`futex_wait = 40`, `futex_wake = 41`) + a `library/runtime` wrapper; user code never names a number.
|
||||
- **Restart granularity stays the process** — a faulting thread kills its process; the
|
||||
supervisor restarts the process, which respawns its threads.
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../system/abi.zig)
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../../system/abi.zig)
|
||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||
exercise and register a `ServiceId` if they must be looked up.
|
||||
|
||||
@@ -58,7 +58,7 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
5) so progress is backed up remotely; **do not push `main`** — merging Phase 2 into
|
||||
`main` stays a human step.
|
||||
3. **Implement** every unchecked item in that milestone, including adding its
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../test/qemu_test.py) (with
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../../test/qemu_test.py) (with
|
||||
`smp: true` / a `mem` bump where noted) so the gate is runnable.
|
||||
4. **Run the gate**: `python3 test/qemu_test.py <case>`, then the full **guardrail
|
||||
set**, then `zig build` (clean) and `zig build test` (green).
|
||||
@@ -67,7 +67,7 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
the whole guardrail set passes, `zig build` is clean, and host tests are green.
|
||||
→ tick this milestone's boxes **and** its `**Gate:**`-referenced case, `git commit`
|
||||
(`threads(M<n>): <summary>`, no `Co-Authored-By` trailer per
|
||||
[coding-standards.md](coding-standards.md)), then **`git push` the working branch to
|
||||
[coding-standards.md](../coding-standards.md)), then **`git push` the working branch to
|
||||
`origin`** (use `-u` on the first push to set upstream). Continue to the next
|
||||
milestone in the same iteration if budget remains; otherwise let the loop re-fire.
|
||||
- **Red** = anything above fails. Diagnose from the captured serial log
|
||||
@@ -111,8 +111,8 @@ task's exit; make destruction happen on the **last** exit.
|
||||
- [x] A refcount keyed by the address-space root, held in `scheduler.zig`
|
||||
(`address_space_refs`): `retainAddressSpace` takes a reference in `spawnUserLocked` (on the
|
||||
success path, after the slot + stack are secured), all under the big kernel lock.
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig):
|
||||
`exitUserLocked` and `destroyTaskLocked`) call `releaseAspace`, which decrements
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../../system/kernel/scheduler.zig):
|
||||
`exitUserLocked` and `destroyTaskLocked`) call `releaseAddressSpace`, which decrements
|
||||
and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test
|
||||
spaces) is destroyed directly, preserving prior behaviour.
|
||||
- [x] `-Dtest-case=address-space-refcount`: spawn and reap several ring-3 processes in sequence
|
||||
@@ -133,10 +133,11 @@ full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `sm
|
||||
Spawn only — no join yet. Prove a second task executes in the **caller's** address
|
||||
space and exits cleanly.
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
process.zig; `thread_spawn` calls `scheduler.spawnThread` (shares the caller's
|
||||
- [x] [abi.zig](../../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
process.zig; `thread_spawn` calls `scheduler.spawnThread` (today, after M3, the
|
||||
handler goes `spawnThreadSupervised` → `scheduler.spawnUserLocked`; shares the caller's
|
||||
address space, `retainAddressSpace`); `thread_exit` ends the task like a process `exit(0)`
|
||||
(`terminateCurrent` → `releaseAspace`). The closure pointer is delivered in the new
|
||||
(`terminateCurrent` → `releaseAddressSpace`). The closure pointer is delivered in the new
|
||||
thread's **rdi** via a new `jump_to_user_arg` asm path (`t.user_arg`, 0 for a
|
||||
process) — no naked runtime asm.
|
||||
- [x] `library/runtime/thread.zig` (barrel-exported as `runtime.Thread`): `spawn` maps a
|
||||
@@ -194,7 +195,7 @@ plus `address-space-refcount`/`thread-spawn`; `zig build` clean, `zig build test
|
||||
|
||||
## M4 — Futex: the one blocking primitive ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
- [x] [abi.zig](../../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
`.blocked` task tagged with `Task.futex_addr` (no queue linkage);
|
||||
`futex_wait(addr, expected, timeout_ns)` reads the user word under the big lock,
|
||||
parks iff `*addr == expected`, and returns on wake or timeout; `futex_wake(addr,
|
||||
@@ -260,7 +261,7 @@ green.
|
||||
restores it.
|
||||
- [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same
|
||||
`Futex`/`Mutex`/`Condition` when wanted.
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../test/qemu_test.py)
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../../test/qemu_test.py)
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`); threading.md + docs/README.md
|
||||
status updated to **built**; the worked example is threading.md's win-condition.
|
||||
- [x] `-Dtest-case=thread-id` (`smp: 4`): two workers read `getCurrentId`; the main
|
||||
@@ -379,8 +380,9 @@ to the std shape and drop M3's per-thread exit endpoint:
|
||||
it enters the kernel to exit — so waking at *exit* time (not reap time) is safe, and
|
||||
no reaper/address-space juggling or user-memory write is needed. This is equally
|
||||
std-shaped (like `pthread_join`) and much simpler/safer than the planned
|
||||
reaper-written completion word. `thread_spawn` no longer takes an exit endpoint (the
|
||||
runtime passes `no_cap`); the per-thread IPC endpoint is gone.
|
||||
reaper-written completion word. The runtime no longer passes `thread_spawn` an exit
|
||||
endpoint (it passes `no_cap`; the kernel's 4th `exit_endpoint` arg remains and is
|
||||
still honored); the runtime's per-thread IPC endpoint is gone.
|
||||
- [x] `thread-join` passes on the new path, and its join mode now runs **40 spawn+join
|
||||
cycles** — under the old per-thread-endpoint scheme those leaked handles would
|
||||
exhaust the 16-slot handle table; here they all succeed, proving join is endpoint-free.
|
||||
@@ -403,7 +405,7 @@ guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-re
|
||||
### M10 — Per-thread TLS: the thread-pointer mechanism ✅
|
||||
|
||||
Give each thread its own thread pointer and private TLS storage — the foundation
|
||||
self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on.
|
||||
self-hosting Zig ([zig-self-hosting.md](../zig-self-hosting.md)) will build `threadlocal` on.
|
||||
|
||||
- [x] **Kernel** stores `thread_pointer` on `Task` and restores it on every context switch
|
||||
**only when it changes** (the same conditional-load discipline as CR3;
|
||||
@@ -457,7 +459,7 @@ clean.
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a
|
||||
physical-address key so two processes share a futex through an [shm](display-v2.md)
|
||||
physical-address key so two processes share a futex through a [shared-memory](../device-driver-development/display-v2.md)
|
||||
region. Not needed for intra-process threads.
|
||||
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||
@@ -465,5 +467,5 @@ clean.
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **A `pthread`/POSIX surface** — the API is `std.Thread`-shaped Zig, nothing more.
|
||||
- **A real `std.Thread` backend** — arrives with self-hosting
|
||||
([zig-self-hosting.md](zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
([zig-self-hosting.md](../zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
swaps the impl under `runtime.Thread`, not the call sites.
|
||||
@@ -1,8 +1,8 @@
|
||||
# Threading: `runtime.Thread`, a std-shaped API over a private thread ABI
|
||||
# Threading: `Thread`, a std-shaped API over a private thread ABI
|
||||
|
||||
A note on danos **threads** — several tasks sharing one address space — provided by a
|
||||
`runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M11, see
|
||||
`Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../../library/kernel). **Built** (M1–M11, see
|
||||
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core parallelism,
|
||||
a futex, `Mutex`/`Condition`/`Semaphore`/`RwLock`/`WaitGroup`, `getCurrentId`/`currentCore`,
|
||||
per-thread thread-pointer TLS, thread-safe allocation, and a task reaper that reclaims dead
|
||||
@@ -17,13 +17,13 @@ treat upstream shapes as "0.16.x."
|
||||
A danos service can write
|
||||
|
||||
```zig
|
||||
const t = try runtime.Thread.spawn(.{}, worker, .{ctx});
|
||||
const t = try Thread.spawn(.{}, worker, .{ctx});
|
||||
// ... do other work concurrently ...
|
||||
t.join();
|
||||
```
|
||||
|
||||
and get real parallelism across cores — with `runtime.Thread.Mutex`,
|
||||
`runtime.Thread.Condition`, and `runtime.Thread.Semaphore` available for
|
||||
and get real parallelism across cores — with `Thread.Mutex`,
|
||||
`Thread.Condition`, and `Thread.Semaphore` available for
|
||||
coordination — **without any code path reaching the kernel except through the
|
||||
runtime**. The call sites read exactly like `std.Thread`, so the day danos becomes a
|
||||
real Zig target (see [self-hosting](#the-self-hosting-endgame)) we swap the
|
||||
@@ -31,18 +31,18 @@ implementation underneath, not the API above.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **We build `runtime.Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
- **We build `Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
features*; the implementation underneath is danos-native. See
|
||||
[Why not literal std.Thread](#why-not-literal-stdthread).
|
||||
- **Threads are a narrow, opt-in capability — not the default concurrency tool.** The
|
||||
default for resilience stays **process + IPC** ([resilience.md](resilience.md),
|
||||
[ipc.md](ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
[ipc.md](../device-driver-development/ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
- **Blocking synchronization is futex-backed, never spin-backed.** Waiters sleep in
|
||||
the kernel so an idle core still halts ([halting.md](halting.md)).
|
||||
- **Per-binary opt-in to multi-threaded codegen.** Only a service that asks for
|
||||
threads is built `single_threaded = false`; the rest stay lean and single-threaded.
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/runtime` wrappers, exactly like
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/kernel` wrappers, exactly like
|
||||
every other danos syscall ([syscall.md](syscall.md)) — numbers stay renumberable.
|
||||
|
||||
## Why not literal `std.Thread`
|
||||
@@ -56,12 +56,12 @@ runtime — rebuilt in lockstep — knows the mapping.
|
||||
`std.Thread` is incompatible with that invariant on two counts:
|
||||
|
||||
1. **It selects its backend from `builtin.os.tag`, and issues syscalls directly.**
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../build.zig)), for which
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../../build.zig)), for which
|
||||
`std.Thread` resolves to an unsupported stub that `@compileError`s. Adding a real
|
||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||
cost that buys nothing the native type doesn't.
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../build.zig)
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../../build.zig)
|
||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||
single-threaded. Threads need this flipped per binary regardless.
|
||||
|
||||
@@ -81,9 +81,11 @@ address space. Threads deliberately remove that boundary *within* a process:
|
||||
|
||||
- Threads share one address space, so one thread's stray write corrupts them all —
|
||||
there is no isolation **between** threads.
|
||||
- Threads share fate: a fault in any thread, or a "kill the process" decision, takes
|
||||
down **all** of them. Restartability lives at the process level, not the thread
|
||||
level.
|
||||
- Threads share fate — by contract: a fault in any thread, or a "kill the process"
|
||||
decision, takes down **all** of them, so restartability lives at the process level,
|
||||
not the thread level. (The kernel does not yet enforce this fan-out — see the
|
||||
Lifecycle note under
|
||||
[Interaction with the rest of the kernel](#interaction-with-the-rest-of-the-kernel).)
|
||||
- Shared mutable state reintroduces data races — the failure class the
|
||||
isolate-and-message model was chosen to avoid.
|
||||
|
||||
@@ -97,22 +99,21 @@ processes. The isolation boundary stays at process granularity.
|
||||
|
||||
## The API surface (mirrors `std.Thread`)
|
||||
|
||||
Lives in `library/runtime/thread.zig`, re-exported as `runtime.Thread`.
|
||||
Lives in `library/kernel/thread.zig`, re-exported as `Thread`.
|
||||
|
||||
```zig
|
||||
pub const Thread = struct {
|
||||
pub const Id = u32; // the kernel task id
|
||||
pub const SpawnConfig = struct {
|
||||
stack_size: usize = default_stack_size,
|
||||
allocator: ?std.mem.Allocator = null, // for the closure + stack bookkeeping
|
||||
stack_size: usize = default_stack_size, // no allocator: the closure lives at the top of the thread's own stack
|
||||
};
|
||||
pub const SpawnError = error{ OutOfMemory, ThreadQuotaExceeded, SystemResources };
|
||||
pub const SpawnError = error{SystemResources};
|
||||
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread;
|
||||
pub fn join(self: Thread) void; // block until the thread ends, reclaim its stack
|
||||
pub fn detach(self: Thread) void; // give up the right to join; kernel reclaims on exit
|
||||
pub fn detach(self: Thread) void; // give up the right to join; stack reclaimed at process exit
|
||||
pub fn getCurrentId() Id;
|
||||
pub fn yield() void; // -> existing `yield` syscall
|
||||
pub fn currentCore() Id; // danos extension: the calling core's dense index
|
||||
|
||||
pub const Mutex = struct { pub fn lock(*Mutex) void; pub fn tryLock(*Mutex) bool; pub fn unlock(*Mutex) void; };
|
||||
pub const Condition = struct { pub fn wait(*Condition, *Mutex) void; pub fn timedWait(*Condition, *Mutex, u64) error{Timeout}!void; pub fn signal(*Condition) void; pub fn broadcast(*Condition) void; };
|
||||
@@ -127,18 +128,21 @@ Deviations from `std.Thread`, called out honestly:
|
||||
- **The thread function's return value is discarded** (as `std.Thread.join` returns
|
||||
`void`). Return data through shared state or a `Semaphore`/`Condition`, not the
|
||||
return.
|
||||
- `getCpuCount()` maps to the existing SMP core count ([smp.md](smp.md)); a service
|
||||
rarely needs it.
|
||||
- No `getCpuCount()` (a service rarely needs it) and no `Thread.yield()` — `yield`
|
||||
lives in the `process` module. Instead `currentCore()` exposes the calling core's dense
|
||||
index ([smp.md](smp.md)), used to observe genuine cross-core parallelism.
|
||||
|
||||
## Kernel primitives (new private syscalls)
|
||||
|
||||
Four new entries extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shm_physical = 36`, each with a `library/runtime` wrapper:
|
||||
Five core entries extend [abi.zig](../../system/abi.zig) `SystemCall` after
|
||||
`shared_memory_physical = 36` (plus small helpers `current_core`, `thread_self`, and
|
||||
`set_thread_pointer`), each with a `library/kernel` wrapper:
|
||||
|
||||
| Syscall | Signature | Purpose |
|
||||
|---|---|---|
|
||||
| `thread_spawn` | `(entry, stack_top, arg) -> tid` | create a task sharing the **caller's** address space |
|
||||
| `thread_exit` | `(stack_base, stack_len)` | end the calling thread; hand back its stack range for reclaim |
|
||||
| `thread_spawn` | `(entry, stack_top, arg, exit_endpoint) -> tid` | create a task sharing the **caller's** address space; the runtime passes `no_cap` for `exit_endpoint` (join is a syscall, not an endpoint) |
|
||||
| `thread_exit` | `()` | end the calling thread; its stack is reclaimed by the joiner's `munmap`, not the kernel |
|
||||
| `thread_join` | `(tid) -> 0` | block until the thread with id `tid` has exited |
|
||||
| `futex_wait` | `(addr, expected, timeout_ns) -> status` | block if `*addr == expected`, until woken or timeout |
|
||||
| `futex_wake` | `(addr, count) -> woken` | wake up to `count` waiters on `addr` |
|
||||
|
||||
@@ -148,56 +152,56 @@ Plus one invariant change with no new syscall: **address-space reference countin
|
||||
|
||||
### Address-space reference counting
|
||||
|
||||
Today an address space is 1:1 with a task: `spawnUserLocked` records `address_space` on the
|
||||
Task, and teardown does `destroyAddressSpace(t.address_space)` when **any** user task exits
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
Before this work an address space was 1:1 with a task: `spawnUserLocked` records
|
||||
`address_space` on the Task (as it still does), and teardown did
|
||||
`destroyAddressSpace(t.address_space)` when **any** user task exited
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
one `address_space`, so the first to exit would rip the address space out from under its
|
||||
siblings.
|
||||
|
||||
Fix: a small refcount keyed by the address-space root (`createAddressSpace` in
|
||||
[process.zig](../system/kernel/process.zig) sets it to 1). `thread_spawn` increments
|
||||
it; task teardown decrements and only calls `destroyAddressSpace` at **zero**. All of
|
||||
Fix: a small refcount keyed by the address-space root, kept in
|
||||
[scheduler.zig](../../system/kernel/scheduler.zig): `retainAddressSpace` takes a
|
||||
reference for every user task `spawnUserLocked` starts (count 1 on the first take, so
|
||||
a thread sharing the caller's space increments it); task teardown calls
|
||||
`releaseAddressSpace`, which only calls `destroyAddressSpace` at **zero**. All of
|
||||
this is already under the big kernel lock, so no new locking. This is the one piece
|
||||
that must land and be proven before anything shares an address space.
|
||||
|
||||
### `thread_spawn` and the trampoline
|
||||
|
||||
The scheduler already accepts an arbitrary `address_space` and does **not** smuggle values
|
||||
through registers — `startUserTask` reads the entry/stack from the Task and
|
||||
`jumpToUser`s ([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread
|
||||
path clean:
|
||||
The scheduler already accepts an arbitrary `address_space` and does **not** smuggle
|
||||
values through scratch registers — `startUserTask` reads the entry/stack (and the
|
||||
thread's closure arg, delivered in `rdi` via `jumpToUserArg`) from the Task
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)). That makes the thread path clean:
|
||||
|
||||
1. The runtime's `spawn` `mmap`s a stack (syscall `4`), heap-allocates a closure —
|
||||
`{ fn_ptr, args_tuple, completion }`, the std "Instance" pattern — and writes the
|
||||
closure pointer to the **top word of the new stack**.
|
||||
2. It calls `thread_spawn(entry = &threadTrampoline, stack_top, arg = closure_ptr)`.
|
||||
The kernel calls the same `spawnUserLocked` path with the **caller's address space**
|
||||
(refcount++), `entry`, and `user_sp = stack_top`.
|
||||
3. `threadTrampoline` (a small runtime shim) reads the closure off its stack, calls
|
||||
the user function, then calls `thread_exit`. No new register ABI — the closure
|
||||
pointer rides the stack the runtime set up, mirroring how `startUserTask` avoids
|
||||
register smuggling.
|
||||
1. The runtime's `spawn` `mmap`s a stack (syscall `4`) and writes the closure —
|
||||
`{ tls_base, args }`, the std "Instance" pattern — at the **top of the new stack
|
||||
itself** (no heap allocation), with a small per-thread TLS block just below it.
|
||||
2. It calls `thread_spawn(entry = &Closure.entry, stack_top, arg = closure_ptr,
|
||||
exit_endpoint = no_cap)`. The kernel calls the same `spawnUserLocked` path with the
|
||||
**caller's address space** (refcount++), `entry`, and `user_sp = stack_top`.
|
||||
3. `Closure.entry` (a small runtime shim) receives the closure pointer in `rdi` — the
|
||||
kernel delivers `arg` as the entry's first C-ABI argument — sets the thread
|
||||
pointer, calls the user function, then calls `thread_exit`.
|
||||
|
||||
Unlike a process start, there is **no** System V argc/argv/auxv block
|
||||
([sysv.md](sysv.md)) — a thread stack carries only the closure pointer.
|
||||
([sysv.md](sysv.md)) — a thread stack carries only the closure and its TLS block.
|
||||
|
||||
### Lifetime: exit, join, detach, stack reclaim
|
||||
|
||||
- **`thread_exit`** marks the task dead and hands the kernel the thread's user-stack
|
||||
range. The kernel reaps the task on the scheduler (already running on a *kernel*
|
||||
stack, so it can safely unmap the user stack), decrements the address-space refcount, and
|
||||
frees the task slot.
|
||||
- **`join` — Stage 1** reuses the existing exit-notification machinery
|
||||
([process-lifecycle.md](process-lifecycle.md)): `spawn` passes a per-thread
|
||||
`exit_endpoint`, and `join` blocks in `ipc_reply_wait` until the child-exit
|
||||
notification for that `tid` arrives, then `munmap`s the stack. No futex needed to
|
||||
land spawn/join.
|
||||
- **`join` — Stage 2 refinement** migrates to the std shape: a `completion` word in
|
||||
the closure that `thread_exit`'s trampoline `futex_wake`s and `join` `futex_wait`s
|
||||
on — dropping the per-thread endpoint. Kept as a refinement so Stage 1 ships first.
|
||||
- **`detach`** relinquishes the join right; the kernel reclaims the stack and slot on
|
||||
`thread_exit` (a detached thread's stack range is unmapped by the reaper, since no
|
||||
joiner will).
|
||||
- **`thread_exit`** (no arguments) marks the task dead. The kernel releases the
|
||||
task's resources, decrements the address-space refcount, and frees the task slot —
|
||||
the user stack is not the kernel's to unmap; the joiner reclaims it.
|
||||
- **`join`** is a dedicated `thread_join(tid)` syscall: the caller blocks in the
|
||||
kernel (`joinThreadLocked`, woken by `wakeJoinersLocked` when the thread exits),
|
||||
then `munmap`s the stack. (The plan staged join over a per-thread `exit_endpoint`
|
||||
first, with a futex `completion` word as a Stage-2 refinement; neither shipped — the
|
||||
dedicated syscall replaced both. `thread_spawn` still accepts an `exit_endpoint`
|
||||
argument, which the runtime passes as `no_cap`.)
|
||||
- **`detach`** relinquishes the join right: no one waits for the thread, and its
|
||||
stack is reclaimed at process exit — kernel-side reclaim of a detached thread's
|
||||
user stack stays deferred (as the intro notes), since `thread_exit` passes no stack
|
||||
range.
|
||||
|
||||
### Futex, and the sync primitives on top
|
||||
|
||||
@@ -210,7 +214,7 @@ Keying: threads share an address space, so a **virtual address within that addre
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`.
|
||||
Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a
|
||||
deliberate forward door: it lets two *processes* share a futex through an
|
||||
[shm](display-v2.md) region later, without changing the API. We start with the
|
||||
[shared-memory](../device-driver-development/display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-address-space key and note the physical-key upgrade.
|
||||
|
||||
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||
@@ -219,21 +223,24 @@ decision, not a "maybe later."
|
||||
|
||||
### TLS and `getCurrentId`
|
||||
|
||||
danos sets up no thread-pointer TLS today (fine under `single_threaded`). Two scoped needs:
|
||||
Per-thread thread-pointer TLS is in place (the `threadlocal` *compiler* layer is not —
|
||||
see the intro). Two scoped pieces, as built:
|
||||
|
||||
- **`getCurrentId`** returns the kernel task id — either a trivial syscall or, better,
|
||||
a value the runtime stashes in a per-thread control block.
|
||||
- **`threadlocal` variables** need a real per-thread TLS block and the thread pointer set per
|
||||
thread. `thread_spawn` sets the thread pointer to a runtime-allocated per-thread block; full
|
||||
`threadlocal` support is Stage 3, only if a consumer needs it. Nothing in the core
|
||||
- **`getCurrentId`** returns the kernel task id via the trivial `thread_self` syscall.
|
||||
- **The thread pointer** is per-thread: `spawn` carves a small TLS block (an `fs:0`
|
||||
self-pointer plus scratch) from the top of the thread's own stack, the trampoline
|
||||
calls `set_thread_pointer` before any user code runs, and the scheduler saves and
|
||||
restores the pointer per task across context switches. Full `threadlocal` support is
|
||||
runtime+linker work on top of this, only if a consumer needs it. Nothing in the core
|
||||
spawn/join/mutex path requires `threadlocal`.
|
||||
|
||||
### Build: multi-threaded codegen, opt-in
|
||||
|
||||
`addUserBinary` gains a `threaded: bool = false` parameter; when set it builds that
|
||||
binary `single_threaded = false` so atomics and (later) TLS are real. Threads and
|
||||
atomics are unsound in a `single_threaded` image, so a binary must opt in **before**
|
||||
it may call `runtime.Thread.spawn`. Everyone else stays single-threaded and lean.
|
||||
A binary opts in by being added with `addThreadedUserBinary` — as `addUserBinary`,
|
||||
but the shared implementation builds it `single_threaded = false` — so atomics and
|
||||
(later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||
so a binary must opt in **before** it may call `Thread.spawn`. Everyone else
|
||||
stays single-threaded and lean.
|
||||
|
||||
## Interaction with the rest of the kernel
|
||||
|
||||
@@ -243,18 +250,41 @@ it may call `runtime.Thread.spawn`. Everyone else stays single-threaded and lean
|
||||
process can run on different cores simultaneously — that is the point.
|
||||
- **Halting** ([halting.md](halting.md)): futex-parked waiters keep the "idle core
|
||||
halts" property intact under lock contention — no busy-wait.
|
||||
- **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): killing a process
|
||||
must kill *all* its threads and only then drop the last address-space ref. The kill path
|
||||
already targets a process; it fans out to every task on that address space.
|
||||
- **Resilience** ([resilience.md](resilience.md)): a faulting thread kills its whole
|
||||
process (shared fate). The supervisor restarts the **process**, which respawns its
|
||||
threads from a known-good state — restart granularity stays the process.
|
||||
- **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): the contract is that
|
||||
killing a process kills *all* its threads and only then drops the last address-space
|
||||
ref — and the kernel now implements exactly that
|
||||
([shared-fate-plan.md](shared-fate-plan.md)): every death path (`exit` from any
|
||||
thread, a fault, `process_kill` aimed at any member id) fans out through the whole
|
||||
group via a `dying` latch on the address space; the supervisor's one exit
|
||||
notification — badged with the leader — fires only when the last member is gone.
|
||||
A worker's voluntary `thread_exit` stays per-thread; the leader's is refused
|
||||
(`-EPERM`).
|
||||
- **Resilience** ([resilience.md](resilience.md)): by the same contract, a faulting
|
||||
thread kills its whole process (shared fate); the supervisor restarts the
|
||||
**process**, which respawns its threads from a known-good state — restart
|
||||
granularity stays the process. The leader's recorded exit reason carries the fault
|
||||
class even when a worker faulted, so restart policy is unchanged.
|
||||
- **IPC — two consequences threads forced ([ipc.md](../device-driver-development/ipc.md)):**
|
||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||
A thread that needs to reach an endpoint another thread owns looks it up
|
||||
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
||||
This is how the display's mouse-listener thread reaches the compositor loop's endpoint
|
||||
to poke it awake (docs/display.md).
|
||||
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
||||
`create_ipc_endpoint`/`ipc_register`/`ipc_lookup` allocate from the kernel heap and
|
||||
mutate the global service registry, endpoint refcounts, and handle tables. Those paths
|
||||
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
||||
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
||||
`send` already did — the kernel heap has no lock of its own (heap.zig: "every kernel
|
||||
entry takes the big kernel lock"), so the big lock is what keeps its callers serialized.
|
||||
|
||||
## Build-out plan (staged, each gate serial-checkable)
|
||||
|
||||
The ordered, `/loop`-runnable milestones live in
|
||||
**[threading-plan.md](threading-plan.md)** (shaped like
|
||||
[display-v2-plan.md](display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
[display-v2-plan.md](../device-driver-development/display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||
|
||||
@@ -263,11 +293,13 @@ a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
*Gate:* the full QEMU suite stays green (no regression) — proves the reframing is
|
||||
invisible until used.
|
||||
- **Stage 1 — spawn / join / detach.** `thread_spawn` + `thread_exit`, the trampoline,
|
||||
stacks via `mmap`, join over the exit-endpoint, the `threaded` build flag.
|
||||
*Gate:* `-Dtest-case=thread-spawn` — a threaded test service spawns N threads that
|
||||
each `@atomicRmw`-increment a shared counter, the parent joins all N, and asserts
|
||||
the total is exactly N × iterations. Runs `smp` (multi-core) to prove real
|
||||
parallelism.
|
||||
stacks via `mmap`, join over the exit-endpoint (as built, join became the dedicated
|
||||
`thread_join` syscall instead), the `addThreadedUserBinary` build opt-in.
|
||||
*Gate:* two cases as built — `-Dtest-case=thread-spawn`, where a worker thread runs
|
||||
in the caller's address space (a shared-memory write, observed by the main thread),
|
||||
and `-Dtest-case=thread-join`, where N workers each atomically increment a shared
|
||||
counter K times, the parent joins all N and asserts the total is exactly N × K —
|
||||
the join case running multi-core (`smp` 4) to prove real parallelism.
|
||||
- **Stage 2 — blocking synchronization.** `futex_wait`/`futex_wake` + `Futex`,
|
||||
`Mutex`, `Condition`, `Semaphore`; optionally migrate join to a futex completion
|
||||
word. *Gate:* `-Dtest-case=thread-mutex` — a bounded producer/consumer over a
|
||||
@@ -275,15 +307,15 @@ a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
the consumer blocked, e.g. via a low idle tick count).
|
||||
- **Stage 3 — polish.** Per-thread TLS / thread pointer and `threadlocal` (only if a
|
||||
consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired
|
||||
into [test/qemu_test.py](../test/qemu_test.py).
|
||||
into [test/qemu_test.py](../../test/qemu_test.py).
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym
|
||||
abbreviations, kebab-case file names, no `Co-Authored-By` trailers. New syscalls
|
||||
extend [abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper
|
||||
([syscall.md](syscall.md)). `runtime.Thread` is a first-class runtime module, the same
|
||||
way `runtime.process` ([process-lifecycle.md](process-lifecycle.md)) and `runtime.ipc`
|
||||
extend [abi.zig](../../system/abi.zig) `SystemCall` + a `library/kernel` wrapper
|
||||
([syscall.md](syscall.md)). `Thread` is a first-class runtime module, the same
|
||||
way `process` ([process-lifecycle.md](process-lifecycle.md)) and `ipc`
|
||||
are — user code never names a syscall.
|
||||
|
||||
## Non-goals
|
||||
@@ -299,19 +331,19 @@ are — user code never names a syscall.
|
||||
## The self-hosting endgame
|
||||
|
||||
When danos becomes a real Zig target and we (eventually) add a danos backend to std
|
||||
([zig-self-hosting.md](zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
([zig-self-hosting.md](../zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
same kernel primitives — the danos `std.Thread.Impl` would call the very
|
||||
`thread_spawn`/`futex_*` wrappers `runtime.Thread` already uses. Because
|
||||
`runtime.Thread` was built API-compatible from day one, that transition swaps the
|
||||
`thread_spawn`/`futex_*` wrappers `Thread` already uses. Because
|
||||
`Thread` was built API-compatible from day one, that transition swaps the
|
||||
implementation, not a single call site. Designing to the std shape now is what makes
|
||||
the later self-hosting lift cheap.
|
||||
|
||||
## Further reading
|
||||
|
||||
- [scheduling.md](scheduling.md), [smp.md](smp.md) — the task model these threads join.
|
||||
- [resilience.md](resilience.md), [vision.md](vision.md) — why isolation is the default
|
||||
- [resilience.md](resilience.md), [vision.md](../vision.md) — why isolation is the default
|
||||
and threads are the exception.
|
||||
- [syscall.md](syscall.md), [ipc.md](ipc.md) — the private ABI and the messaging model
|
||||
- [syscall.md](syscall.md), [ipc.md](../device-driver-development/ipc.md) — the private ABI and the messaging model
|
||||
threads sit beside.
|
||||
- [halting.md](halting.md) — the idle/halt property futex-backed blocking preserves.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the target this bends toward.
|
||||
- [zig-self-hosting.md](../zig-self-hosting.md) — the target this bends toward.
|
||||
@@ -7,7 +7,7 @@ Two different needs hide under the word "timer", and danos keeps them apart:
|
||||
|
||||
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||
are built in [device-interrupts.md](../device-driver-development/device-interrupts.md); the scheduler's blocking and
|
||||
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||
service.**
|
||||
@@ -32,11 +32,11 @@ danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on I
|
||||
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
[device-interrupts.md](device-interrupts.md).
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md).
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||
model; that role now lives in [drivers.md](../device-driver-development/drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
|
||||
@@ -55,20 +55,20 @@ Time and waiting are three entries in the small syscall table ([syscall.md](sysc
|
||||
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||
[device-manager.md](device-manager.md)).
|
||||
[device-manager.md](../device-driver-development/device-manager.md)).
|
||||
|
||||
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||
both riding the scheduler tick.
|
||||
|
||||
## `runtime.time` — the generic interface
|
||||
## `time` — the generic interface
|
||||
|
||||
Applications don't call the syscalls directly; they use `runtime.time`
|
||||
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
Applications don't call the syscalls directly; they use `time`
|
||||
(`library/kernel/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
front door, not new mechanism.
|
||||
|
||||
```zig
|
||||
const time = @import("runtime").time;
|
||||
const time = @import("time");
|
||||
|
||||
const start = time.now(); // Instant — monotonic
|
||||
doWork();
|
||||
@@ -91,8 +91,9 @@ _ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||
|
||||
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||
The raw wrappers (`clock`, `sleepMillis`, `timerOnce`) and the ergonomic
|
||||
`Instant`/`Duration` layer both live in the `time` module
|
||||
(`library/kernel/time.zig`); the latter is what everyday code uses.
|
||||
|
||||
## Wall-clock time (not built)
|
||||
|
||||
@@ -106,10 +107,10 @@ owns covers every current use.
|
||||
|
||||
## Verifying it
|
||||
|
||||
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
`time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
|
||||
```
|
||||
$ zig build test # includes library/runtime/time.zig
|
||||
$ zig build test # includes library/kernel/time.zig
|
||||
```
|
||||
|
||||
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||
@@ -1,7 +1,7 @@
|
||||
# The vDSO — the public system-call boundary
|
||||
|
||||
> **Status:** design note, not built. The runtime today issues raw `syscall`
|
||||
> instructions from `library/runtime/system-call.zig` using the numbers in
|
||||
> instructions from `library/kernel/system-call.zig` using the numbers in
|
||||
> `system/abi.zig`. This note designs the layer that replaces that arrangement:
|
||||
> a **kernel-supplied, C-ABI entry library** mapped into every process — the
|
||||
> only supported way into the kernel — so the raw numbers can stay private,
|
||||
@@ -25,8 +25,8 @@ ourselves:
|
||||
this mistake: it issued XNU syscalls directly instead of going through
|
||||
libSystem, and macOS updates repeatedly broke every Go binary until Go
|
||||
switched to the library like everyone else.
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the `runtime`
|
||||
module. The public boundary has to be expressible in the one calling
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the danos Zig
|
||||
modules. The public boundary has to be expressible in the one calling
|
||||
convention every language speaks: the C ABI.
|
||||
3. **Randomised syscall numbers** — a hardening option we want open — only
|
||||
work if no user binary anywhere knows a number at build time. The binding
|
||||
@@ -49,10 +49,10 @@ The public danos ABI then has exactly two layers, neither of which is
|
||||
|
||||
| Layer | Contract | Spoken by |
|
||||
|-------|----------|-----------|
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (`runtime.system` for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (the `system-call` module for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](../file-system-development/vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
|
||||
Everything above those — the heap, `runtime.fs`, the service harness — is
|
||||
Everything above those — the heap, `file_system`, the service harness — is
|
||||
per-language convenience, compiled into each binary from source, exactly as
|
||||
today. Nothing about the Zig runtime's shape changes; it just stops being the
|
||||
*only* door.
|
||||
@@ -72,7 +72,7 @@ into every process's address space. The blob is:
|
||||
- **Architecture-specific.** The x86-64 blob wraps `syscall`; an aarch64 blob
|
||||
wraps `svc #0`. It lives beside the other per-architecture kernel sources
|
||||
(`system/kernel/architecture/<arch>/`), selected the same way the
|
||||
`architecture` module is (docs/arch.md).
|
||||
`architecture` module is (docs/architecture.md).
|
||||
|
||||
### Shape: a function table, not an ELF
|
||||
|
||||
@@ -107,7 +107,7 @@ convenience, not a requirement.)
|
||||
|
||||
The kernel already builds a System V entry block — argc, argv, envp
|
||||
terminator, **auxiliary vector** — on every new process's stack
|
||||
(`buildEntryStack`, read by `runtime.start`). The vDSO base rides in a new
|
||||
(`buildEntryStack`, read by the `start` module). The vDSO base rides in a new
|
||||
auxv entry, exactly Linux's `AT_SYSINFO_EHDR` move. No new syscall, no magic
|
||||
address, and a language shim finds it the same portable way on every
|
||||
architecture.
|
||||
@@ -119,21 +119,29 @@ One table entry per kernel call, C ABI (System V AMD64), names prefixed
|
||||
returns are `u64`, errors return as negative values exactly as today.
|
||||
|
||||
The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||
(virtual_address + physical_address), `msi_bind` (address + data), `shm_create` (virtual_address + handle) —
|
||||
(virtual_address + physical_address), `msi_bind` (address + data), `shared_memory_create` (virtual_address + handle),
|
||||
`fs_resolve` (route tag + node token / backend handle) —
|
||||
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||
C-ABI spelling of the existing convention, at zero cost.
|
||||
C-ABI spelling of the existing convention, at zero cost. The one call that
|
||||
returns *three* values — `ipc_reply_wait` (receive_len in `rax`, badge in
|
||||
`rdx`, received capability in `r8`) — exceeds the two-register return: its
|
||||
function returns a three-`u64` struct, which the ABI passes via a hidden
|
||||
result pointer, so that one stub stores `rax`/`rdx`/`r8` through the pointer
|
||||
after the `syscall` — a few instructions rather than one.
|
||||
|
||||
Grouped as `abi.zig` groups them:
|
||||
|
||||
| Group | Functions |
|
||||
|-------|-----------|
|
||||
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shm_create`, `danos_shm_map`, `danos_shm_physical` |
|
||||
| threads | `danos_thread_spawn`, `danos_thread_exit`, `danos_current_core`, `danos_futex_wait`, `danos_futex_wake`, `danos_thread_self`, `danos_thread_join`, `danos_set_thread_pointer` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shared_memory_create`, `danos_shared_memory_map`, `danos_shared_memory_physical` |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
||||
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||
| diagnostics | `danos_debug_write`, `danos_klog_read` |
|
||||
| diagnostics | `danos_debug_write` (leveled, kernel-stamped records), `danos_klog_read`, `danos_klog_status` |
|
||||
| filesystem naming | `danos_fs_resolve`, `danos_fs_node`, `danos_fs_mount`, `danos_fs_unmount` (naming only — file DATA still crosses the vfs-protocol IPC, see below) |
|
||||
|
||||
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
||||
@@ -175,11 +183,11 @@ second — but the design should never be sold as more than that.
|
||||
Phased so every step ships alone (the M-milestone discipline):
|
||||
|
||||
1. **The blob + the table.** Build the vDSO, map it at spawn, deliver the
|
||||
base via auxv. `runtime.system-call.zig` binds through the table when the
|
||||
base via auxv. `library/kernel/system-call.zig` binds through the table when the
|
||||
auxv entry is present, falls back to raw `syscall` when absent — the whole
|
||||
tree keeps booting during the transition.
|
||||
2. **Cut the runtime over.** Delete the raw stubs; `runtime` no longer
|
||||
imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
2. **Cut the system library over.** Delete the raw stubs; the `system-call`
|
||||
module no longer imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
kernel-internal). The QEMU suite passing proves the table carries the
|
||||
whole system.
|
||||
3. **Enforce + randomise.** Add the `rip`-range check, then per-boot number
|
||||
@@ -197,8 +205,10 @@ Phased so every step ships alone (the M-milestone discipline):
|
||||
across processes stays what it is today: a service behind IPC, or source
|
||||
compiled into each binary.
|
||||
- **No file/device I/O in the vDSO.** The microkernel line doesn't move: the
|
||||
vDSO wraps the same deliberately tiny table (docs/syscall.md); files are
|
||||
still the VFS server's business over IPC.
|
||||
vDSO wraps the same deliberately tiny table (docs/syscall.md). The kernel
|
||||
resolves file NAMES (`fs_resolve` — the mount table moved in-kernel), but
|
||||
file data is still the filesystem server's business over the vfs-protocol
|
||||
IPC; the kernel never blocks on a userspace filesystem.
|
||||
- **No fast-path user-mode implementations yet.** Linux's vDSO exists mostly
|
||||
to answer `gettimeofday` without a kernel entry. `danos_clock` could one
|
||||
day read the calibrated TSC in user mode the same way — the blob is where
|
||||
+54
-45
@@ -81,10 +81,11 @@ they are guidance, not a tested-hardware list.)
|
||||
|
||||
- **Firmware must be UEFI.** Many 2011-era machines could do either UEFI or
|
||||
legacy BIOS — danos needs it set to UEFI. There is no BIOS boot path.
|
||||
- **Input is PS/2 only, for now.** danos does not yet support USB
|
||||
keyboards/mice. This is fine on most **laptops** (their built-in keyboards are
|
||||
wired to a PS/2-style i8042 controller) but means a **desktop with only USB
|
||||
ports** currently has no usable keyboard. USB HID input is planned.
|
||||
- **Internal disks are invisible.** The only block-device driver is USB mass
|
||||
storage (`usb-storage`) — there is no AHCI / NVMe / IDE driver — so danos
|
||||
boots from and stores to a **USB stick**, not the machine's internal
|
||||
SATA/NVMe disk. Keyboards and mice are fine either way: both PS/2 (most
|
||||
laptops' built-in keyboards) and USB HID work.
|
||||
|
||||
Virtual machines are the easiest way to meet every requirement: QEMU (with OVMF/
|
||||
UEFI, a `qemu-xhci` controller, and the default Q35 machine type), or any
|
||||
@@ -94,40 +95,43 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
|
||||
| Requirement | Detail | Source |
|
||||
|---|---|---|
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:285`, `boot/efi.zig:418` |
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:481`, `boot/efi.zig:622` |
|
||||
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:282`, `trampoline.s:62` |
|
||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:59`, `isr.s:169` |
|
||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:62`, `apic.zig:414` |
|
||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:279`, `apic.zig:84` |
|
||||
| **SMP (optional)** | Multi-core supported via INIT–SIPI–SIPI; ceiling `maximum_cpus = 128`. Single core is fine. Cores beyond the ceiling are parked. | `system/parameters.zig:16`, `apic.zig:144` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:477`, `trampoline.s:62` |
|
||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:71`, `isr.s:196` |
|
||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:67`, `apic.zig:646` |
|
||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:333`, `apic.zig:113` |
|
||||
| **SMP (optional)** | Multi-core supported via INIT–SIPI–SIPI; ceiling `maximum_cpus = 128`. Single core is fine. Cores beyond the ceiling are parked. | `system/parameters.zig:16`, `apic.zig:173` |
|
||||
|
||||
## Firmware / boot
|
||||
|
||||
- **UEFI only.** A custom UEFI application loader is installed to
|
||||
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||
(`build.zig:464`, `boot/efi.zig`)
|
||||
(`build.zig:246`, `boot/efi.zig`)
|
||||
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||
(`efi.zig:578`, `boot-handoff.zig:144`)
|
||||
(`efi.zig:790`, `boot-handoff.zig:149`)
|
||||
- **Required ACPI tables:** MADT (interrupt topology), MCFG (PCIe ECAM base),
|
||||
FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR.
|
||||
(`system/devices/acpi.zig:3`)
|
||||
- The loader reads `/system/kernel`, `/system/services/init`, and
|
||||
`/boot/initial-ramdisk.img` off the FAT boot volume. The kernel can boot
|
||||
"kernel-only" without init or the ramdisk. (`efi.zig:14`, `efi.zig:66`)
|
||||
(`system/kernel/acpi.zig:3`)
|
||||
- The loader reads `/system/kernel` off the FAT boot volume, then loads user
|
||||
space: a prebuilt `boot\system.img` capsule
|
||||
([system-image.md](os-development/system-image.md)) when present, otherwise it walks
|
||||
the volume's `/system` and optional `/test` trees (init included) into the
|
||||
initial ramdisk. The kernel can boot "kernel-only" without either.
|
||||
(`efi.zig:16`, `efi.zig:68`)
|
||||
|
||||
## Interrupt controller
|
||||
|
||||
- **Local APIC + I/O APIC required.** I/O APIC base, GSI base, and MADT
|
||||
interrupt-source overrides come from ACPI. (`cpu.zig:365`, `apic.zig:119`)
|
||||
interrupt-source overrides come from ACPI. (`cpu.zig:388`, `ioapic.zig:35`)
|
||||
- **MSI supported** — edge-triggered, keyed by vector, no I/O APIC mask cycle.
|
||||
Vector window 33–46, timer on 32, spurious on 47. (`system/kernel/irq.zig:70`,
|
||||
`cpu.zig:397`)
|
||||
`cpu.zig:427`)
|
||||
- The legacy 8259 PIC is remapped and masked **only if present** (MADT
|
||||
`PCAT_COMPAT`); it is not required. (`apic.zig:103`)
|
||||
`PCAT_COMPAT`); it is not required. (`apic.zig:149`)
|
||||
|
||||
## PCI / PCIe
|
||||
|
||||
@@ -135,46 +139,47 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
bridge's ECAM window (1 MiB config space per bus) and computes config
|
||||
addresses directly. **There is no legacy CF8/CFC port-IO config path** — the
|
||||
driver bails if the bridge exposes no ECAM window. The ECAM base comes from
|
||||
the ACPI MCFG table. (`system/drivers/pci-bus/pci-bus.zig:41`, `acpi.zig:6`)
|
||||
the ACPI MCFG table. (`system/drivers/pci-bus/pci-bus.zig:86`, `acpi.zig:7`)
|
||||
|
||||
## USB
|
||||
|
||||
- **xHCI only.** The sole USB driver is `usb-xhci-bus`, and the device manager
|
||||
binds it strictly to PCI prog-IF `0x30` (xHCI). UHCI / OHCI / EHCI exist only
|
||||
as report strings with no driver behind them — **USB 1.x/2.0-only controllers
|
||||
are not supported.** (`system/drivers/usb-xhci-bus/`,
|
||||
`system/services/device-manager/device-manager.zig:34`)
|
||||
- USB input (keyboard/mouse over HID) is future work; the current input stack is
|
||||
PS/2. See [Buses & devices](#buses--devices).
|
||||
- **xHCI only.** The sole USB *host-controller* driver is `usb-xhci-bus`, and
|
||||
the device manager binds it strictly to PCI prog-IF `0x30` (xHCI). UHCI /
|
||||
OHCI / EHCI exist only as report strings with no driver behind them — **USB
|
||||
1.x/2.0-only controllers are not supported.** (`system/drivers/usb-xhci-bus/`,
|
||||
`system/services/device-manager/device-manager.zig:30`)
|
||||
- USB class drivers ride on top of it: HID keyboard + mouse (`usb-hid`, feeding
|
||||
the same input service as PS/2) and mass storage (`usb-storage`). See
|
||||
[Buses & devices](#buses--devices).
|
||||
|
||||
## Timers
|
||||
|
||||
Calibration prefers, in order: (1) CPUID leaf `0x15` TSC frequency, (2) HPET,
|
||||
(3) ACPI PM timer (3.579545 MHz, from FADT), (4) legacy PIT. Any one suffices —
|
||||
HPET/PM-timer/PIT are optional fallbacks when CPUID `0x15` is absent.
|
||||
(`apic.zig:180`)
|
||||
(`apic.zig:209`)
|
||||
|
||||
- **TSC** — monotonic high-resolution clock.
|
||||
- **LAPIC timer** — scheduler heartbeat, periodic at `timer_hz = 1000 Hz`.
|
||||
(`parameters.zig:39`)
|
||||
(`parameters.zig:42`)
|
||||
|
||||
## Memory
|
||||
|
||||
**Target: 128 MiB RAM.** The system uses 4 KiB pages and a bitmap physical-frame
|
||||
allocator built from the firmware memory map. There is no hardcoded minimum-RAM
|
||||
constant — the allocator only panics if there is no usable region, or none large
|
||||
enough to hold its own bitmap. (`system/kernel/pmm.zig:13`, `pmm.zig:77`)
|
||||
enough to hold its own bitmap. (`system/kernel/pmm.zig:15`, `pmm.zig:77`)
|
||||
|
||||
Where the budget goes:
|
||||
|
||||
| Consumer | Size | Source |
|
||||
|---|---|---|
|
||||
| Kernel heap (cap, grown one page at a time) | up to **64 MiB** | `system/kernel/heap.zig:26` |
|
||||
| Kernel stack, per CPU | 16 KiB | `parameters.zig:26` |
|
||||
| IST stack, per CPU | 16 KiB | `parameters.zig:36` |
|
||||
| User stack, per task | 8 pages / 32 KiB | `parameters.zig:32` |
|
||||
| Max concurrent tasks | 32 | `parameters.zig:23` |
|
||||
| Boot page-table pool | 64 frames / 256 KiB | `efi.zig:299` |
|
||||
| Kernel heap (cap, grown one page at a time) | up to **64 MiB** | `system/kernel/heap.zig:28` |
|
||||
| Kernel stack, per CPU | 16 KiB | `parameters.zig:29` |
|
||||
| IST stack, per CPU | 16 KiB | `parameters.zig:39` |
|
||||
| User stack, per task | 8 pages / 32 KiB | `parameters.zig:35` |
|
||||
| Max concurrent tasks | 48 | `parameters.zig:26` |
|
||||
| Boot page-table pool | 64 frames / 256 KiB | `efi.zig:307` |
|
||||
|
||||
The 64 MiB heap cap plus kernel image, per-CPU stacks, task stacks, the frame
|
||||
bitmap, and DMA-contiguous allocations fit comfortably within 128 MiB on a
|
||||
@@ -184,9 +189,9 @@ ceiling) add per-CPU stack overhead and push toward more RAM.
|
||||
**Note on the 4 GiB physmap:** the loader identity-maps and physmaps the low
|
||||
4 GiB of address space with 2 MiB leaves. This is *virtual address* reach, not a
|
||||
RAM requirement — RAM above 4 GiB simply needs an extra mapping window and is not
|
||||
needed to boot. (`efi.zig:305`)
|
||||
needed to boot. (`efi.zig:313`)
|
||||
|
||||
Virtual-memory layout (`boot-handoff.zig:47`):
|
||||
Virtual-memory layout (`boot-handoff.zig:58`):
|
||||
|
||||
| Region | Base |
|
||||
|---|---|
|
||||
@@ -201,19 +206,23 @@ Buses with real drivers today:
|
||||
|
||||
- **PCIe** via ECAM (`pci-bus`)
|
||||
- **xHCI USB** (`usb-xhci-bus`)
|
||||
- **PS/2** keyboard + mouse (`ps2-bus`) — the current input stack
|
||||
- **PS/2** keyboard + mouse (`ps2-bus`)
|
||||
- **USB HID** keyboard + mouse (`usb-hid`) — PS/2 and USB HID feed the same
|
||||
input service
|
||||
- **Serial UART** (16550/16450), configured from the ACPI SPCR table
|
||||
|
||||
**No storage driver exists yet.** AHCI / NVMe / IDE are named for reporting only;
|
||||
there is no block-device driver. Persistent storage is future work.
|
||||
**Storage is USB mass storage only.** The block-device driver is `usb-storage`
|
||||
(bulk-only transport + SCSI), and the FAT server reads and writes it —
|
||||
create / truncate / mkdir / unlink / rename, with mtime. AHCI / NVMe / IDE are
|
||||
named for reporting only; internal SATA / NVMe / IDE disks have no driver.
|
||||
|
||||
## IOMMU
|
||||
|
||||
**Detection only; enforcement deferred.** The ACPI DMAR table is parsed for the
|
||||
first VT-d DRHD unit and its capabilities are exposed via `PlatformInfo`
|
||||
first VT-d DRHD unit and its capabilities are exposed via `PlatformInformation`
|
||||
(`iommu_present`, `iommu_base`, `iommu_version`). No DMA-remapping tables are
|
||||
programmed and no translation is enforced. An IOMMU is therefore **not required**
|
||||
and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||
and does not currently constrain devices. (`system/kernel/acpi.zig:96`)
|
||||
|
||||
## What is explicitly NOT supported
|
||||
|
||||
@@ -223,5 +232,5 @@ and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||
- Legacy port-IO (CF8/CFC) PCI configuration
|
||||
- Non-xHCI USB (UHCI / OHCI / EHCI)
|
||||
- Machines without ACPI (no device discovery)
|
||||
- Persistent storage (no AHCI / NVMe / IDE driver yet)
|
||||
- USB HID input (PS/2 only for now)
|
||||
- Internal-disk storage (no AHCI / NVMe / IDE driver — persistent storage means
|
||||
USB mass storage)
|
||||
|
||||
+32
-14
@@ -7,10 +7,13 @@ without a human staring at the screen.
|
||||
|
||||
There are two layers:
|
||||
|
||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
||||
the shared contracts (`system/boot-handoff.zig`, `system/abi.zig`,
|
||||
`system/devices/device-abi.zig`), which also compile-checks the three-way split
|
||||
stays self-consistent. These compile for the host and run natively.
|
||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic.
|
||||
What began as the three shared contracts (`system/boot-handoff.zig`,
|
||||
`system/abi.zig`, `library/device/model/device-abi.zig`) now spans ~26 modules:
|
||||
protocol and on-wire definitions (VFS, USB, virtio-gpu), the FAT engine, the
|
||||
display compositor, PS/2 and HID decoding, the kernel log ring, and the
|
||||
runtime's `time`/`thread` — the full list is the test step in `build.zig`.
|
||||
These compile for the host and run natively.
|
||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||
and check its behaviour. This is the interesting part.
|
||||
|
||||
@@ -18,13 +21,13 @@ There are two layers:
|
||||
|
||||
The framebuffer console draws pixels, which a test can't read without
|
||||
screen-scraping. So the kernel also writes everything to a **serial port**
|
||||
(`system/kernel/architecture/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
||||
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
||||
appears on serial as plain text.
|
||||
(`system/kernel/architecture/x86_64/serial.zig`, a 16550 UART on COM1). The diagnostic log fans out
|
||||
to registered sinks, and the serial UART is one of them, so all kernel output —
|
||||
boot log, memory summary, exception reports — appears on serial as plain text.
|
||||
|
||||
QEMU captures that with `-serial file:serial.log`, giving a machine-readable
|
||||
transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
||||
memory-mapped UART), so it lives behind the [arch](arch.md) boundary — and adding
|
||||
memory-mapped UART), so it lives behind the [architecture](os-development/architecture.md) boundary — and adding
|
||||
a new architecture's UART is what makes the same tests run there.
|
||||
|
||||
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||
@@ -51,13 +54,18 @@ DANOS-TEST-RESULT: PASS (6 passed, 0 failed)
|
||||
DANOS-TEST-DONE
|
||||
```
|
||||
|
||||
Current cases:
|
||||
The core kernel cases (the suite has since grown far beyond this table — SMP,
|
||||
threads, processes, display, USB, FAT and more; the full list is the `CASES`
|
||||
table in `test/qemu_test.py`):
|
||||
|
||||
| Case | What it checks | How the harness confirms it |
|
||||
|------|----------------|-----------------------------|
|
||||
| `smoke` | memory map has usable RAM; frame alloc/free; paging active | `DANOS-TEST-RESULT: PASS` |
|
||||
| `discovery` | ACPI discovery populated the platform facts: MADT (LAPIC base, CPU count) and FADT (PM/reset registers) | `DANOS-TEST-RESULT: PASS` |
|
||||
| `wx` | W^X audit: kernel code is executable; rodata, data, heap, and stack are NX | `DANOS-TEST-RESULT: PASS` |
|
||||
| `timer` | device interrupts fire and return (tick count advances) | `DANOS-TEST-RESULT: PASS` |
|
||||
| `clock` | LAPIC + TSC calibrated; monotonic uptime advances; `nanos()` has sub-ms resolution | `DANOS-TEST-RESULT: PASS` |
|
||||
| `wall-clock` | the CMOS RTC read at boot yields a plausible current epoch | `DANOS-TEST-RESULT: PASS` |
|
||||
| `vmm` | on-demand `map` works: a mapped page is writable and reads back | `DANOS-TEST-RESULT: PASS` |
|
||||
| `heap` | kernel heap: alloc/free, block reuse, growth, and a std container on it | `DANOS-TEST-RESULT: PASS` |
|
||||
| `sched` | preemption: three non-yielding tasks all make progress | `DANOS-TEST-RESULT: PASS` |
|
||||
@@ -65,14 +73,22 @@ Current cases:
|
||||
| `sleep` | a task blocks for ~50 ms (real block, not a busy-wait) | `DANOS-TEST-RESULT: PASS` |
|
||||
| `event` | a task blocks on a wait queue and is woken (preempting) | `DANOS-TEST-RESULT: PASS` |
|
||||
| `ipc` | producer/consumer pass 100 messages through a 4-slot channel intact | `DANOS-TEST-RESULT: PASS` |
|
||||
| `ipc-call` | synchronous IPC: client and server ping-pong 100 calls through one endpoint (rendezvous, reply routing, cross-copy) | `DANOS-TEST-RESULT: PASS` |
|
||||
| `ipc-cap` | capability passing: endpoints handed over in a call and its reply arrive as the same object, shared not moved | `DANOS-TEST-RESULT: PASS` |
|
||||
| `dma` | DMA memory: contiguous frame allocation, below-4G cap, coherent mapping, reclaim on teardown | `DANOS-TEST-RESULT: PASS` |
|
||||
| `msi` | an MSI vector is allocated and delivered as an endpoint notification (a self-IPI stands in for the device write) | `DANOS-TEST-RESULT: PASS` |
|
||||
| `iommu` | the VT-d unit is found in the DMAR table and its registers read back (detection only; boots with an emulated IOMMU) | `DANOS-TEST-RESULT: PASS` |
|
||||
| `ioport` | port I/O grants: an `io_port` resource admits in-range reads, refuses out-of-range/unclaimed | `DANOS-TEST-RESULT: PASS` |
|
||||
| `fault-ud` | invalid-opcode exception is caught | serial shows `invalid opcode (vector 6)` |
|
||||
| `fault-pf` | page fault caught with CR2 | `page fault (vector 14)` |
|
||||
| `fault-df` | double fault caught on IST1 (not a triple-fault reset) | `double fault (vector 8)` |
|
||||
| `fault-ap-df` | a double fault pinned to an application processor is caught by that core's own TSS/IST (boots with `-smp 4`) | `core N: double fault (vector 8)`, N ≥ 1 |
|
||||
| `fault-nx` | executing a data page (NX) faults | `page fault (vector 14)` |
|
||||
| `fault-null` | dereferencing the unmapped page 0 faults | `page fault (vector 14)` |
|
||||
| `fault-recovery` | a ring-3 process that faults is killed and reaped while init keeps heartbeating — the OS survives | `DANOS-TEST-RESULT: PASS` |
|
||||
|
||||
The faulting cases don't print a result line — they deliberately raise a CPU
|
||||
exception, and the harness asserts on the [exception report](interrupts.md) the
|
||||
exception, and the harness asserts on the [exception report](os-development/interrupts.md) the
|
||||
handler prints (which also reaches serial). This reuses the real fault path as the
|
||||
test oracle: if the IDT/TSS weren't wired up, `fault-df` would triple-fault and the
|
||||
marker would never appear.
|
||||
@@ -81,8 +97,10 @@ marker would never appear.
|
||||
|
||||
`test/qemu_test.py` ties it together. For each case it:
|
||||
|
||||
1. builds the kernel with `-Dtest-case=<name>`,
|
||||
2. assembles a fresh EFI System Partition from the built binaries,
|
||||
1. builds the kernel with `-Dtest-case=<name>` (the build produces the bootable
|
||||
FAT32 USB image, `zig-out/danos-usb.img`),
|
||||
2. copies that image to a fresh per-run boot volume, so the guest's mutations
|
||||
don't dirty the build artifact,
|
||||
3. boots it headless in QEMU with serial captured to a file and `-no-reboot`
|
||||
(so a triple fault exits rather than looping),
|
||||
4. polls the serial log until the case's expected regex appears (**pass**), a
|
||||
@@ -118,8 +136,8 @@ firmware, boot method, serial device). The cases are architecture-neutral —
|
||||
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
||||
one — means:
|
||||
|
||||
1. implement `system/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||
tables) behind the same `arch` interface,
|
||||
1. implement `system/kernel/architecture/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||
tables) behind the same `architecture` interface,
|
||||
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
||||
|
||||
and the *same* `smoke` / `fault-*` cases run against it: `python3 test/qemu_test.py
|
||||
|
||||
-117
@@ -1,117 +0,0 @@
|
||||
# Vision: a microkernel, built to learn
|
||||
|
||||
danos exists first and foremost as a **learning-by-doing project**: the point is to
|
||||
build a real operating system, bump into the hard constraints for real, and research
|
||||
them from a position of having actually hit them. The docs in this folder are part of
|
||||
that — they're where a constraint gets understood once it's been met.
|
||||
|
||||
That framing sets the priorities. danos is not chasing a spec or a product; it's
|
||||
chasing understanding, with a concrete, motivating **win condition** to aim at.
|
||||
|
||||
## The win condition
|
||||
|
||||
danos is a "win" when it:
|
||||
|
||||
- **boots and runs on real hardware** — the author's **PC** (x86-64) and **both
|
||||
Raspberry Pis**: the **Zero 2 W** and the **Pi 5** (both `aarch64`, one backend —
|
||||
see [arm.md](arm.md)),
|
||||
- **has a graphical user interface**, ideally — building on the framebuffer it
|
||||
already draws to.
|
||||
|
||||
Everything below serves that, or serves the curiosity that the project runs on.
|
||||
|
||||
## Why a microkernel: resilience
|
||||
|
||||
The kernel stays **minimal** — only what genuinely must run privileged:
|
||||
|
||||
- scheduling,
|
||||
- inter-process communication (IPC),
|
||||
- memory management (address spaces, page tables),
|
||||
- low-level interrupt dispatch.
|
||||
|
||||
Everything else — device drivers, filesystems, the GUI, the network stack — runs as
|
||||
an **isolated user-space server**, each in its own address space with only the
|
||||
privileges it needs.
|
||||
|
||||
The reason for this shape is **resilience**: the ability to **re-initialise parts of
|
||||
the OS while it runs**. A driver bug can't corrupt the kernel or another driver; a
|
||||
crashed or wedged component is contained, killed, and **restarted** — "if I break
|
||||
something, I can just fix it," without rebooting. Keeping the kernel tiny is part of
|
||||
that strategy: the one thing that *can't* be restarted is the trusted base, so the
|
||||
less code in it, the less that can take the whole system down. This is the project's
|
||||
real motivation, and it has its own design note: [resilience.md](resilience.md).
|
||||
|
||||
The cost is that **IPC becomes the backbone**: what used to be a function call inside
|
||||
a monolithic kernel is now a message between address spaces. In a microkernel, IPC
|
||||
performance essentially *is* system performance (the lesson of L4), so it's a
|
||||
first-class concern. Hardware interrupts become IPC too: the kernel turns an IRQ into
|
||||
a message to the driver that owns the device.
|
||||
|
||||
## On real-time: an option, not a commitment
|
||||
|
||||
danos was originally framed as a hard **real-time** OS. That's now held as **one
|
||||
interesting constraint to explore, not a requirement** — because real-time is a
|
||||
*pervasive* invariant (every operation must be provably time-bounded, everywhere)
|
||||
that would slow every milestone, whereas resilience is a set of *structural* features
|
||||
that's lighter to build and is what the project actually wants. The trade-off is
|
||||
written up in [smp.md](smp.md#does-the-right-choice-depend-on-real-time-vs-resilience).
|
||||
|
||||
What danos keeps from the real-time direction, because it's cheap and useful anyway:
|
||||
|
||||
- **Fixed-priority preemptive scheduling** — the highest-priority ready task runs, and
|
||||
preemption lets a runaway component be interrupted and killed (which *serves
|
||||
resilience*). Already built ([scheduling.md](scheduling.md)).
|
||||
- **A calibrated, deterministic clock** — already built ([device-interrupts.md](device-interrupts.md)).
|
||||
|
||||
What danos does *not* owe anyone unless it deliberately chooses real-time later:
|
||||
timing *guarantees*, priority inheritance, bounded allocators, tickless timers, MCS
|
||||
scheduling contexts. Concretely, the current [heap](heap.md) is a first-fit free list
|
||||
with unbounded allocation time — fine here, and only a problem *if* a hard-real-time
|
||||
path is ever added. Note that **QNX is both** a real-time and a restartable
|
||||
microkernel, so choosing resilience now doesn't close the real-time door — it just
|
||||
doesn't pay the tax yet.
|
||||
|
||||
## The roadmap — tracks, not a strict line
|
||||
|
||||
Because the driver is curiosity plus the win condition, the roadmap is a set of
|
||||
**tracks** with dependencies, not a rigid sequence. Pick by interest; mind the
|
||||
prerequisites.
|
||||
|
||||
**Done:** UEFI boot, framebuffer + [serial](testing.md), [physical frames](frame-allocator.md)
|
||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||
own page tables — plus a [test harness](testing.md).
|
||||
|
||||
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||
higher-half kernel with a physmap (the low half is user space), per-process
|
||||
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||
`syscall`/`sysret`, a user-ELF loader, and `/system/services/init` running as a real
|
||||
preemptive ring-3 process (PID 1). Remaining polish: an address-space/stack
|
||||
reaper for exited tasks, SMAP + fault-recovering copy-in/out, the real IPC
|
||||
syscalls (IPC_Call/IPC_ReplyWait — they arrive with the second user server),
|
||||
and TLB shootdown once a process has more than one thread.*
|
||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||
See [resilience.md](resilience.md).
|
||||
- **ARM track** — the `aarch64` port so danos runs on the Zero 2 W and Pi 5. Largely
|
||||
independent of the others (it's the [arch layer](arch.md)); directly serves the win
|
||||
condition. Likely via aarch64-UEFI first (QEMU `virt` + AAVMF), then real boards.
|
||||
See [arm.md](arm.md), and [discovery.md](discovery.md) for the device tree it needs.
|
||||
- **GUI track** — a framebuffer-based windowing/compositor, and the input + display
|
||||
drivers under it. Builds on the neutral framebuffer (so it's arch-independent), and
|
||||
on the driver model from the isolation/resilience tracks. The visible payoff.
|
||||
|
||||
The natural spine is **isolation → (resilience + drivers) → GUI**, with the **ARM
|
||||
track** pursued alongside whenever the itch to see it boot on a Pi wins out.
|
||||
|
||||
## How to use this page
|
||||
|
||||
Read it before adding anything structural. When a design decision comes up, the
|
||||
question is: does it serve the **win condition** (runs on the three machines, with a
|
||||
GUI), or the **learning** (a constraint worth meeting)? If it serves neither — e.g.
|
||||
paying the full real-time tax with no payoff in sight — it can wait.
|
||||
+31
-27
@@ -108,7 +108,7 @@ localised (below).
|
||||
## The architecture decision: `runtime.os` + `runtime.fs`, and retire `posix`
|
||||
|
||||
danos already has the right split ([the private-ABI boundary](../README.md)): the
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](syscall.md)); the **`runtime`**
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](os-development/syscall.md)); the **`runtime`**
|
||||
library is the stable, danos-native application ABI. What this roadmap adds:
|
||||
|
||||
- **`runtime.os` — the seam.** A C-ABI-shaped module of the ~30 operations
|
||||
@@ -137,20 +137,22 @@ be. `runtime.os` is only the interim staging ground: developed against the stock
|
||||
toolchain so Phase 1 need not wait on the fork, then promoted near-verbatim into the
|
||||
fork's `std/os/danos.zig`.
|
||||
|
||||
### Retire `library/posix`
|
||||
### Retire `library/posix` (done)
|
||||
|
||||
The `posix` compatibility layer (`unistd`, `stdio`) was the right instinct too early.
|
||||
Its whole value is POSIX *spellings* for POSIX software — and danos has no POSIX
|
||||
software; every current caller is danos-native code that could use `runtime.fs`
|
||||
Its whole value was POSIX *spellings* for POSIX software — and danos has no POSIX
|
||||
software; every caller was danos-native code that could use `runtime.fs`
|
||||
directly. The real POSIX story arrives later and from elsewhere (musl, or upstream
|
||||
`std`'s own posix over `std.os.danos`), which supersedes a hand-rolled shim. So it is
|
||||
premature abstraction that adds a "which layer do I use?" fork with no payoff yet.
|
||||
`std`'s own posix over `std.os.danos`), which supersedes a hand-rolled shim. So it was
|
||||
premature abstraction that added a "which layer do I use?" fork with no payoff.
|
||||
|
||||
Its footprint is tiny: **five** call sites, all `unistd` file operations —
|
||||
Its footprint was tiny: **five** call sites, all `unistd` file operations —
|
||||
`system/services/fat/fat.zig` (`mount`), the `vfs-test` and `fat-test` clients, and
|
||||
(from the boot-log work) `init.zig` and `log-flush.zig`. `stdio.zig` is dead — nothing
|
||||
imports it. The plan: build `runtime.fs`, migrate those five to it, delete
|
||||
`library/posix/`, and drop the `posix` module from `build.zig`'s `addUserBinary`.
|
||||
(from the boot-log work) `init.zig` and `log-flush.zig`. `stdio.zig` was dead — nothing
|
||||
imported it. The plan — build `runtime.fs`, migrate those five to it, delete
|
||||
`library/posix/`, and drop the `posix` module from `build.zig`'s `addUserBinary` — has
|
||||
since been carried out: `library/` today holds only `mmio`, `runtime`, and
|
||||
`xkeyboard-config`.
|
||||
|
||||
## Where danos stands: coverage vs. the gaps
|
||||
|
||||
@@ -158,12 +160,12 @@ What the seam needs, and what danos already provides:
|
||||
|
||||
| std need | danos today | Gap |
|
||||
|----------|-------------|-----|
|
||||
| open / read / write / close / lseek | VFS (via the current `unistd`, → `runtime.fs`) | none — repackage |
|
||||
| open / read / write / close / lseek | VFS (via `runtime.fs`; the `unistd` shim is retired) | none — repackage |
|
||||
| directory read (`getdents`) | VFS `readdir` | none — repackage |
|
||||
| mmap / munmap | native syscalls ([abi.zig](../system/abi.zig)) | none |
|
||||
| page allocator | over `mmap`, via `root.os.heap.page_allocator` override | ~30-line hook |
|
||||
| monotonic clock | `clock` syscall | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](sysv.md)), `runtime.process.Init` | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](os-development/sysv.md)), `runtime.process.Init` | none |
|
||||
| stdout / stderr | `debug_write` today | wire fd 1/2 to a console **byte** stream |
|
||||
| mkdir / unlink / rename / truncate | done — engine + VFS + `runtime.fs` (Phase 2) | — |
|
||||
| stat fields | `{size, kind, mtime}` | **mode / inode** still missing (cache validity) |
|
||||
@@ -172,7 +174,7 @@ What the seam needs, and what danos already provides:
|
||||
| **cwd / chdir** | paths are absolute or bare | missing (no cwd anchor) |
|
||||
| **entropy / random** | — | missing (needed behind `vtable.random`) |
|
||||
| process spawn + exit status | `system_spawn` starts a *named ramdisk binary*; `ExitReason` is a *category* | no exec-of-path, no numeric `WEXITSTATUS` |
|
||||
| threads | one thread per process | avoided via `-fsingle-threaded` (below) |
|
||||
| threads | native threads — `thread_spawn`/futex/`thread_join` (`runtime.Thread`) | `std.Thread` seam unwritten; avoided via `-fsingle-threaded` (below) |
|
||||
| symlinks | `NodeKind` has the tag; unimplemented | low priority |
|
||||
|
||||
The clustering is clear: reads and memory are basically done; the real work is
|
||||
@@ -207,7 +209,7 @@ build); point danos's `build.zig`/CI at the resulting binary. Four localised pat
|
||||
plan9/serenity;
|
||||
- add `danos` to the freestanding/other **no-op `_start` list** in `std`'s `start.zig`,
|
||||
so std does *not* emit its own System-V `_start` — danos keeps owning the entry shim
|
||||
and `Init`/argv construction it already builds ([sysv.md](sysv.md));
|
||||
and `Init`/argv construction it already builds ([sysv.md](os-development/sysv.md));
|
||||
- wire the `system` selector `.danos => std.os.danos` in `std.posix`;
|
||||
- add `std/os/danos.zig` — **the seam itself**, promoted near-verbatim from the
|
||||
`runtime.os` developed first in Phase 1 (against the stock toolchain, so the fork is
|
||||
@@ -241,7 +243,7 @@ readdir/isatty/args/exit) exists. Those are downstream and out of scope here.
|
||||
danos's biggest genuine gap, and the correctness-critical one:
|
||||
|
||||
- Add **mkdir / unlink / rename / truncate** to *both* the VFS wire protocol
|
||||
([protocol.zig](../system/services/vfs/protocol.zig)) and the FAT engine
|
||||
([vfs-protocol.zig](../library/protocol/vfs/vfs-protocol.zig)) and the FAT engine
|
||||
([engine.zig](../system/services/fat/engine.zig)), then expose them via `runtime.os`.
|
||||
- Extend `stat` beyond `{size, kind}` to carry **mtime + inode + mode** — `std`'s file
|
||||
stat needs them for build-cache validity — which in turn needs **wall-clock** time
|
||||
@@ -270,8 +272,9 @@ fork):** the `runtime.os` seam, `cwd`, stdio-as-fds, and the compiler bring-up.
|
||||
Build the compiler with **two load-bearing flags**:
|
||||
|
||||
- **`-fsingle-threaded`** removes `std.Thread` entirely — `Thread.spawn` is a hard
|
||||
compile error under it, and `std.Io`'s threaded backend runs inline. danos being
|
||||
one-thread-per-process is therefore **not** a blocker. Parallel codegen is a
|
||||
compile error under it, and `std.Io`'s threaded backend runs inline. The unwritten
|
||||
`std.Thread` seam is therefore **not** a blocker (danos has native threads now —
|
||||
`thread_spawn`/futex — but the seam need not cover them yet). Parallel codegen is a
|
||||
throughput optimisation, not a correctness requirement.
|
||||
- **`-fno-llvm -fno-lld`** keeps codegen and linking **in-process** (the self-hosted
|
||||
x86-64 backend + self-linker), so a single `build-exe` **never forks a child**. That
|
||||
@@ -306,17 +309,18 @@ today), symlinks, and musl.
|
||||
`system_spawn` only starts a *named ramdisk binary*, not exec of an arbitrary path.
|
||||
Verify the self-hosted backend covers the target output before assuming child
|
||||
processes are optional.
|
||||
- **The shim cannot host the compiler.** danos's current `runtime`/`posix` is fine for
|
||||
- **The shim cannot host the compiler.** danos's current `runtime` is fine for
|
||||
danos's *own* native programs, but the compiler `import`s *upstream* `std`, which on
|
||||
a non-target hits the void `system` stub. So the compiler forces the real target
|
||||
(Phase 0's fork). Do not over-invest in extending the hand-shim for compiler
|
||||
purposes; put that effort into `runtime.os` + the VFS/FAT operations, which both the
|
||||
fork *and* a future musl consume.
|
||||
- **`"w"`/`O_CREAT` does not truncate — a silent-corruption bug on this road.** The FAT
|
||||
engine's `writeFile` only *grows* `node.size`, so overwriting a shorter file leaves
|
||||
trailing garbage. Harmless for the boot log today, but for a compiler it means
|
||||
**corrupt `.o`/cache files that look like nondeterministic compiler bugs.** Land
|
||||
`truncate` (Phase 2) before the compiler ever writes cache.
|
||||
- **`"w"`/`O_CREAT` not truncating was a silent-corruption bug on this road — fixed in
|
||||
Phase 2.** The FAT engine's `writeFile` only *grows* `node.size`, so overwriting a
|
||||
shorter file used to leave trailing garbage — for a compiler that means **corrupt
|
||||
`.o`/cache files that look like nondeterministic compiler bugs.** `engine.truncate`
|
||||
+ the O_TRUNC open flag (wired through the VFS protocol and `runtime.fs`) closed
|
||||
this before the compiler ever writes cache.
|
||||
- **Exit status is categorical, not numeric.** `process_exit_reason` returns an
|
||||
`ExitReason` *category*, not a numeric code (`WEXITSTATUS`). Fine while spawn is
|
||||
stubbed; the day `zig build` or external tools arrive, plan a kernel exit-record
|
||||
@@ -340,10 +344,10 @@ Two current decisions fall out of this roadmap:
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the north star this serves.
|
||||
- [syscall.md](syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md) — the
|
||||
- [syscall.md](os-development/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](os-development/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](device-driver-development/ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md) — the
|
||||
filesystem layout the file surface serves.
|
||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||
are confined, and now retired).
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
# /etc/devices.csv — the device→driver registry.
|
||||
#
|
||||
# The device manager reads this at boot and binds each device a bus driver
|
||||
# reports to the driver named here. It is AUTHORITATIVE: a device that no row
|
||||
# matches goes unbound (logged), never guessed. Edit this file to teach the
|
||||
# system new hardware — no recompile of the device manager required.
|
||||
#
|
||||
# One rule per line, nine comma-separated fields. '#' starts a comment
|
||||
# (whole-line or trailing); blank lines are ignored. Whitespace around a field
|
||||
# is trimmed, so columns may be padded for readability.
|
||||
#
|
||||
# bus which bus reported the device: pci | usb | acpi
|
||||
# base PCI base class / USB class (hex)
|
||||
# class PCI subclass / USB subclass (hex)
|
||||
# prog_if PCI prog-IF / USB protocol (hex)
|
||||
# vendor PCI vendor id / USB idVendor (hex)
|
||||
# device PCI device id / USB idProduct (hex)
|
||||
# subsystem PCI subsystem, packed (ssvid<<16)|ssid (hex)
|
||||
# hid ACPI _HID string (e.g. PNP0303); blank for pci/usb
|
||||
# driver full ramdisk path of the driver to spawn
|
||||
#
|
||||
# '*' or an empty field is a wildcard. When several rows match one device the
|
||||
# MOST SPECIFIC wins (pinning vendor/device/hid beats pinning only a class), so
|
||||
# a generic class rule and a precise vendor:device rule can coexist.
|
||||
#
|
||||
# bus base class prog_if vendor device subsystem hid driver
|
||||
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||
usb, 03, 01, 02, *, *, *, *, /system/drivers/usb-hid-mouse
|
||||
usb, 08, 06, 50, *, *, *, *, /system/drivers/usb-storage
|
||||
acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||
acpi, *, *, *, *, *, *, PNP0F13, /system/drivers/ps2-bus
|
||||
|
@@ -4,9 +4,9 @@
|
||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("display-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const display_protocol = @import("display-protocol");
|
||||
|
||||
/// The display's current mode, as `info()` reports it.
|
||||
pub const Info = struct {
|
||||
@@ -29,50 +29,50 @@ fn service() ?ipc.Handle {
|
||||
handle = h;
|
||||
return h;
|
||||
}
|
||||
system.sleep(50);
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Send one request, receive its reply; true on a zero status. `out` receives the reply
|
||||
/// so callers can read `info`/`layer` fields on success.
|
||||
fn transact(request: protocol.Request, out: *protocol.Reply) bool {
|
||||
fn transact(request: display_protocol.Request, out: *display_protocol.Reply) bool {
|
||||
const h = service() orelse return false;
|
||||
var req = request;
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var reply: [display_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&req), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
out.* = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
if (len < display_protocol.reply_size) return false;
|
||||
out.* = std.mem.bytesToValue(display_protocol.Reply, reply[0..display_protocol.reply_size]);
|
||||
return out.status == 0;
|
||||
}
|
||||
|
||||
/// The display's current mode, or null if the service never came up.
|
||||
pub fn info() ?Info {
|
||||
var reply: protocol.Reply = undefined;
|
||||
if (!transact(.{ .operation = @intFromEnum(protocol.Operation.info) }, &reply)) return null;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
if (!transact(.{ .operation = @intFromEnum(display_protocol.Operation.info) }, &reply)) return null;
|
||||
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||
}
|
||||
|
||||
/// Composite the dirty layers and flush the frame to the screen.
|
||||
pub fn present() bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.present) }, &reply);
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(display_protocol.Operation.present) }, &reply);
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = protocol.Mode;
|
||||
pub const Mode = display_protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const h = service() orelse return 0;
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.get_modes) };
|
||||
var reply: [protocol.modes_reply_size]u8 = undefined;
|
||||
var request = display_protocol.Request{ .operation = @intFromEnum(display_protocol.Operation.get_modes) };
|
||||
var reply: [display_protocol.modes_reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (len < protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(protocol.ModesReply, reply[0..protocol.modes_reply_size]);
|
||||
if (len < display_protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(display_protocol.ModesReply, reply[0..display_protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, protocol.max_modes), out.len);
|
||||
const count = @min(@min(answer.count, display_protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
@@ -80,8 +80,8 @@ pub fn modes(out: []Mode) usize {
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(display_protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
@@ -100,7 +100,7 @@ fn cachedInfo() ?Info {
|
||||
/// client packs colours through this so it never has to know the byte order itself.
|
||||
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||
const format = if (cachedInfo()) |i| i.format else 0;
|
||||
return protocol.pack(format, r, g, b);
|
||||
return display_protocol.pack(format, r, g, b);
|
||||
}
|
||||
|
||||
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||
@@ -111,9 +111,9 @@ pub const Layer = struct {
|
||||
|
||||
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.fill_rect),
|
||||
.operation = @intFromEnum(display_protocol.Operation.fill_rect),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
@@ -125,10 +125,10 @@ pub const Layer = struct {
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||
/// layer at (`x`, `y`). The tile rides inline in the request, so `w*h*4` must fit
|
||||
/// `protocol.maximum_payload`.
|
||||
/// `display_protocol.maximum_payload`.
|
||||
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
var request = protocol.Request{
|
||||
.operation = @intFromEnum(protocol.Operation.blit_tile),
|
||||
var request = display_protocol.Request{
|
||||
.operation = @intFromEnum(display_protocol.Operation.blit_tile),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
@@ -136,22 +136,22 @@ pub const Layer = struct {
|
||||
.height = h,
|
||||
};
|
||||
const header = std.mem.asBytes(&request);
|
||||
if (header.len + pixels.len > protocol.message_maximum) return false;
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
if (header.len + pixels.len > display_protocol.message_maximum) return false;
|
||||
var buffer: [display_protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(buffer[0..header.len], header);
|
||||
@memcpy(buffer[header.len..][0..pixels.len], pixels);
|
||||
const h_svc = service() orelse return false;
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var reply: [display_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h_svc, buffer[0 .. header.len + pixels.len], &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
if (len < display_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(display_protocol.Reply, reply[0..display_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Move / restack / show or hide the layer.
|
||||
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.configure_layer),
|
||||
.operation = @intFromEnum(display_protocol.Operation.configure_layer),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
@@ -163,9 +163,9 @@ pub const Layer = struct {
|
||||
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.damage),
|
||||
.operation = @intFromEnum(display_protocol.Operation.damage),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
@@ -176,17 +176,17 @@ pub const Layer = struct {
|
||||
|
||||
/// Release the layer and its surface.
|
||||
pub fn destroy(self: Layer) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(display_protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||
}
|
||||
};
|
||||
|
||||
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||
var reply: protocol.Reply = undefined;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
if (!transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.create_layer),
|
||||
.operation = @intFromEnum(display_protocol.Operation.create_layer),
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
@@ -23,26 +23,26 @@
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("input-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const input_protocol = @import("input-protocol");
|
||||
|
||||
pub const DeviceKind = protocol.DeviceKind;
|
||||
pub const InputEvent = protocol.InputEvent;
|
||||
pub const KeyEvent = protocol.KeyEvent;
|
||||
pub const MouseEvent = protocol.MouseEvent;
|
||||
pub const JoystickEvent = protocol.JoystickEvent;
|
||||
pub const EventKind = protocol.EventKind;
|
||||
pub const MouseEventKind = protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = protocol.JoystickEventKind;
|
||||
pub const Keycode = protocol.Keycode;
|
||||
pub const DeviceKind = input_protocol.DeviceKind;
|
||||
pub const InputEvent = input_protocol.InputEvent;
|
||||
pub const KeyEvent = input_protocol.KeyEvent;
|
||||
pub const MouseEvent = input_protocol.MouseEvent;
|
||||
pub const JoystickEvent = input_protocol.JoystickEvent;
|
||||
pub const EventKind = input_protocol.EventKind;
|
||||
pub const MouseEventKind = input_protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = input_protocol.JoystickEventKind;
|
||||
pub const Keycode = input_protocol.Keycode;
|
||||
|
||||
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||
/// input.device_mouse)`.
|
||||
pub const device_keyboard = protocol.device_keyboard;
|
||||
pub const device_mouse = protocol.device_mouse;
|
||||
pub const device_joystick = protocol.device_joystick;
|
||||
pub const device_all = protocol.device_all;
|
||||
pub const device_keyboard = input_protocol.device_keyboard;
|
||||
pub const device_mouse = input_protocol.device_mouse;
|
||||
pub const device_joystick = input_protocol.device_joystick;
|
||||
pub const device_all = input_protocol.device_all;
|
||||
|
||||
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||
@@ -51,7 +51,7 @@ fn lookupService() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.input)) |handle| return handle;
|
||||
system.sleep(50);
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -66,7 +66,7 @@ pub const Subscriber = struct {
|
||||
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||
/// was handed to the service as a capability at subscribe time).
|
||||
endpoint: ipc.Handle,
|
||||
receive: [protocol.event_size]u8 = undefined,
|
||||
receive: [input_protocol.event_size]u8 = undefined,
|
||||
|
||||
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||
@@ -74,8 +74,8 @@ pub const Subscriber = struct {
|
||||
/// (there should be none), so callers can loop.
|
||||
pub fn next(self: *Subscriber) ?InputEvent {
|
||||
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||
if (!got.isMessage() or got.len < protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..protocol.event_size]);
|
||||
if (!got.isMessage() or got.len < input_protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..input_protocol.event_size]);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -86,11 +86,11 @@ pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||
const service = lookupService() orelse return null;
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var request = input_protocol.Request{ .operation = @intFromEnum(input_protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [input_protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status != 0) return null;
|
||||
if (result.len < input_protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(input_protocol.Reply, reply[0..input_protocol.reply_size]).status != 0) return null;
|
||||
return .{ .endpoint = endpoint };
|
||||
}
|
||||
|
||||
@@ -149,11 +149,11 @@ pub const Publisher = struct {
|
||||
service: ipc.Handle,
|
||||
|
||||
fn publish(self: Publisher, event: InputEvent) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.publish), .event = event };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var request = input_protocol.Request{ .operation = @intFromEnum(input_protocol.Operation.publish), .event = event };
|
||||
var reply: [input_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
if (len < input_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(input_protocol.Reply, reply[0..input_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||
@@ -204,8 +204,8 @@ pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = input_protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = input_protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = input_protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
@@ -8,9 +8,9 @@
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("block-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
@@ -19,11 +19,11 @@ pub const Device = struct {
|
||||
|
||||
/// The device's block size and total block count.
|
||||
pub fn geometry(self: Device) ?Geometry {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const result = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
if (n < block_protocol.reply_size) return null;
|
||||
const result = std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]);
|
||||
if (result.status != 0) return null;
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
@@ -45,15 +45,22 @@ pub const Device = struct {
|
||||
return self.transfer(.flush, 0, 0, 0);
|
||||
}
|
||||
|
||||
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
fn transfer(self: Device, operation: block_protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (n < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
if (n < block_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// One lookup attempt, no waiting — for a server that retries on its own
|
||||
/// timer (the fat service) instead of blocking its harness in here.
|
||||
pub fn tryOpen() ?Device {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Look up the block device, retrying generously while the USB storage chain
|
||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||
pub fn open() ?Device {
|
||||
@@ -61,9 +68,12 @@ pub fn open() ?Device {
|
||||
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||
// tens of seconds under emulation.
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 1200) : (attempts += 1) {
|
||||
// 30 s covers the slowest observed healthy chain (a flaky QEMU enumeration
|
||||
// completed at ~24 s); a machine whose stick genuinely failed setup should
|
||||
// not sit a further minute pretending otherwise.
|
||||
while (attempts < 600) : (attempts += 1) {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
system.sleep(50);
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -1,12 +1,15 @@
|
||||
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||
//! mapping registers or routing an IRQ.
|
||||
//! library/device/driver — the driver author's interface: enumerate the kernel's device
|
||||
//! table, claim a device, map its MMIO, bind its interrupt (the claim is the capability the
|
||||
//! kernel checks before mapping registers or routing an IRQ), and say `hello` to the device
|
||||
//! manager at startup. The whole kernel + manager surface a driver needs, in one import.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
|
||||
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||
@@ -129,3 +132,42 @@ pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- device-manager handshake (folded in from the former device-manager.zig) ---
|
||||
|
||||
/// What kind of driver is announcing itself (a bus that reports children, or a leaf
|
||||
/// device). Re-exported so callers name it without importing the protocol.
|
||||
pub const Role = device_manager_protocol.Role;
|
||||
|
||||
const lookup_attempts: u32 = 100;
|
||||
const lookup_pause_ms: u64 = 20;
|
||||
|
||||
/// Say hello to the device manager and return its endpoint, or null if there is no manager
|
||||
/// (best-effort standalone bring-up) or it refused the handshake. Bus drivers keep the handle
|
||||
/// to report children through; a driver that runs fine unsupervised discards it with `_ =`,
|
||||
/// and one that requires supervision bails on null. Logs the outcome itself.
|
||||
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
||||
var attempts: u32 = 0;
|
||||
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||
time.sleepMillis(lookup_pause_ms);
|
||||
} else {
|
||||
std.log.info("no device manager to hello", .{});
|
||||
return null;
|
||||
};
|
||||
|
||||
const message = device_manager_protocol.Hello{ .role = @intFromEnum(role), .device_id = device_id };
|
||||
var reply: [device_manager_protocol.reply_size]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&message), &reply) catch {
|
||||
std.log.info("hello call failed", .{});
|
||||
return null;
|
||||
};
|
||||
if (length < device_manager_protocol.reply_size or
|
||||
std.mem.bytesToValue(device_manager_protocol.HelloReply, reply[0..device_manager_protocol.reply_size]).status != 0)
|
||||
{
|
||||
std.log.info("hello refused", .{});
|
||||
return null;
|
||||
}
|
||||
std.log.info("hello acknowledged", .{});
|
||||
return manager;
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||
//! /lib/device/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
|
||||
//!
|
||||
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
|
||||
@@ -11,13 +11,13 @@
|
||||
//! doorbell.* = i; // volatile store to UC MMIO
|
||||
//! // nothing orders these; the device can read a stale descriptor
|
||||
//!
|
||||
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
|
||||
//! reason they are a named primitive and not scattered `asm volatile`:
|
||||
//! Put a `writeMemoryBarrier()` between them. The barriers lower per-architecture — which
|
||||
//! is the whole reason they are a named primitive and not scattered `asm volatile`:
|
||||
//!
|
||||
//! x86_64 aarch64
|
||||
//! mb() mfence dsb sy
|
||||
//! rmb() lfence dsb ld
|
||||
//! wmb() sfence dsb st
|
||||
//! x86_64 aarch64
|
||||
//! memoryBarrier() mfence dsb sy
|
||||
//! readMemoryBarrier() lfence dsb ld
|
||||
//! writeMemoryBarrier() sfence dsb st
|
||||
//!
|
||||
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
|
||||
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
|
||||
@@ -29,52 +29,52 @@ const builtin = @import("builtin");
|
||||
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
|
||||
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
|
||||
/// another volatile access.
|
||||
pub inline fn read(comptime T: type, addr: usize) T {
|
||||
pub inline fn readRegister(comptime T: type, addr: usize) T {
|
||||
return @as(*const volatile T, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
|
||||
pub inline fn write(comptime T: type, addr: usize, value: T) void {
|
||||
pub inline fn writeRegister(comptime T: type, addr: usize, value: T) void {
|
||||
@as(*volatile T, @ptrFromInt(addr)).* = value;
|
||||
}
|
||||
|
||||
/// Full barrier: all loads and stores before it are globally visible before any after
|
||||
/// it. Use when an MMIO write must complete before a following read.
|
||||
pub inline fn mb() void {
|
||||
/// Full memory barrier: all loads and stores before it are globally visible before any
|
||||
/// after it. Use when an MMIO write must complete before a following read.
|
||||
pub inline fn memoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.mb: unsupported architecture"),
|
||||
else => @compileError("mmio.memoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
|
||||
/// Read memory barrier: loads before it complete before loads after it. Use after an IRQ
|
||||
/// wake, before reading what the device wrote to shared memory.
|
||||
pub inline fn rmb() void {
|
||||
pub inline fn readMemoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.rmb: unsupported architecture"),
|
||||
else => @compileError("mmio.readMemoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Write barrier: stores before it become visible before stores after it. Use between
|
||||
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||
pub inline fn wmb() void {
|
||||
/// Write memory barrier: stores before it become visible before stores after it. Use
|
||||
/// between filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||
pub inline fn writeMemoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.wmb: unsupported architecture"),
|
||||
else => @compileError("mmio.writeMemoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
test "barriers emit and registers round-trip through a RAM cell" {
|
||||
// The barriers must at least assemble for the host arch; ordering can't be unit
|
||||
// tested, but a missing/mistyped mnemonic is caught here.
|
||||
wmb();
|
||||
rmb();
|
||||
mb();
|
||||
writeMemoryBarrier();
|
||||
readMemoryBarrier();
|
||||
memoryBarrier();
|
||||
var cell: u64 = 0;
|
||||
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
|
||||
writeRegister(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), readRegister(u64, @intFromPtr(&cell)));
|
||||
}
|
||||
@@ -6,7 +6,7 @@
|
||||
//! by name, and neither reaches into the other's files.
|
||||
//!
|
||||
//! It is also the **single source of truth** for `DeviceClass` and `ResourceKind`:
|
||||
//! the kernel's rich, pointer-based device tree (system/devices/device-model.zig,
|
||||
//! the kernel's rich, pointer-based device tree (system/kernel/device-model.zig,
|
||||
//! which user space must never import) re-exports these, so the enum that a driver
|
||||
//! matches on and the enum the kernel classifies with are the *same* type — no
|
||||
//! hand-kept "mirror in order" to drift. The core kernel↔user ABI is [[abi]]; the
|
||||
@@ -97,6 +97,7 @@ pub const DisplayInfo = extern struct {
|
||||
height: u32 = 0, // visible rows
|
||||
pitch: u32 = 0, // bytes from one row's start to the next
|
||||
format: u32 = 0, // a DisplayFormat value
|
||||
refresh_hz: u32 = 0, // panel refresh rate from EDID (0 = unknown); see boot-handoff
|
||||
};
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||
@@ -124,6 +125,15 @@ pub const DeviceDescriptor = extern struct {
|
||||
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
||||
// names with the pci-class module.
|
||||
pci_class: u64,
|
||||
// Numeric identity beyond the class triple, mirrored in the bus report's
|
||||
// ChildAdded so /etc/devices.csv can bind on it: `vendor`/`device` are the PCI
|
||||
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
||||
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
||||
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
||||
// out identically until they choose to set them.
|
||||
vendor: u16 = 0,
|
||||
device: u16 = 0,
|
||||
subsystem: u32 = 0,
|
||||
hid_len: u64,
|
||||
resource_count: u64,
|
||||
hid: [8]u8,
|
||||
@@ -41,6 +41,35 @@ pub const ClassCode = struct {
|
||||
}
|
||||
};
|
||||
|
||||
// --- Configuration-space layout ---------------------------------------------------------
|
||||
// The offsets and bit layouts of the PCI configuration header (PCI spec; see
|
||||
// https://wiki.osdev.org/PCI). Pure data — named here so both a device driver's view of
|
||||
// its own claimed function (library/device/pci/pci.zig) and the bus enumerator name the
|
||||
// same bytes instead of scattering bare 0x04/0x34/0xFFFF_FFF0 magic across the tree.
|
||||
|
||||
/// Header field offsets (byte offsets into the 256-byte configuration space).
|
||||
pub const config_vendor_id: usize = 0x00;
|
||||
pub const config_device_id: usize = 0x02;
|
||||
pub const config_command: usize = 0x04;
|
||||
pub const config_status: usize = 0x06;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4
|
||||
|
||||
/// Command register: Memory-Space enable (bit 1) | Bus-Master enable (bit 2).
|
||||
pub const command_memory_and_bus_master: u16 = 0x06;
|
||||
/// Status register bit 4: a capability list is present at config_capabilities_pointer.
|
||||
pub const status_capabilities_list: u16 = 0x10;
|
||||
/// Capability pointers are dword-aligned; the low two bits are reserved.
|
||||
pub const capability_pointer_mask: u8 = 0xFC;
|
||||
|
||||
/// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1
|
||||
/// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is
|
||||
/// the dword with the low 4 flag bits masked off.
|
||||
pub const bar_io_space: u32 = 0x1;
|
||||
pub const bar_type_mask: u32 = 0x6;
|
||||
pub const bar_type_64bit: u32 = 0x4;
|
||||
pub const bar_memory_base_mask: u32 = 0xFFFF_FFF0;
|
||||
|
||||
/// Base class (config byte 0x0B). Non-exhaustive: an unlisted code is a real but
|
||||
/// unnamed class, decoded as "Unknown" rather than rejected.
|
||||
pub const BaseClass = enum(u8) {
|
||||
@@ -0,0 +1,108 @@
|
||||
//! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has
|
||||
//! claimed. Config space is mapped as resource 0; this gives header-field accessors, BAR
|
||||
//! decode + map, and a capability-list iterator, so a driver never re-derives the
|
||||
//! config-space layout by hand.
|
||||
//!
|
||||
//! This is the *device-owned* view: read my own function's live config, map my own BARs.
|
||||
//! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing
|
||||
//! their BARs — is a different mechanism and lives in the pci-bus driver. The pure
|
||||
//! config-space layout both need (offsets, BAR bit fields) is named once in the `pci-class`
|
||||
//! data module; this logic module adds the parts that need `mmio` + the `driver` client.
|
||||
|
||||
const std = @import("std");
|
||||
const mmio = @import("mmio");
|
||||
const pci_class = @import("pci-class");
|
||||
const device = @import("driver");
|
||||
|
||||
/// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor`
|
||||
/// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole
|
||||
/// bring-up. Header reads and the capability walk hit live config space; `mapBar` caches.
|
||||
pub const Function = struct {
|
||||
device_id: u64,
|
||||
descriptor: *const device.DeviceDescriptor,
|
||||
config: usize, // virtual base of mapped resource 0
|
||||
bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 }, // per-BAR mmio_map cache
|
||||
|
||||
/// Map config space (resource 0) of the already-claimed `device_id`. null if the map
|
||||
/// fails (not claimed, or no config resource).
|
||||
pub fn map(device_id: u64, descriptor: *const device.DeviceDescriptor) ?Function {
|
||||
const base = device.mmioMap(device_id, 0) orelse return null;
|
||||
return .{ .device_id = device_id, .descriptor = descriptor, .config = base };
|
||||
}
|
||||
|
||||
pub fn vendorId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_vendor_id);
|
||||
}
|
||||
pub fn deviceId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_device_id);
|
||||
}
|
||||
pub fn command(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_command);
|
||||
}
|
||||
pub fn status(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_status);
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master enable in the command register. Firmware often leaves
|
||||
/// a secondary display's decode off; a bus-mastering device must enable both.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | pci_class.command_memory_and_bus_master);
|
||||
}
|
||||
|
||||
/// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs,
|
||||
/// combine the high dword for a 64-bit BAR, mask the base, then correlate that physical
|
||||
/// base with one of the descriptor's memory resources and `mmio_map` it — a BAR names a
|
||||
/// *number*, while `mmio_map` takes a *resource index*, and gaps/config-space shift the
|
||||
/// numbering. Cached per BAR. null if the BAR is I/O-space or is not a mapped resource.
|
||||
pub fn mapBar(self: *Function, bar: u8) ?usize {
|
||||
if (bar >= 6) return null;
|
||||
if (self.bar_virtual[bar] != 0) return self.bar_virtual[bar];
|
||||
|
||||
const low = mmio.readRegister(u32, self.config + pci_class.config_bar0 + @as(usize, bar) * 4);
|
||||
if (low & pci_class.bar_io_space != 0) return null; // an I/O-space BAR
|
||||
var base: u64 = low & pci_class.bar_memory_base_mask;
|
||||
if ((low & pci_class.bar_type_mask) == pci_class.bar_type_64bit) { // 64-bit: high half is the next dword
|
||||
const high = mmio.readRegister(u32, self.config + pci_class.config_bar0 + (@as(usize, bar) + 1) * 4);
|
||||
base |= @as(u64, high) << 32;
|
||||
}
|
||||
|
||||
for (self.descriptor.resources[0..@intCast(self.descriptor.resource_count)], 0..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and resource.start == base) {
|
||||
const v = device.mmioMap(self.device_id, index) orelse return null;
|
||||
self.bar_virtual[bar] = v;
|
||||
return v;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Iterate the capability list. Empty when the function advertises none.
|
||||
pub fn capabilities(self: *const Function) CapabilityIterator {
|
||||
const present = self.status() & pci_class.status_capabilities_list != 0;
|
||||
const first = if (present)
|
||||
mmio.readRegister(u8, self.config + pci_class.config_capabilities_pointer) & pci_class.capability_pointer_mask
|
||||
else
|
||||
0;
|
||||
return .{ .config = self.config, .cursor = first };
|
||||
}
|
||||
};
|
||||
|
||||
/// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the
|
||||
/// caller reads its body with `mmio.readRegister(T, cap.offset + n)`.
|
||||
pub const Capability = struct { id: u8, offset: usize };
|
||||
|
||||
pub const CapabilityIterator = struct {
|
||||
config: usize,
|
||||
cursor: u8,
|
||||
guard: u32 = 0, // bounds a malformed/looping chain (48 = the 256-byte space in dwords)
|
||||
|
||||
pub fn next(self: *CapabilityIterator) ?Capability {
|
||||
if (self.cursor == 0 or self.guard >= 48) return null;
|
||||
self.guard += 1;
|
||||
const at = self.config + self.cursor;
|
||||
const id = mmio.readRegister(u8, at + 0);
|
||||
self.cursor = mmio.readRegister(u8, at + 1) & pci_class.capability_pointer_mask;
|
||||
return .{ .id = id, .offset = at };
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,348 @@
|
||||
//! The device registry: parse `/etc/devices.csv` into match rules and bind a
|
||||
//! reported device to a driver. This is the data-driven replacement for the
|
||||
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
||||
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
||||
//! — a device that no row matches goes unbound (logged), never guessed.
|
||||
//!
|
||||
//! Pure logic: no hardware access, no syscalls, no allocator. `parse` fills a
|
||||
//! caller-provided `[]Rule` whose string fields (`hid`, `driver`) are slices
|
||||
//! *into the CSV source*, so the source buffer must outlive the rules (the
|
||||
//! manager holds it in a static buffer for the life of the process — zero-copy).
|
||||
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
||||
//!
|
||||
//! The file format (docs/device-driver-development/device-manager.md, and the
|
||||
//! `/etc/devices.csv` header itself): one rule per line, nine comma-separated
|
||||
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
||||
//!
|
||||
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||
//!
|
||||
//! `bus` is `pci`/`usb`/`acpi`; the numeric fields are hex (with or without a
|
||||
//! `0x` prefix); `*` or an empty field is a wildcard (matches anything). For PCI
|
||||
//! the class triple is base/subclass/prog-IF; for USB it is class/subclass/
|
||||
//! protocol with vendor/device the idVendor/idProduct; ACPI matches on `hid`
|
||||
//! (e.g. "PNP0303") with the triple left blank. `driver` is a full ramdisk path.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Which bus a rule or a reported device belongs to. `unknown` is what an
|
||||
/// unrecognised `bus` token parses to — such a rule never matches (its bus
|
||||
/// equals no real device's), so a typo fails safe rather than binding wrongly.
|
||||
pub const Bus = enum {
|
||||
pci,
|
||||
usb,
|
||||
acpi,
|
||||
unknown,
|
||||
|
||||
pub fn fromToken(token: []const u8) Bus {
|
||||
if (std.mem.eql(u8, token, "pci")) return .pci;
|
||||
if (std.mem.eql(u8, token, "usb")) return .usb;
|
||||
if (std.mem.eql(u8, token, "acpi")) return .acpi;
|
||||
return .unknown;
|
||||
}
|
||||
};
|
||||
|
||||
/// A reported device's full identity, as the manager assembles it from a
|
||||
/// `child_added`: the bus-native class triple plus the numeric ids the widened
|
||||
/// ABI now carries, or the ACPI `_HID` string. Fields a given bus does not have
|
||||
/// are zero / empty (a PCI function has no `hid`; an ACPI device has no vendor).
|
||||
pub const Identity = struct {
|
||||
bus: Bus,
|
||||
base: u8 = 0,
|
||||
subclass: u8 = 0,
|
||||
prog_if: u8 = 0,
|
||||
vendor: u16 = 0,
|
||||
device: u16 = 0,
|
||||
subsystem: u32 = 0,
|
||||
hid: []const u8 = "",
|
||||
};
|
||||
|
||||
/// One parsed registry row. A `null` field is a wildcard — it matches any value
|
||||
/// and contributes nothing to specificity. String fields point into the CSV
|
||||
/// source that was parsed (see the module doc).
|
||||
pub const Rule = struct {
|
||||
bus: Bus,
|
||||
base: ?u8 = null,
|
||||
subclass: ?u8 = null,
|
||||
prog_if: ?u8 = null,
|
||||
vendor: ?u16 = null,
|
||||
device: ?u16 = null,
|
||||
subsystem: ?u32 = null,
|
||||
hid: ?[]const u8 = null,
|
||||
driver: []const u8,
|
||||
};
|
||||
|
||||
/// Specificity weights: how much each pinned field counts toward "most specific
|
||||
/// wins". Doubling from the coarsest (`base`) so that each level outweighs *all*
|
||||
/// coarser levels combined (1+2+4+8+16 = 31 < 32) — a rule that pins `device`
|
||||
/// always beats any rule that does not, no matter how many coarse fields the
|
||||
/// latter pins. `hid` and `device` share the top tier (the user's "hid and
|
||||
/// device weigh heaviest"); they never co-occur, since `hid` is ACPI-only and
|
||||
/// `device` is a PCI/USB numeric id.
|
||||
const weight_base: u32 = 1;
|
||||
const weight_subclass: u32 = 2;
|
||||
const weight_prog_if: u32 = 4;
|
||||
const weight_vendor: u32 = 8;
|
||||
const weight_subsystem: u32 = 16;
|
||||
const weight_device: u32 = 32;
|
||||
const weight_hid: u32 = 32;
|
||||
|
||||
/// The outcome of `matchDriver`: the winning rule's driver path, its specificity,
|
||||
/// and whether another rule tied it at that specificity. `ambiguous` is a
|
||||
/// registry authoring error (two equally-specific rules claiming one device); the
|
||||
/// manager logs it loudly and binds the first, so a shadowed rule is visible
|
||||
/// rather than silently dropped.
|
||||
pub const Match = struct {
|
||||
driver: []const u8,
|
||||
specificity: u32,
|
||||
ambiguous: bool,
|
||||
};
|
||||
|
||||
/// Whether `rule` matches `id`: same bus, and every pinned (non-wildcard) field
|
||||
/// equal. `hid` compares as a string; the rest as integers.
|
||||
fn matches(rule: Rule, id: Identity) bool {
|
||||
if (rule.bus != id.bus) return false;
|
||||
if (rule.base) |b| if (b != id.base) return false;
|
||||
if (rule.subclass) |s| if (s != id.subclass) return false;
|
||||
if (rule.prog_if) |p| if (p != id.prog_if) return false;
|
||||
if (rule.vendor) |v| if (v != id.vendor) return false;
|
||||
if (rule.device) |d| if (d != id.device) return false;
|
||||
if (rule.subsystem) |s| if (s != id.subsystem) return false;
|
||||
if (rule.hid) |h| if (!std.mem.eql(u8, h, id.hid)) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The specificity score of a rule — the sum of the weights of its pinned fields.
|
||||
fn specificity(rule: Rule) u32 {
|
||||
var score: u32 = 0;
|
||||
if (rule.base != null) score += weight_base;
|
||||
if (rule.subclass != null) score += weight_subclass;
|
||||
if (rule.prog_if != null) score += weight_prog_if;
|
||||
if (rule.vendor != null) score += weight_vendor;
|
||||
if (rule.device != null) score += weight_device;
|
||||
if (rule.subsystem != null) score += weight_subsystem;
|
||||
if (rule.hid != null) score += weight_hid;
|
||||
return score;
|
||||
}
|
||||
|
||||
/// Bind a reported device to a driver: of every rule that matches `id`, return
|
||||
/// the most specific. `null` when nothing matches (the device goes unbound —
|
||||
/// the authoritative registry does not guess). On an exact specificity tie the
|
||||
/// first such rule in file order wins and `ambiguous` is set.
|
||||
pub fn matchDriver(rules: []const Rule, id: Identity) ?Match {
|
||||
var best: ?Match = null;
|
||||
for (rules) |rule| {
|
||||
if (!matches(rule, id)) continue;
|
||||
const score = specificity(rule);
|
||||
if (best) |current| {
|
||||
if (score > current.specificity) {
|
||||
best = .{ .driver = rule.driver, .specificity = score, .ambiguous = false };
|
||||
} else if (score == current.specificity) {
|
||||
// Two equally-specific rules claim this device — keep the first,
|
||||
// flag the ambiguity for the manager to log.
|
||||
best.?.ambiguous = true;
|
||||
}
|
||||
} else {
|
||||
best = .{ .driver = rule.driver, .specificity = score, .ambiguous = false };
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
// --- parsing -----------------------------------------------------------------
|
||||
|
||||
/// What one CSV line parsed to. `malformed` is a non-comment, non-blank line the
|
||||
/// parser could not read (wrong field count, unparsable number, empty driver) —
|
||||
/// the manager counts these and logs, so a broken registry is loud, not silent.
|
||||
const Line = union(enum) {
|
||||
rule: Rule,
|
||||
ignorable, // blank or comment
|
||||
malformed,
|
||||
};
|
||||
|
||||
/// The result of `parse`: how many rules landed in the caller's buffer, and how
|
||||
/// many non-ignorable lines were malformed (for the manager to log). `truncated`
|
||||
/// is set if there were more valid rules than the buffer could hold.
|
||||
pub const ParseResult = struct {
|
||||
count: usize,
|
||||
malformed: usize,
|
||||
truncated: bool,
|
||||
};
|
||||
|
||||
/// Strip a trailing `#` comment and surrounding whitespace from one raw line.
|
||||
fn stripComment(raw: []const u8) []const u8 {
|
||||
const body = if (std.mem.indexOfScalar(u8, raw, '#')) |hash| raw[0..hash] else raw;
|
||||
return std.mem.trim(u8, body, " \t\r\n");
|
||||
}
|
||||
|
||||
/// Parse one hex field into `T`, honouring `*`/empty as a wildcard (`null`) and
|
||||
/// an optional `0x` prefix. Returns an error only for a genuinely unparsable
|
||||
/// non-wildcard token, so the caller can mark the whole line malformed.
|
||||
fn parseHexField(comptime T: type, field: []const u8) !?T {
|
||||
const token = std.mem.trim(u8, field, " \t");
|
||||
if (token.len == 0 or std.mem.eql(u8, token, "*")) return null;
|
||||
const digits = if (std.mem.startsWith(u8, token, "0x") or std.mem.startsWith(u8, token, "0X"))
|
||||
token[2..]
|
||||
else
|
||||
token;
|
||||
return try std.fmt.parseInt(T, digits, 16);
|
||||
}
|
||||
|
||||
/// Parse a wildcard-or-string field (the `hid` column): `*`/empty → wildcard.
|
||||
fn parseStringField(field: []const u8) ?[]const u8 {
|
||||
const token = std.mem.trim(u8, field, " \t");
|
||||
if (token.len == 0 or std.mem.eql(u8, token, "*")) return null;
|
||||
return token;
|
||||
}
|
||||
|
||||
/// Classify and (if a rule) parse one line. Split out from `parse` so it can be
|
||||
/// unit-tested directly. `line` is the raw line including no newline.
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
|
||||
// Nine comma-separated fields: bus, base, class, prog_if, vendor, device,
|
||||
// subsystem, hid, driver.
|
||||
var fields: [9][]const u8 = undefined;
|
||||
var count: usize = 0;
|
||||
var it = std.mem.splitScalar(u8, body, ',');
|
||||
while (it.next()) |field| {
|
||||
if (count >= fields.len) return .malformed; // too many columns
|
||||
fields[count] = field;
|
||||
count += 1;
|
||||
}
|
||||
if (count != fields.len) return .malformed; // too few columns
|
||||
|
||||
const bus = Bus.fromToken(std.mem.trim(u8, fields[0], " \t"));
|
||||
if (bus == .unknown) return .malformed;
|
||||
|
||||
const driver = std.mem.trim(u8, fields[8], " \t");
|
||||
if (driver.len == 0) return .malformed;
|
||||
|
||||
return .{ .rule = .{
|
||||
.bus = bus,
|
||||
.base = parseHexField(u8, fields[1]) catch return .malformed,
|
||||
.subclass = parseHexField(u8, fields[2]) catch return .malformed,
|
||||
.prog_if = parseHexField(u8, fields[3]) catch return .malformed,
|
||||
.vendor = parseHexField(u16, fields[4]) catch return .malformed,
|
||||
.device = parseHexField(u16, fields[5]) catch return .malformed,
|
||||
.subsystem = parseHexField(u32, fields[6]) catch return .malformed,
|
||||
.hid = parseStringField(fields[7]),
|
||||
.driver = driver,
|
||||
} };
|
||||
}
|
||||
|
||||
/// Parse a whole `/etc/devices.csv` into `out_rules`. The string fields of the
|
||||
/// returned rules point into `source`, which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.rule => |rule| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = rule;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// The worked example from the design: a specific virtio-gpu rule (pins vendor +
|
||||
// device) and a generic display rule (class only) both match the virtio card;
|
||||
// the specific one must win. And a plain VGA adapter still falls to the generic
|
||||
// rule. This is the whole point of widening the ABI to carry vendor/device.
|
||||
test "virtio device rule beats the generic display rule" {
|
||||
const csv =
|
||||
\\# bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display
|
||||
\\pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(csv, &rules);
|
||||
try testing.expectEqual(@as(usize, 2), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
|
||||
// The virtio-gpu function: display / other, vendor 1AF4 device 1050.
|
||||
const virtio = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x80, .prog_if = 0x00,
|
||||
.vendor = 0x1AF4, .device = 0x1050,
|
||||
}).?;
|
||||
try testing.expect(!virtio.ambiguous);
|
||||
try testing.expectEqualStrings("/system/drivers/virtio-gpu", virtio.driver);
|
||||
|
||||
// A plain VGA adapter (display / VGA) still binds the generic display driver.
|
||||
const vga = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x00, .prog_if = 0x00,
|
||||
.vendor = 0x1234, .device = 0x1111,
|
||||
}).?;
|
||||
try testing.expectEqualStrings("/system/drivers/display", vga.driver);
|
||||
}
|
||||
|
||||
test "no matching row leaves the device unbound" {
|
||||
const csv = "pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus\n";
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(csv, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
|
||||
// An AHCI controller (mass storage / SATA / AHCI) has no row — unbound.
|
||||
const unmatched = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x01, .subclass = 0x06, .prog_if = 0x01,
|
||||
});
|
||||
try testing.expect(unmatched == null);
|
||||
}
|
||||
|
||||
test "acpi rows match on hid" {
|
||||
const csv =
|
||||
\\acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||
\\acpi, *, *, *, *, *, *, PNP0F13, /system/drivers/ps2-bus
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(csv, &rules);
|
||||
try testing.expectEqual(@as(usize, 2), parsed.count);
|
||||
|
||||
const keyboard = matchDriver(rules[0..parsed.count], .{ .bus = .acpi, .hid = "PNP0303" }).?;
|
||||
try testing.expectEqualStrings("/system/drivers/ps2-bus", keyboard.driver);
|
||||
const nothing = matchDriver(rules[0..parsed.count], .{ .bus = .acpi, .hid = "PNP0A03" });
|
||||
try testing.expect(nothing == null);
|
||||
}
|
||||
|
||||
test "equally specific rules flag ambiguity" {
|
||||
const csv =
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display-a
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display-b
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(csv, &rules);
|
||||
const hit = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x00, .prog_if = 0x00,
|
||||
}).?;
|
||||
try testing.expect(hit.ambiguous);
|
||||
try testing.expectEqualStrings("/system/drivers/display-a", hit.driver); // first wins
|
||||
}
|
||||
|
||||
test "comments, blanks, and malformed lines" {
|
||||
const csv =
|
||||
\\# a header comment
|
||||
\\
|
||||
\\pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus # trailing comment
|
||||
\\pci, ZZ, 03, 30, *, *, *, *, /system/drivers/broken
|
||||
\\pci, 03, 00, 00, *, *, *, *,
|
||||
\\bogus-bus, *, *, *, *, *, *, *, /system/drivers/x
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(csv, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the xhci row is valid
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // bad hex, empty driver, bad bus
|
||||
try testing.expectEqualStrings("/system/drivers/usb-xhci-bus", rules[0].driver);
|
||||
try testing.expect(rules[0].hid == null); // trailing comment stripped, hid still wildcard
|
||||
}
|
||||
@@ -70,6 +70,104 @@ pub const Class = enum(u8) {
|
||||
_,
|
||||
};
|
||||
|
||||
/// A human-readable name for a device/interface class code, for logs. Unknown
|
||||
/// codes fall through to "class 0xNN".
|
||||
pub fn className(class: u8) []const u8 {
|
||||
return switch (@as(Class, @enumFromInt(class))) {
|
||||
.per_interface => "per-interface",
|
||||
.audio => "Audio",
|
||||
.communications => "Communications",
|
||||
.hid => "HID",
|
||||
.physical => "Physical",
|
||||
.image => "Image",
|
||||
.printer => "Printer",
|
||||
.mass_storage => "Mass Storage",
|
||||
.hub => "Hub",
|
||||
.cdc_data => "CDC Data",
|
||||
.smart_card => "Smart Card",
|
||||
.content_security => "Content Security",
|
||||
.video => "Video",
|
||||
.personal_healthcare => "Personal Healthcare",
|
||||
.audio_video => "Audio/Video",
|
||||
.billboard => "Billboard",
|
||||
.type_c_bridge => "Type-C Bridge",
|
||||
.bulk_display => "Bulk Display",
|
||||
.mctp => "MCTP",
|
||||
.i3c => "I3C",
|
||||
.diagnostic => "Diagnostic",
|
||||
.wireless_controller => "Wireless Controller",
|
||||
.miscellaneous => "Miscellaneous",
|
||||
.application_specific => "Application-specific",
|
||||
.vendor_specific => "Vendor-specific",
|
||||
_ => "Unknown",
|
||||
};
|
||||
}
|
||||
|
||||
/// The USB speed class (as xHCI reports it in PORTSC/slot contexts) named.
|
||||
pub fn speedName(speed: u32) []const u8 {
|
||||
return switch (speed) {
|
||||
1 => "Full-speed",
|
||||
2 => "Low-speed",
|
||||
3 => "High-speed",
|
||||
4 => "SuperSpeed",
|
||||
5 => "SuperSpeedPlus",
|
||||
else => "unknown-speed",
|
||||
};
|
||||
}
|
||||
|
||||
/// A USB3 Port Link State (xHCI PORTSC PLS field) named.
|
||||
pub fn linkStateName(pls: u32) []const u8 {
|
||||
return switch (pls) {
|
||||
0 => "U0",
|
||||
1 => "U1",
|
||||
2 => "U2",
|
||||
3 => "U3-suspended",
|
||||
4 => "Disabled",
|
||||
5 => "RxDetect",
|
||||
6 => "Inactive",
|
||||
7 => "Polling",
|
||||
8 => "Recovery",
|
||||
9 => "HotReset",
|
||||
10 => "Compliance",
|
||||
11 => "Test",
|
||||
15 => "Resume",
|
||||
else => "reserved",
|
||||
};
|
||||
}
|
||||
|
||||
/// The most useful readable name for an interface's (class, subclass, protocol)
|
||||
/// triple, decoding the well-known combinations recognizable in a log — e.g.
|
||||
/// "HID boot keyboard", "Mass Storage SCSI Bulk-Only", "Bluetooth". Falls back
|
||||
/// to the class name (and then "Unknown") for codes without a spelled-out combo.
|
||||
pub fn interfaceName(class: u8, subclass: u8, protocol: u8) []const u8 {
|
||||
return switch (@as(Class, @enumFromInt(class))) {
|
||||
.hid => if (subclass == @intFromEnum(hid.SubClass.boot)) switch (@as(hid.Protocol, @enumFromInt(protocol))) {
|
||||
.keyboard => "HID boot keyboard",
|
||||
.mouse => "HID boot mouse",
|
||||
else => "HID boot device",
|
||||
} else "HID",
|
||||
.mass_storage => switch (@as(mass_storage.Protocol, @enumFromInt(protocol))) {
|
||||
.bulk_only => "Mass Storage (Bulk-Only)",
|
||||
.uas => "Mass Storage (UAS)",
|
||||
else => "Mass Storage",
|
||||
},
|
||||
.hub => switch (@as(hub.Protocol, @enumFromInt(protocol))) {
|
||||
.super_speed => "Hub (SuperSpeed)",
|
||||
.hi_speed_multi_tt => "Hub (Hi-Speed multi-TT)",
|
||||
.hi_speed_single_tt => "Hub (Hi-Speed single-TT)",
|
||||
else => "Hub",
|
||||
},
|
||||
.wireless_controller => if (subclass == @intFromEnum(wireless_controller.SubClass.radio_frequency))
|
||||
wireless_controller.protocolName(protocol)
|
||||
else
|
||||
"Wireless Controller",
|
||||
.communications => communications.subclassName(subclass),
|
||||
.application_specific => application_specific.subclassName(subclass),
|
||||
.miscellaneous => "Miscellaneous",
|
||||
else => className(class),
|
||||
};
|
||||
}
|
||||
|
||||
// Subclass and protocol codes qualified by Class.hub. Hubs have no subclass codes; the
|
||||
// protocol distinguishes the hub's transaction-translator arrangement.
|
||||
pub const hub = struct {
|
||||
@@ -84,6 +182,16 @@ pub const hub = struct {
|
||||
super_speed = 0x03,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.full_speed => "full-speed",
|
||||
.hi_speed_single_tt => "Hi-Speed single-TT",
|
||||
.hi_speed_multi_tt => "Hi-Speed multi-TT",
|
||||
.super_speed => "SuperSpeed",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.hid.
|
||||
@@ -104,6 +212,23 @@ pub const hid = struct {
|
||||
mouse = 0x02,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.none => "none",
|
||||
.boot => "boot",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.none => "none",
|
||||
.keyboard => "keyboard",
|
||||
.mouse => "mouse",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.mass_storage. The subclass identifies the
|
||||
@@ -147,6 +272,33 @@ pub const mass_storage = struct {
|
||||
vendor_specific = 0xFF,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.not_reported => "SCSI (not reported)",
|
||||
.rbc => "RBC",
|
||||
.atapi => "ATAPI",
|
||||
.qic_157 => "QIC-157",
|
||||
.ufi => "UFI",
|
||||
.sff_8070i => "SFF-8070i",
|
||||
.scsi => "SCSI",
|
||||
.lsd_fs => "LSD FS",
|
||||
.ieee_1667 => "IEEE 1667",
|
||||
.vendor_specific => "vendor-specific",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.cbi_completion_interrupt => "CBI",
|
||||
.cbi => "CBI (no completion IRQ)",
|
||||
.bulk_only => "Bulk-Only",
|
||||
.uas => "UAS",
|
||||
.vendor_specific => "vendor-specific",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.communications (CDC). The protocol codes
|
||||
@@ -182,6 +334,25 @@ pub const communications = struct {
|
||||
network_control = 0x0D,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.direct_line => "Direct Line",
|
||||
.abstract_control => "Abstract Control (modem/serial)",
|
||||
.telephone => "Telephone",
|
||||
.multi_channel => "Multi-Channel",
|
||||
.capi => "CAPI",
|
||||
.ethernet => "Ethernet",
|
||||
.atm => "ATM",
|
||||
.wireless_handset => "Wireless Handset",
|
||||
.device_management => "Device Management",
|
||||
.mobile_direct_line => "Mobile Direct Line",
|
||||
.obex => "OBEX",
|
||||
.ethernet_emulation => "Ethernet Emulation",
|
||||
.network_control => "Network Control",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.wireless_controller.
|
||||
@@ -204,6 +375,23 @@ pub const wireless_controller = struct {
|
||||
bluetooth_amp = 0x04,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.radio_frequency => "RF",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.bluetooth => "Bluetooth",
|
||||
.ultra_wideband => "Ultra-Wideband",
|
||||
.remote_ndis => "Remote NDIS",
|
||||
.bluetooth_amp => "Bluetooth AMP",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.miscellaneous.
|
||||
@@ -221,6 +409,20 @@ pub const miscellaneous = struct {
|
||||
interface_association = 0x01,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.common => "common",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
pub fn protocolName(protocol: u8) []const u8 {
|
||||
return switch (@as(Protocol, @enumFromInt(protocol))) {
|
||||
.interface_association => "Interface Association",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Subclass and protocol codes qualified by Class.application_specific.
|
||||
@@ -234,6 +436,15 @@ pub const application_specific = struct {
|
||||
test_and_measurement = 0x03,
|
||||
_,
|
||||
};
|
||||
|
||||
pub fn subclassName(subclass: u8) []const u8 {
|
||||
return switch (@as(SubClass, @enumFromInt(subclass))) {
|
||||
.firmware_upgrade => "Device Firmware Upgrade",
|
||||
.irda_bridge => "IrDA Bridge",
|
||||
.test_and_measurement => "Test & Measurement",
|
||||
_ => "unknown",
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// Pack a (class, subclass, protocol) triple into one 0xCCSSPP value — the
|
||||
@@ -280,6 +491,30 @@ test "class codes match the USB-IF assignments" {
|
||||
_ = application_specific.SubClass.firmware_upgrade;
|
||||
}
|
||||
|
||||
test "readable names decode the well-known triples" {
|
||||
const std = @import("std");
|
||||
const eql = std.testing.expectEqualStrings;
|
||||
|
||||
try eql("Hub", className(0x09));
|
||||
try eql("Unknown", className(0x42));
|
||||
|
||||
// interfaceName decodes the combos we log.
|
||||
try eql("HID boot keyboard", interfaceName(0x03, 0x01, 0x01));
|
||||
try eql("HID boot mouse", interfaceName(0x03, 0x01, 0x02));
|
||||
try eql("Mass Storage (Bulk-Only)", interfaceName(0x08, 0x06, 0x50));
|
||||
try eql("Hub (SuperSpeed)", interfaceName(0x09, 0x00, 0x03));
|
||||
try eql("Bluetooth", interfaceName(0xE0, 0x01, 0x01));
|
||||
|
||||
// The per-enum name functions.
|
||||
try eql("Bulk-Only", mass_storage.protocolName(0x50));
|
||||
try eql("SCSI", mass_storage.subclassName(0x06));
|
||||
try eql("Bluetooth", wireless_controller.protocolName(0x01));
|
||||
try eql("SuperSpeed", hub.protocolName(0x03));
|
||||
try eql("keyboard", hid.protocolName(0x01));
|
||||
try eql("SuperSpeed", speedName(4));
|
||||
try eql("Polling", linkStateName(7));
|
||||
}
|
||||
|
||||
test "packTriple / unpackTriple round-trip the identity a bus driver reports" {
|
||||
const std = @import("std");
|
||||
const expectEqual = std.testing.expectEqual;
|
||||
@@ -5,7 +5,7 @@
|
||||
//! service and `device.zig` over the raw device calls.
|
||||
//!
|
||||
//! A class driver, spawned with its interface's assigned device id as argv[1]:
|
||||
//! if (!usb.helloManager(id)) return; // meet the spawn deadline
|
||||
//! if (device_manager.hello(.device, id) == null) return; // meet the spawn deadline
|
||||
//! var device = usb.open(id) orelse return; // open + get its endpoints
|
||||
//! _ = device.controlOut(usb_abi.setProtocol(...));// class requests, descriptors
|
||||
//! _ = device.subscribeInterrupt(address, length); // reports arrive asynchronously
|
||||
@@ -16,14 +16,18 @@
|
||||
//! the service harness drops buffered-message payloads — see service.zig).
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("usb-transfer-protocol");
|
||||
const device_manager = @import("device-manager-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||
|
||||
pub const Endpoint = protocol.Endpoint;
|
||||
pub const InterruptReport = protocol.InterruptReport;
|
||||
pub const max_report_data = protocol.max_report_data;
|
||||
/// The USB chapter-9 wire ABI and the class taxonomy, re-exported so a class driver reaches
|
||||
/// the whole USB domain through its one `usb` import (`usb.abi.getDescriptor`, `usb.ids.Class`).
|
||||
pub const abi = @import("usb-abi");
|
||||
pub const ids = @import("usb-ids");
|
||||
|
||||
pub const Endpoint = usb_transfer_protocol.Endpoint;
|
||||
pub const InterruptReport = usb_transfer_protocol.InterruptReport;
|
||||
pub const max_report_data = usb_transfer_protocol.max_report_data;
|
||||
|
||||
// Endpoint transfer types (EndpointDescriptor attributes), for `findEndpoint`.
|
||||
pub const transfer_type_bulk: u8 = 2;
|
||||
@@ -41,7 +45,7 @@ pub const Device = struct {
|
||||
protocol_code: u8,
|
||||
interface_number: u8,
|
||||
endpoint_count: usize = 0,
|
||||
endpoints: [protocol.max_reported_endpoints]Endpoint = undefined,
|
||||
endpoints: [usb_transfer_protocol.max_reported_endpoints]Endpoint = undefined,
|
||||
|
||||
/// The interface's first endpoint of the given transfer type and direction
|
||||
/// (`transfer_type_bulk` / `transfer_type_interrupt`), or null.
|
||||
@@ -53,17 +57,17 @@ pub const Device = struct {
|
||||
}
|
||||
|
||||
fn controlTransfer(self: *Device, setup: [8]u8, direction_in: bool, data: []u8) ?usize {
|
||||
var request = protocol.ControlRequest{
|
||||
var request = usb_transfer_protocol.ControlRequest{
|
||||
.device_token = self.token,
|
||||
.setup = setup,
|
||||
.direction_in = @intFromBool(direction_in),
|
||||
.data_length = @intCast(data.len),
|
||||
};
|
||||
if (!direction_in and data.len > 0) @memcpy(request.data[0..data.len], data);
|
||||
var reply: [@sizeOf(protocol.ControlReply)]u8 = undefined;
|
||||
var reply: [@sizeOf(usb_transfer_protocol.ControlReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (length < @sizeOf(protocol.ControlReply)) return null;
|
||||
const control_reply = std.mem.bytesToValue(protocol.ControlReply, reply[0..@sizeOf(protocol.ControlReply)]);
|
||||
if (length < @sizeOf(usb_transfer_protocol.ControlReply)) return null;
|
||||
const control_reply = std.mem.bytesToValue(usb_transfer_protocol.ControlReply, reply[0..@sizeOf(usb_transfer_protocol.ControlReply)]);
|
||||
if (control_reply.status != 0) return null;
|
||||
const actual = @min(control_reply.actual_length, data.len);
|
||||
if (direction_in and actual > 0) @memcpy(data[0..actual], control_reply.data[0..actual]);
|
||||
@@ -84,30 +88,30 @@ pub const Device = struct {
|
||||
/// Begin periodic IN polling of an interrupt endpoint; reports flow back to
|
||||
/// `self.endpoint` as asynchronous `InterruptReport` messages.
|
||||
pub fn subscribeInterrupt(self: *Device, endpoint_address: u8, max_length: u16) bool {
|
||||
var request = protocol.InterruptSubscribeRequest{
|
||||
var request = usb_transfer_protocol.InterruptSubscribeRequest{
|
||||
.device_token = self.token,
|
||||
.endpoint_address = endpoint_address,
|
||||
.max_length = max_length,
|
||||
};
|
||||
var reply: [@sizeOf(protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||
var reply: [@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (length < @sizeOf(protocol.InterruptSubscribeReply)) return false;
|
||||
return std.mem.bytesToValue(protocol.InterruptSubscribeReply, reply[0..@sizeOf(protocol.InterruptSubscribeReply)]).status == 0;
|
||||
if (length < @sizeOf(usb_transfer_protocol.InterruptSubscribeReply)) return false;
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.InterruptSubscribeReply, reply[0..@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
var request = protocol.BulkRequest{
|
||||
var request = usb_transfer_protocol.BulkRequest{
|
||||
.device_token = self.token,
|
||||
.physical_address = physical,
|
||||
.length = length,
|
||||
.endpoint_address = endpoint_address,
|
||||
};
|
||||
var reply: [@sizeOf(protocol.BulkReply)]u8 = undefined;
|
||||
var reply: [@sizeOf(usb_transfer_protocol.BulkReply)]u8 = undefined;
|
||||
const replied = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (replied < @sizeOf(protocol.BulkReply)) return null;
|
||||
const bulk_reply = std.mem.bytesToValue(protocol.BulkReply, reply[0..@sizeOf(protocol.BulkReply)]);
|
||||
if (replied < @sizeOf(usb_transfer_protocol.BulkReply)) return null;
|
||||
const bulk_reply = std.mem.bytesToValue(usb_transfer_protocol.BulkReply, reply[0..@sizeOf(usb_transfer_protocol.BulkReply)]);
|
||||
if (bulk_reply.status != 0) return null;
|
||||
return bulk_reply.actual_length;
|
||||
}
|
||||
@@ -120,15 +124,15 @@ pub fn open(device_id: u64) ?Device {
|
||||
var attempts: usize = 0;
|
||||
const bus = while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.usb_bus)) |handle| break handle;
|
||||
system.sleep(20);
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
var request = protocol.OpenRequest{ .device_id = device_id };
|
||||
var reply: [@sizeOf(protocol.OpenReply)]u8 = undefined;
|
||||
var request = usb_transfer_protocol.OpenRequest{ .device_id = device_id };
|
||||
var reply: [@sizeOf(usb_transfer_protocol.OpenReply)]u8 = undefined;
|
||||
const result = ipc.callCap(bus, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < @sizeOf(protocol.OpenReply)) return null;
|
||||
const open_reply = std.mem.bytesToValue(protocol.OpenReply, reply[0..@sizeOf(protocol.OpenReply)]);
|
||||
if (result.len < @sizeOf(usb_transfer_protocol.OpenReply)) return null;
|
||||
const open_reply = std.mem.bytesToValue(usb_transfer_protocol.OpenReply, reply[0..@sizeOf(usb_transfer_protocol.OpenReply)]);
|
||||
if (open_reply.status != 0) return null;
|
||||
|
||||
var device = Device{
|
||||
@@ -139,24 +143,8 @@ pub fn open(device_id: u64) ?Device {
|
||||
.subclass = open_reply.interface_subclass,
|
||||
.protocol_code = open_reply.interface_protocol,
|
||||
.interface_number = open_reply.interface_number,
|
||||
.endpoint_count = @min(open_reply.endpoint_count, protocol.max_reported_endpoints),
|
||||
.endpoint_count = @min(open_reply.endpoint_count, usb_transfer_protocol.max_reported_endpoints),
|
||||
};
|
||||
for (0..device.endpoint_count) |index| device.endpoints[index] = open_reply.endpoints[index];
|
||||
return device;
|
||||
}
|
||||
|
||||
/// Hello the device manager as a class driver (Role.device) so a supervised
|
||||
/// spawn meets its hello deadline. Retries while the manager comes up.
|
||||
pub fn helloManager(device_id: u64) bool {
|
||||
var attempts: usize = 0;
|
||||
const manager = while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||
system.sleep(20);
|
||||
} else return false;
|
||||
|
||||
const hello = device_manager.Hello{ .role = @intFromEnum(device_manager.Role.device), .device_id = device_id };
|
||||
var reply: [device_manager.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch return false;
|
||||
if (length < device_manager.reply_size) return false;
|
||||
return std.mem.bytesToValue(device_manager.HelloReply, reply[0..device_manager.reply_size]).status == 0;
|
||||
}
|
||||
@@ -0,0 +1,427 @@
|
||||
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
|
||||
//! lists files through the kernel VFS root (resolve + redirect), each call
|
||||
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
|
||||
//! danos programs use directly; it is also where the file operations that later
|
||||
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
|
||||
//! the old POSIX `unistd` shim — a compatibility spelling danos does not need yet.
|
||||
//!
|
||||
//! Handles are *values*, not entries in a global descriptor table: a `File` /
|
||||
//! `Directory` owns its VFS node id and (for files) a byte offset. So there is no
|
||||
//! per-process fd limit and no shared table to synchronise — the danos-native
|
||||
//! shape, unlike the POSIX fd model the old shim emulated.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||
/// wire protocol.
|
||||
pub const Kind = vfs_protocol.NodeKind;
|
||||
|
||||
/// A node's metadata (the answer to a status request).
|
||||
pub const Attributes = struct {
|
||||
size: u64,
|
||||
kind: Kind,
|
||||
/// Modification time — Unix epoch seconds, UTC. 0 if the filesystem has none.
|
||||
mtime: u64 = 0,
|
||||
};
|
||||
|
||||
// Map a wire `NodeKind` value to the enum, defaulting anything unrecognised to
|
||||
// `.regular` (the server is trusted, but a value outside the enum would be
|
||||
// illegal to `@enumFromInt` directly).
|
||||
fn kindFromWire(value: u32) Kind {
|
||||
return switch (value) {
|
||||
@intFromEnum(Kind.directory) => .directory,
|
||||
@intFromEnum(Kind.character_device) => .character_device,
|
||||
@intFromEnum(Kind.block_device) => .block_device,
|
||||
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
||||
@intFromEnum(Kind.fifo) => .fifo,
|
||||
@intFromEnum(Kind.socket) => .socket,
|
||||
else => .regular,
|
||||
};
|
||||
}
|
||||
|
||||
/// How to open a path.
|
||||
pub const OpenOptions = struct {
|
||||
/// Create the file if it does not exist.
|
||||
create: bool = false,
|
||||
/// Open a directory node (for listing) rather than a file.
|
||||
directory: bool = false,
|
||||
/// Truncate an existing file to zero length on open (O_TRUNC) — replace its
|
||||
/// contents rather than overwriting in place.
|
||||
truncate: bool = false,
|
||||
|
||||
fn wireFlags(self: OpenOptions) u32 {
|
||||
var f: u32 = 0;
|
||||
if (self.create) f |= vfs_protocol.create;
|
||||
if (self.directory) f |= vfs_protocol.directory;
|
||||
if (self.truncate) f |= vfs_protocol.truncate;
|
||||
return f;
|
||||
}
|
||||
};
|
||||
|
||||
// The route to a path: the kernel resolves (fs_resolve) and either serves the
|
||||
// node itself (the initrd at /system — a permanent token) or redirects us to
|
||||
// the owning filesystem backend's endpoint, to which we speak the vfs-protocol
|
||||
// rendezvous directly with the rewritten mount-relative path.
|
||||
const Route = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: ipc.Handle, path: [224]u8, path_len: usize },
|
||||
|
||||
fn backendPath(self: *const Route) []const u8 {
|
||||
return self.backend.path[0..self.backend.path_len];
|
||||
}
|
||||
};
|
||||
|
||||
fn resolve(path: []const u8, flags: usize) ?Route {
|
||||
var out: [224]u8 = undefined;
|
||||
const route = fsResolve(path, flags, &out) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .kernel = token },
|
||||
.backend => |b| {
|
||||
var r: Route = .{ .backend = .{ .handle = b.handle, .path = undefined, .path_len = b.path_len } };
|
||||
@memcpy(r.backend.path[0..b.path_len], out[0..b.path_len]);
|
||||
return r;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
const Result = struct { reply: vfs_protocol.Reply, payload: []u8 };
|
||||
|
||||
// One request/reply round trip: [Request header][send payload] -> backend ->
|
||||
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||
fn transact(h: ipc.Handle, request: vfs_protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
var message: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..vfs_protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, vfs_protocol.maximum_payload);
|
||||
@memcpy(message[vfs_protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. vfs_protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < vfs_protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(vfs_protocol.Reply, rbuf[0..vfs_protocol.reply_size]);
|
||||
const rpl = @min(n - vfs_protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[vfs_protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
/// An open file: a VFS node plus a byte cursor. Read and write advance the cursor.
|
||||
pub const File = struct {
|
||||
node: u64,
|
||||
offset: u64 = 0,
|
||||
/// The owning backend's endpoint, or null for a kernel-served node (the
|
||||
/// read-only /system tree), whose `node` is a permanent fs_node token.
|
||||
backend: ?ipc.Handle = null,
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
|
||||
/// null on error.
|
||||
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||
const h = self.backend orelse {
|
||||
const n = fsNodeRead(self.node, self.offset, buffer) orelse return null;
|
||||
self.offset += n;
|
||||
return n;
|
||||
};
|
||||
const want: u32 = @intCast(@min(buffer.len, vfs_protocol.maximum_payload));
|
||||
const request = vfs_protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(h, request, &.{}, buffer) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write `data` at the current offset; returns the count written. A single
|
||||
/// call is capped at the VFS payload size, so the return may be short — use
|
||||
/// `writeAll` to write the whole slice. Null on error (kernel-served nodes
|
||||
/// are read-only).
|
||||
pub fn write(self: *File, data: []const u8) ?usize {
|
||||
const h = self.backend orelse return null;
|
||||
const want: u32 = @intCast(@min(data.len, vfs_protocol.maximum_payload));
|
||||
const request = vfs_protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(h, request, data[0..want], &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write all of `data`, looping past the per-call payload cap. Returns the
|
||||
/// total written, or null if a write failed before any progress.
|
||||
pub fn writeAll(self: *File, data: []const u8) ?usize {
|
||||
var written: usize = 0;
|
||||
while (written < data.len) {
|
||||
const n = self.write(data[written..]) orelse return if (written == 0) null else written;
|
||||
if (n == 0) return written; // no forward progress; stop rather than spin
|
||||
written += n;
|
||||
}
|
||||
return written;
|
||||
}
|
||||
|
||||
/// Move the read/write cursor to an absolute byte position.
|
||||
pub fn seekTo(self: *File, position: u64) void {
|
||||
self.offset = position;
|
||||
}
|
||||
|
||||
/// This file's metadata.
|
||||
pub fn attributes(self: *File) ?Attributes {
|
||||
const h = self.backend orelse {
|
||||
const a = fsNodeStatus(self.node) orelse return null;
|
||||
return .{ .size = a.size, .kind = if (a.kind == file_kind_directory) .directory else .regular, .mtime = a.mtime };
|
||||
};
|
||||
const request = vfs_protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(vfs_protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return null;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(vfs_protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(vfs_protocol.FileStatus, buffer[0..@sizeOf(vfs_protocol.FileStatus)]);
|
||||
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||
}
|
||||
|
||||
/// Release the backend's open handle for this file. Kernel-served node
|
||||
/// tokens are permanent — nothing to release.
|
||||
pub fn close(self: *File) void {
|
||||
const h = self.backend orelse return;
|
||||
const request = vfs_protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(h, request, &.{}, &.{});
|
||||
}
|
||||
};
|
||||
|
||||
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
|
||||
pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||
const route = resolve(path, options.wireFlags()) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .node = token, .backend = null },
|
||||
.backend => |b| {
|
||||
const relative = route.backendPath();
|
||||
const request = vfs_protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() };
|
||||
const r = transact(b.handle, request, relative, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node, .backend = b.handle };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// A path's metadata without keeping it open (open -> status -> close).
|
||||
pub fn attributes(path: []const u8) ?Attributes {
|
||||
var file = open(path, .{}) orelse return null;
|
||||
defer file.close();
|
||||
return file.attributes();
|
||||
}
|
||||
|
||||
/// Whether `path` resolves — handy as a readiness check (e.g. waiting for a mount
|
||||
/// to come up before writing to it).
|
||||
pub fn exists(path: []const u8) bool {
|
||||
return attributes(path) != null;
|
||||
}
|
||||
|
||||
/// One entry returned by `Directory.next`.
|
||||
pub const Entry = struct {
|
||||
kind: Kind = .regular,
|
||||
size: u64 = 0,
|
||||
name_buffer: [64]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
|
||||
pub fn name(self: *const Entry) []const u8 {
|
||||
return self.name_buffer[0..self.name_len];
|
||||
}
|
||||
};
|
||||
|
||||
/// An open directory being listed, cursor-advanced by `next`.
|
||||
pub const Directory = struct {
|
||||
node: u64,
|
||||
cursor: u64 = 0,
|
||||
backend: ?ipc.Handle = null,
|
||||
|
||||
/// Fill `entry` with the next directory entry; false at end of directory or
|
||||
/// on error.
|
||||
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||
const h = self.backend orelse {
|
||||
var buffer: [@sizeOf(DirectoryEntryHeader) + 64]u8 = undefined;
|
||||
const n = fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false;
|
||||
if (n < @sizeOf(DirectoryEntryHeader)) return false; // end
|
||||
const header = std.mem.bytesToValue(DirectoryEntryHeader, buffer[0..@sizeOf(DirectoryEntryHeader)]);
|
||||
entry.kind = if (header.kind == file_kind_directory) .directory else .regular;
|
||||
entry.size = header.size;
|
||||
const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(DirectoryEntryHeader)..][0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
};
|
||||
const request = vfs_protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return false;
|
||||
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||
if (r.payload.len < vfs_protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(vfs_protocol.DirectoryEntry, r.payload[0..vfs_protocol.directory_entry_size]);
|
||||
entry.kind = kindFromWire(header.kind);
|
||||
entry.size = header.size;
|
||||
const source = r.payload[vfs_protocol.directory_entry_size..];
|
||||
const nlen = @min(@min(@as(usize, header.name_len), source.len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], source[0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release the backend's open handle for this directory.
|
||||
pub fn close(self: *Directory) void {
|
||||
var f = File{ .node = self.node, .backend = self.backend };
|
||||
f.close();
|
||||
}
|
||||
};
|
||||
|
||||
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
|
||||
pub fn openDirectory(path: []const u8) ?Directory {
|
||||
const file = open(path, .{ .directory = true }) orelse return null;
|
||||
return .{ .node = file.node, .backend = file.backend };
|
||||
}
|
||||
|
||||
// A path-based request that returns only a status (mkdir, unlink). Kernel-served
|
||||
// paths (the read-only /system) refuse mutation by construction: the resolve
|
||||
// must land on a backend.
|
||||
fn pathOperation(operation: vfs_protocol.Operation, path: []const u8) bool {
|
||||
const route = resolve(path, 0) orelse return false;
|
||||
if (route != .backend) return false;
|
||||
const relative = route.backendPath();
|
||||
const request = vfs_protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
const r = transact(route.backend.handle, request, relative, &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Create a directory at `path` (its parent must already exist). Returns true on
|
||||
/// success. Only works under a mounted filesystem that supports directories.
|
||||
pub fn makeDirectory(path: []const u8) bool {
|
||||
return pathOperation(.mkdir, path);
|
||||
}
|
||||
|
||||
/// Create every missing directory along `path` (mkdir -p). Probes each prefix
|
||||
/// with `exists` first — a FAT mkdir of an existing name is refused, and the
|
||||
/// probe keeps the common "already there" case cheap. Returns true when the
|
||||
/// whole path exists afterwards.
|
||||
pub fn makePath(path: []const u8) bool {
|
||||
var end: usize = 0;
|
||||
while (end < path.len) {
|
||||
end += 1;
|
||||
while (end < path.len and path[end] != '/') end += 1;
|
||||
const prefix = path[0..end];
|
||||
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
||||
// Best-effort per prefix: components at or above a mount point ("/mnt")
|
||||
// are router names, not filesystem nodes — they neither exist as nodes
|
||||
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
||||
if (!exists(prefix)) _ = makeDirectory(prefix);
|
||||
}
|
||||
return exists(path);
|
||||
}
|
||||
|
||||
/// Remove the file at `path`. Returns true on success. Directories are refused
|
||||
/// (a separate directory-removal would have to check emptiness).
|
||||
pub fn remove(path: []const u8) bool {
|
||||
return pathOperation(.unlink, path);
|
||||
}
|
||||
|
||||
/// Rename `old_path` to `new_path`. Both must resolve to the SAME filesystem
|
||||
/// backend (same-directory, 8.3-name rename only for now). Returns true on
|
||||
/// success.
|
||||
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
const old_route = resolve(old_path, 0) orelse return false;
|
||||
const new_route = resolve(new_path, 0) orelse return false;
|
||||
if (old_route != .backend or new_route != .backend) return false;
|
||||
if (old_route.backend.handle != new_route.backend.handle) return false; // cross-filesystem
|
||||
const old_relative = old_route.backendPath();
|
||||
const new_relative = new_route.backendPath();
|
||||
const total = old_relative.len + 1 + new_relative.len;
|
||||
if (total > vfs_protocol.maximum_payload) return false;
|
||||
var payload: [vfs_protocol.maximum_payload]u8 = undefined;
|
||||
@memcpy(payload[0..old_relative.len], old_relative);
|
||||
payload[old_relative.len] = 0;
|
||||
@memcpy(payload[old_relative.len + 1 ..][0..new_relative.len], new_relative);
|
||||
const request = vfs_protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const r = transact(old_route.backend.handle, request, payload[0..total], &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||
/// the kernel VFS then routes everything under `target` to that backend.
|
||||
/// Possession of the endpoint handle is the capability. Returns true on success.
|
||||
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||
return fsMount(target, backend, "");
|
||||
}
|
||||
|
||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||
return fsMount(target, backend, rewrite);
|
||||
}
|
||||
|
||||
// --- raw filesystem syscalls, formerly in the system.zig dumping ground ---
|
||||
|
||||
pub const FileAttributes = abi.FileAttributes;
|
||||
pub const DirectoryEntryHeader = abi.DirectoryEntryHeader;
|
||||
pub const file_kind_regular = abi.file_kind_regular;
|
||||
pub const file_kind_directory = abi.file_kind_directory;
|
||||
|
||||
/// Where fs_resolve routed a path: served by the kernel (a permanent node token for
|
||||
/// `fs_node`) or by a user-space filesystem backend (an endpoint handle plus the rewritten
|
||||
/// mount-relative path returned in the caller's buffer).
|
||||
pub const FsRoute = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: usize, path_len: usize },
|
||||
};
|
||||
|
||||
/// Route `path` through the kernel VFS. For a backend route the rewritten mount-relative
|
||||
/// path lands in `out` (behind a kernel-written length prefix, already stripped here:
|
||||
/// out[0..path_len] is the path).
|
||||
pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)),
|
||||
[a0] "{rdi}" (@intFromPtr(path.ptr)),
|
||||
[a1] "{rsi}" (path.len),
|
||||
[a3] "{r10}" (@intFromPtr(out.ptr)),
|
||||
[a4] "{r8}" (out.len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (@as(isize, @bitCast(rax)) < 0) return null;
|
||||
if (rax == abi.fs_route_kernel) return .{ .kernel = rdx };
|
||||
if (rax != abi.fs_route_backend) return null;
|
||||
const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8);
|
||||
if (path_len + 2 > out.len) return null;
|
||||
std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]);
|
||||
return .{ .backend = .{ .handle = rdx, .path_len = path_len } };
|
||||
}
|
||||
|
||||
/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read).
|
||||
pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// A kernel-served node's metadata (fs_node status).
|
||||
pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes {
|
||||
var attrs: abi.FileAttributes = undefined;
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attrs), @sizeOf(abi.FileAttributes));
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return attrs;
|
||||
}
|
||||
|
||||
/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills `out` with
|
||||
/// [DirectoryEntryHeader][name]; returns total bytes (0 = end).
|
||||
pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// Mount a userspace filesystem's endpoint at `prefix`, with an optional backend-side
|
||||
/// `rewrite` prefix ("" = none). Possession of the endpoint handle is the capability.
|
||||
pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool {
|
||||
return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0;
|
||||
}
|
||||
|
||||
pub fn fsUnmount(prefix: []const u8) bool {
|
||||
return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0;
|
||||
}
|
||||
@@ -4,7 +4,7 @@
|
||||
//! added with the first server binary.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const sc = @import("system-call");
|
||||
|
||||
/// A small-int handle into the calling process's handle table.
|
||||
pub const Handle = usize;
|
||||
@@ -0,0 +1,97 @@
|
||||
//! The per-process logger: std.log wired to the tagged kernel log ring.
|
||||
//!
|
||||
//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped
|
||||
//! logger); this backend formats the line into a fixed buffer and emits ONE
|
||||
//! `debug_write` record carrying the level. The kernel stamps the record with
|
||||
//! the sender's pid and task name (its binary path) — the process does NOT put
|
||||
//! its own name in the payload; attribution is the kernel's, structural and
|
||||
//! unforgeable. Serial shows the kernel-rendered `<path>: message` line, and
|
||||
//! the logger service demultiplexes the ring into one file per process.
|
||||
//!
|
||||
//! Installed for every user binary by the root shim (library/runtime/root.zig)
|
||||
//! via `std_options`; a program can override by declaring its own
|
||||
//! `pub const std_options`.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
|
||||
// --- the tagged log ring: raw wrappers + record types, formerly in the system.zig dump ---
|
||||
|
||||
/// A log record's level and the ring's framing types (re-exported from the shared ABI so
|
||||
/// callers and the logger service don't import `abi` themselves).
|
||||
pub const KlogLevel = abi.KlogLevel;
|
||||
pub const KlogStatus = abi.KlogStatus;
|
||||
pub const KlogRecordHeader = abi.KlogRecordHeader;
|
||||
pub const klog_record_header_size = abi.klog_record_header_size;
|
||||
pub const klog_record_alignment = abi.klog_record_alignment;
|
||||
pub const klog_record_magic = abi.klog_record_magic;
|
||||
pub const klog_flag_truncated = abi.klog_flag_truncated;
|
||||
pub const klog_maximum_message = abi.klog_maximum_message;
|
||||
pub const maximum_process_name = abi.maximum_process_name;
|
||||
|
||||
/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary output goes
|
||||
/// through std.log -> writeRecord). The kernel stamps the record with this process's id
|
||||
/// and name. Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return writeRecord(.raw, message);
|
||||
}
|
||||
|
||||
/// Emit one leveled record into the tagged kernel log ring. The kernel stamps
|
||||
/// pid/name/sequence/timestamp; the payload should be a single line.
|
||||
pub fn writeRecord(level: KlogLevel, message: []const u8) usize {
|
||||
return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level));
|
||||
}
|
||||
|
||||
/// Copy framed records out of the tagged kernel log ring starting at stream `offset` into
|
||||
/// `out`. Returns the byte count (0 = caught up), or null when `offset` fell behind the
|
||||
/// ring's tail or lies past its head (re-sync via `klogStatus`).
|
||||
pub fn klogRead(offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// The log ring's live cursors (oldest retained offset, end of stream, next sequence)
|
||||
/// plus the wall-clock time of boot — how a log reader starts, detects loss, and names a
|
||||
/// per-boot log directory.
|
||||
pub fn klogStatus() ?KlogStatus {
|
||||
var status: KlogStatus = undefined;
|
||||
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null;
|
||||
return status;
|
||||
}
|
||||
|
||||
fn levelOf(comptime level: std.log.Level) KlogLevel {
|
||||
return switch (level) {
|
||||
.err => .err,
|
||||
.warn => .warn,
|
||||
.info => .info,
|
||||
.debug => .debug,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn logFn(
|
||||
comptime level: std.log.Level,
|
||||
comptime scope: @EnumLiteral(),
|
||||
comptime format: []const u8,
|
||||
args: anytype,
|
||||
) void {
|
||||
// One record = one line = at most klog_maximum_message bytes of payload.
|
||||
// On overflow keep what fits and end with "~" so the record is still a
|
||||
// whole line (the kernel would split an embedded rest anyway).
|
||||
var buffer: [256]u8 = undefined;
|
||||
const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") ";
|
||||
const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: {
|
||||
buffer[buffer.len - 1] = '~';
|
||||
break :truncated buffer[0..];
|
||||
};
|
||||
_ = writeRecord(levelOf(level), line);
|
||||
}
|
||||
|
||||
/// The std.Options the root shim installs unless the program overrides it.
|
||||
/// Debug level: filtering is the log *reader's* job here — the ring is cheap,
|
||||
/// serial is a dev convenience, and the logger service keeps everything.
|
||||
pub const default_options: std.Options = .{
|
||||
.log_level = .debug,
|
||||
.logFn = logFn,
|
||||
};
|
||||
@@ -5,7 +5,7 @@
|
||||
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const sc = @import("system-call");
|
||||
|
||||
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||
/// opt-in for specific hardware — see `abi`.
|
||||
@@ -19,8 +19,8 @@
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const system_calls = @import("system.zig");
|
||||
const Mutex = @import("thread.zig").Thread.Mutex;
|
||||
const sc = @import("system-call");
|
||||
const Mutex = @import("thread").Thread.Mutex;
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
@@ -63,8 +63,8 @@ fn payloadOf(block: *Block) [*]u8 {
|
||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||
if (system_calls.mmapFailed(ret)) return false;
|
||||
const ret = sc.systemCall2(.mmap, bytes, abi.prot_read | abi.prot_write);
|
||||
if (ret > ~@as(usize, 0) - 4095) return false; // a wrapped -errno lands in the top page
|
||||
|
||||
const block: *Block = @ptrFromInt(ret);
|
||||
block.size = bytes;
|
||||
@@ -0,0 +1,48 @@
|
||||
//! library/kernel/memory — the process's memory interface: the heap allocator, DMA-capable
|
||||
//! buffers, shared-memory regions, and the raw `mmap` grant they all sit on. One flat module
|
||||
//! (formerly runtime.heap / runtime.dma / runtime.shared_memory, plus the `mmap` wrappers that
|
||||
//! lived in the system.zig dumping ground). Its private files are heap.zig, dma.zig, and
|
||||
//! shared-memory.zig — imported only here, so the heap's state and C symbols exist once.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
const heap = @import("heap.zig");
|
||||
const dma = @import("dma.zig");
|
||||
const shared = @import("shared-memory.zig");
|
||||
|
||||
// --- the heap: a std.mem.Allocator over a first-fit free list (C malloc/free are also
|
||||
// exported from heap.zig, compiled once here) ---
|
||||
pub const allocator = heap.allocator;
|
||||
|
||||
// --- the raw grant every allocation sits on ---
|
||||
pub const PROT_READ: usize = abi.prot_read;
|
||||
pub const PROT_WRITE: usize = abi.prot_write;
|
||||
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable memory and
|
||||
/// return the base virtual address. On failure returns a value in the top page (`mmapFailed`).
|
||||
pub fn mmap(len: usize, prot: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, prot);
|
||||
}
|
||||
/// Release a range previously handed out by `mmap`.
|
||||
pub fn munmap(base: usize, len: usize) usize {
|
||||
return sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
/// Whether an `mmap` return value is an error (a wrapped -errno lands in the top page).
|
||||
pub inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
// --- DMA-capable buffers: physically contiguous, pinned, uncacheable, physical address known ---
|
||||
pub const DmaRegion = dma.Region;
|
||||
pub const dma_coherent = dma.coherent;
|
||||
pub const dma_write_combining = dma.write_combining;
|
||||
pub const dma_below_4g = dma.below_4g;
|
||||
pub const dmaAlloc = dma.alloc;
|
||||
pub const dmaFree = dma.free;
|
||||
|
||||
// --- shared-memory regions: a capability handed to another process over an ipc_call send_cap ---
|
||||
pub const SharedRegion = shared.Region;
|
||||
pub const sharedCreate = shared.create;
|
||||
pub const sharedMap = shared.map;
|
||||
pub const sharedPhysical = shared.physical;
|
||||
@@ -1,13 +1,13 @@
|
||||
//! User-space shared memory: `shm_create` / `shm_map`. A process creates a shareable,
|
||||
//! User-space shared memory: `shared_memory_create` / `shared_memory_map`. A process creates a shareable,
|
||||
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||
//! `shm_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! `shared_memory_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||
//! generalization of capability passing from endpoints to memory objects.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||
@@ -30,7 +30,7 @@ pub fn create(len: usize) ?Region {
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shm_create)),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shared_memory_create)),
|
||||
[a0] "{rdi}" (len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
@@ -41,7 +41,7 @@ pub fn create(len: usize) ?Region {
|
||||
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||
/// Returns the pointer, or null on failure.
|
||||
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
const r = sc.systemCall1(.shm_map, handle);
|
||||
const r = sc.systemCall1(.shared_memory_map, handle);
|
||||
if (failed(r)) return null;
|
||||
return @ptrFromInt(r);
|
||||
}
|
||||
@@ -51,7 +51,7 @@ pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||
/// `attach_backing`. Returns null on failure.
|
||||
pub fn physical(handle: ipc.Handle) ?usize {
|
||||
const r = sc.systemCall1(.shm_physical, handle);
|
||||
const r = sc.systemCall1(.shared_memory_physical, handle);
|
||||
if (failed(r)) return null;
|
||||
return r;
|
||||
}
|
||||
@@ -7,9 +7,9 @@
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
|
||||
/// Everything a program receives at entry. Passed to
|
||||
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||
@@ -112,14 +112,14 @@ pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||
_ = sendSignal(id, .terminate);
|
||||
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||
_ = time.timerOnce(exit_endpoint, deadline_ms);
|
||||
var receive: [8]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||
}
|
||||
_ = system.kill(id);
|
||||
_ = kill(id);
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
@@ -135,3 +135,78 @@ pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||
pub fn subscribeExits(endpoint: usize) bool {
|
||||
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||
}
|
||||
|
||||
// --- raw process syscalls, formerly in the system.zig dumping ground ---
|
||||
|
||||
/// One `processes` entry — re-exported from the shared ABI so a program can declare its
|
||||
/// snapshot buffer without importing `abi` itself.
|
||||
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
pub fn exit(code: usize) noreturn {
|
||||
_ = sc.systemCall1(.exit, code);
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3 process,
|
||||
/// returning the child's process id (or null). argv[0] is `name`, and the caller becomes
|
||||
/// its **supervisor** — the only process allowed to `kill` it.
|
||||
pub fn spawn(name: []const u8) ?u32 {
|
||||
return spawnSupervised(name, &.{}, null);
|
||||
}
|
||||
|
||||
/// Like `spawn`, but hands the child argv[1..] (argv[0] is still `name`).
|
||||
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||
return spawnSupervised(name, arguments, null);
|
||||
}
|
||||
|
||||
/// The full spawn: argv[1..] for the child, and an optional endpoint the kernel notifies
|
||||
/// when the child ends (any way — clean exit, fault, or `kill`), delivered via
|
||||
/// `ipc.replyWait` as a child-exit badge (`ipc.Received.isChildExit`/`childProcessId`), so
|
||||
/// one endpoint can supervise many children. Returns the child's process id, or null.
|
||||
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||
var blob: [256]u8 = undefined;
|
||||
var len: usize = 0;
|
||||
for (arguments, 0..) |argument, i| {
|
||||
if (i != 0) {
|
||||
if (len >= blob.len) return null;
|
||||
blob[len] = 0;
|
||||
len += 1;
|
||||
}
|
||||
if (len + argument.len > blob.len) return null;
|
||||
@memcpy(blob[len..][0..argument.len], argument);
|
||||
len += argument.len;
|
||||
}
|
||||
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @intCast(r);
|
||||
}
|
||||
|
||||
/// Snapshot the process table into `out` and return the total number of live processes
|
||||
/// (which may exceed `out.len`; call again with a larger buffer). Kernel tasks are
|
||||
/// included, with an empty name. The primitive `ps` is built on.
|
||||
pub fn processes(out: []ProcessDescriptor) usize {
|
||||
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||
}
|
||||
|
||||
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||
pub fn isProcessRunning(name: []const u8) bool {
|
||||
var table: [32]ProcessDescriptor = undefined;
|
||||
const total = processes(&table);
|
||||
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// End process `id`. Only its supervisor — the process that spawned it — may; anyone else
|
||||
/// gets false, as does a stale or unknown id. Delivery is prompt but asynchronous, like a
|
||||
/// signal. True means the kill is accepted and irrevocable.
|
||||
pub fn kill(id: u32) bool {
|
||||
return sc.systemCall1(.process_kill, id) == 0;
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
//! The root module every user binary is compiled through (build.zig,
|
||||
//! `addUserBinary`). The program's own file is imported as `program`, and this
|
||||
//! shim contributes the declarations Zig resolves from the compilation root —
|
||||
//! `main` (dispatched by start's comptime dispatch) and the panic handler — and
|
||||
//! pulls in the `_start` entry shim. A program therefore only defines
|
||||
//! `pub fn main`; nothing else is required in its source file.
|
||||
|
||||
const start = @import("start");
|
||||
const logging = @import("logging");
|
||||
const program = @import("program");
|
||||
|
||||
/// Resolved as `@import("root").main` by start's comptime dispatch.
|
||||
pub const main = program.main;
|
||||
|
||||
/// The panic handler for every safety check in the image (start.panic).
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// std.log for every user binary goes to the tagged kernel log ring (the kernel
|
||||
/// stamps the sender; see the logging module). A program overrides by declaring
|
||||
/// its own `pub const std_options`.
|
||||
pub const std_options: @import("std").Options =
|
||||
if (@hasDecl(program, "std_options")) program.std_options else logging.default_options;
|
||||
|
||||
comptime {
|
||||
_ = &start._start; // pull the entry shim into the image
|
||||
}
|
||||
@@ -13,8 +13,8 @@
|
||||
//! is the diagnosis (see docs/ipc.md).
|
||||
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const process = @import("process.zig");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Called once with the service's endpoint before the loop starts — the
|
||||
@@ -4,8 +4,8 @@
|
||||
//! the whole runtime is linked in.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
const process = @import("process.zig");
|
||||
const logging = @import("logging");
|
||||
const process = @import("process");
|
||||
|
||||
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||
@@ -31,7 +31,7 @@ export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||
.count = stack[0],
|
||||
.vector = @ptrCast(stack + 1),
|
||||
} };
|
||||
system.exit(callMain(init));
|
||||
process.exit(callMain(init));
|
||||
}
|
||||
|
||||
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||
@@ -68,7 +68,7 @@ fn callMain(init: process.Init) u8 {
|
||||
const payload = @call(.auto, root.main, call_arguments) catch |err| {
|
||||
var buffer: [128]u8 = undefined;
|
||||
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
|
||||
_ = system.write(line);
|
||||
_ = logging.write(line);
|
||||
return 1; // distinct from panic's 127
|
||||
};
|
||||
if (@TypeOf(payload) == void) return 0;
|
||||
@@ -82,6 +82,6 @@ fn callMain(init: process.Init) u8 {
|
||||
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||
system.exit(127);
|
||||
process.exit(127);
|
||||
}
|
||||
}.panic);
|
||||
@@ -13,8 +13,19 @@
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const system = @import("system.zig");
|
||||
const sc = @import("system-call");
|
||||
|
||||
// A thread allocates its own stack straight from the mmap syscall (not through the
|
||||
// `memory` module) so `memory`'s heap can depend on this module's Mutex without a cycle.
|
||||
inline fn mmapStack(len: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, abi.prot_read | abi.prot_write);
|
||||
}
|
||||
inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
inline fn munmapStack(base: usize, len: usize) void {
|
||||
_ = sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
|
||||
/// True in a real danos binary; false when this module is compiled for host unit tests.
|
||||
/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state
|
||||
@@ -66,8 +77,8 @@ pub const Thread = struct {
|
||||
}
|
||||
};
|
||||
|
||||
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(base)) return error.SystemResources;
|
||||
const base = mmapStack(config.stack_size);
|
||||
if (mmapFailed(base)) return error.SystemResources;
|
||||
|
||||
// Top of the thread's own stack, downward: the closure, then a small per-thread TLS
|
||||
// block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is
|
||||
@@ -88,7 +99,7 @@ pub const Thread = struct {
|
||||
|
||||
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr);
|
||||
if (threadSpawnFailed(tid)) {
|
||||
_ = system.munmap(base, config.stack_size);
|
||||
munmapStack(base, config.stack_size);
|
||||
return error.SystemResources;
|
||||
}
|
||||
return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size };
|
||||
@@ -99,7 +110,7 @@ pub const Thread = struct {
|
||||
/// child-exit notification on it is this thread's.
|
||||
pub fn join(self: Thread) void {
|
||||
_ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited
|
||||
_ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||
munmapStack(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||
}
|
||||
|
||||
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
|
||||
@@ -120,6 +131,15 @@ pub const Thread = struct {
|
||||
return @intCast(sc.systemCall0(.current_core));
|
||||
}
|
||||
|
||||
/// Ask the kernel to end the calling thread. A WORKER never returns from this;
|
||||
/// the process's MAIN thread gets the kernel's refusal (-EPERM — the group ends
|
||||
/// only through exit, a fault, or process_kill, docs/shared-fate-plan.md) and
|
||||
/// the call returns. Exists for exactly that refusal path; workers end through
|
||||
/// the spawn trampoline, and a process ends through `system.exit`.
|
||||
pub fn tryExitCurrent() void {
|
||||
_ = sc.systemCall0(.thread_exit);
|
||||
}
|
||||
|
||||
/// `std.Thread.Futex`-shaped block/wake on a `u32` atomic — the primitive the
|
||||
/// blocking `Mutex`/`Condition`/`Semaphore` are built on. Waiters park in the
|
||||
/// kernel (no busy-wait), so an idle core still halts (docs/halting.md).
|
||||
@@ -11,7 +11,33 @@
|
||||
//! CLOCK_REALTIME) layered on top later.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
const sc = @import("system-call");
|
||||
|
||||
// --- raw syscall wrappers, formerly in the system.zig dumping ground ---
|
||||
|
||||
/// Monotonic nanoseconds since boot — the raw reading; `now()` wraps it in an `Instant`.
|
||||
/// Never runs backward. Not wall-clock time (see `wallClock`).
|
||||
pub fn clock() u64 {
|
||||
return @intCast(sc.systemCall0(.clock));
|
||||
}
|
||||
|
||||
/// Wall-clock time in Unix epoch seconds (UTC), from the RTC — the real date/time, what a
|
||||
/// filesystem stamps as an mtime. Unlike `clock` (monotonic since boot), this is calendar time.
|
||||
pub fn wallClock() u64 {
|
||||
return @intCast(sc.systemCall0(.wall_clock));
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds — the raw, coarse, allocation-free form.
|
||||
pub fn sleepMillis(ms: u64) void {
|
||||
_ = sc.systemCall1(.sleep, ms);
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer: after `ms` the kernel posts a timer notification
|
||||
/// (`ipc.Received.isTimer`) to `endpoint` (a handle from `ipc.createIpcEndpoint`). Unlike
|
||||
/// `sleep`, does not block — a service keeps serving IPC while the deadline is pending.
|
||||
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||
}
|
||||
|
||||
const nanos_per_micro: u64 = 1_000;
|
||||
const nanos_per_milli: u64 = 1_000_000;
|
||||
@@ -90,32 +116,27 @@ pub const Instant = struct {
|
||||
|
||||
/// The current monotonic time.
|
||||
pub fn now() Instant {
|
||||
return .{ .ns = system.clock() };
|
||||
return .{ .ns = clock() };
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||
/// want a plain integer instead of an `Instant`.
|
||||
pub fn monotonicNanos() u64 {
|
||||
return system.clock();
|
||||
return clock();
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||
/// "unavailable" instead of assuming the clock advances.
|
||||
pub fn available() bool {
|
||||
return system.clock() != 0;
|
||||
return clock() != 0;
|
||||
}
|
||||
|
||||
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||
/// `spin`.
|
||||
/// `spin`. (The raw millisecond form is `sleepMillis`.)
|
||||
pub fn sleep(d: Duration) void {
|
||||
system.sleep(d.ceilMillis());
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||
pub fn sleepMillis(ms: u64) void {
|
||||
system.sleep(ms);
|
||||
sleepMillis(d.ceilMillis());
|
||||
}
|
||||
|
||||
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||
@@ -126,13 +147,10 @@ pub fn spin(d: Duration) void {
|
||||
while (!deadline.reached()) {}
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||
/// The ergonomic `Duration` form of `timerOnce`: arm a one-shot timer against `endpoint`
|
||||
/// for `d` (rounded up to milliseconds). Returns false if the timer could not be armed.
|
||||
pub fn after(endpoint: usize, d: Duration) bool {
|
||||
return system.timerOnce(endpoint, d.ceilMillis());
|
||||
return timerOnce(endpoint, d.ceilMillis());
|
||||
}
|
||||
|
||||
test "Duration unit conversions round toward zero" {
|
||||
+28
-1
@@ -11,6 +11,19 @@
|
||||
/// startup instead of quiet corruption later.
|
||||
pub const version: u16 = 1;
|
||||
|
||||
/// Which bus a `child_added` came from — stated by the reporting bus driver so
|
||||
/// the manager's /etc/devices.csv matcher knows how to read the report's identity
|
||||
/// (a PCI class triple vs a USB class triple are the same 24 bits but different
|
||||
/// namespaces) and which `bus` column a rule must name to bind it. `unknown` is
|
||||
/// the zero default, so an un-upgraded reporter fails to match rather than
|
||||
/// binding to the wrong bus's rule.
|
||||
pub const BusKind = enum(u8) {
|
||||
unknown = 0,
|
||||
pci = 1,
|
||||
usb = 2,
|
||||
acpi = 3,
|
||||
};
|
||||
|
||||
/// What kind of driver is talking (docs/driver-model.md's shapes).
|
||||
pub const Role = enum(u8) {
|
||||
/// Owns a controller and reports the devices behind it (`child_added`).
|
||||
@@ -66,7 +79,9 @@ pub const reply_size = @sizeOf(HelloReply);
|
||||
/// restarted instance rediscovers and re-reports.
|
||||
pub const ChildAdded = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.child_added),
|
||||
reserved0: u8 = 0,
|
||||
/// A `BusKind` value: which bus reported this child, so the manager reads the
|
||||
/// identity in the right namespace and matches against the right `bus` column.
|
||||
bus: u8 = @intFromEnum(BusKind.unknown),
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
/// The reporting driver's own device (the controller) — the child's parent.
|
||||
@@ -80,6 +95,18 @@ pub const ChildAdded = extern struct {
|
||||
/// manager hands a matched driver as its argv assignment — or `no_device`
|
||||
/// for an unregistered leaf (a USB port before the descriptor track).
|
||||
device_id: u64 = no_device,
|
||||
/// The vendor id (PCI vendor / USB idVendor), or 0 when the bus has no such
|
||||
/// concept (ACPI). Carried so the manager's /etc/devices.csv matcher can bind
|
||||
/// on vendor — a level the bus-native `identity` (a class triple) cannot express.
|
||||
vendor: u16 = 0,
|
||||
/// The device id (PCI device / USB idProduct), or 0. The most specific numeric
|
||||
/// level: this is what lets one virtio-gpu (1AF4:1050) be told from any other
|
||||
/// virtio display function without the driver re-confirming after it is spawned.
|
||||
device: u16 = 0,
|
||||
/// The PCI subsystem id, packed `(subsystem_vendor << 16) | subsystem_device`
|
||||
/// (so it reads vendor-first, matching the CSV's `ssvid:ssid`), or 0 when the
|
||||
/// device has no subsystem id (a bridge, or a non-PCI bus).
|
||||
subsystem: u32 = 0,
|
||||
/// The ACPI hardware id (`_HID`), EISA-decoded (e.g. "PNP0303"), for devices
|
||||
/// discovered by firmware string rather than a numeric bus identity. Empty
|
||||
/// (all zero) otherwise. Widens for FDT `compatible` strings later.
|
||||
+7
-5
@@ -24,11 +24,13 @@ pub const Operation = enum(u32) {
|
||||
damage = 6,
|
||||
/// present(): composite the dirty layers and flush to the screen.
|
||||
present = 7,
|
||||
/// attach_scanout(x=stride, width, height, colour=format) + <surface capability>: a native
|
||||
/// scanout driver announces itself, handing over the shared scanout surface as an `ipc_call`
|
||||
/// send_cap. The compositor maps it, looks up the driver's `.scanout` present channel, and
|
||||
/// upgrades off the GOP floor (docs/display-v2.md V4). `x` is the surface's row stride in
|
||||
/// pixels, `colour` the DisplayFormat.
|
||||
/// attach_scanout(x=stride, y=refresh_hz, width, height, colour=format) + <surface
|
||||
/// capability>: a native scanout driver announces itself, handing over the shared scanout
|
||||
/// surface as an `ipc_call` send_cap. The compositor maps it, looks up the driver's
|
||||
/// `.scanout` present channel, and upgrades off the GOP floor (docs/display-v2.md V4).
|
||||
/// `x` is the surface's row stride in pixels, `y` the panel refresh rate from the
|
||||
/// driver's EDID read (0 = unknown; paces the compositor's frame clock), `colour` the
|
||||
/// DisplayFormat.
|
||||
attach_scanout = 8,
|
||||
/// set_mode(width, height): change the display resolution — only a native backend that
|
||||
/// reports `canModeSet` honours it; on the GOP floor it fails (docs/display-v2.md V5).
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user