Compare commits
33
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0a4388c3bc | ||
|
|
0045fd87ba | ||
|
|
ad0fd52cb8 | ||
|
|
65a44a5568 | ||
|
|
e186858315 | ||
|
|
7c5645fe48 | ||
|
|
127ea2dad9 | ||
|
|
30d6ea622a | ||
|
|
d0c1b3e45b | ||
|
|
f480c5d790 | ||
|
|
9f18d8340e | ||
|
|
0a84e52bf8 | ||
|
|
1cf9985da6 | ||
|
|
2d0858caf6 | ||
|
|
ec6e888076 | ||
|
|
4f02f75602 | ||
|
|
16618d2cdc | ||
|
|
15107f54be | ||
|
|
23f915c593 | ||
|
|
981a4af7e0 | ||
|
|
5ab7263c9c | ||
|
|
7f415e724f | ||
|
|
cf140eb772 | ||
|
|
e2dddc941f | ||
|
|
28b3635979 | ||
|
|
6101e429ba | ||
|
|
a4e44e8f31 | ||
|
|
c7e9b5a4f6 | ||
|
|
6bc329456a | ||
|
|
ed3b3f1c45 | ||
|
|
8259678f0a | ||
|
|
f4813c8e99 | ||
|
|
8b7f1d009c |
+144
-49
@@ -2,6 +2,7 @@ const std = @import("std");
|
||||
const uefi = std.os.uefi;
|
||||
const elf = std.elf;
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||
@@ -15,11 +16,11 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
/// The kernel image: /system/kernel.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||
|
||||
/// The init program: /system/services/init.
|
||||
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||
|
||||
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||
/// The user binaries: everything under /system except the kernel itself. The
|
||||
/// loader walks this tree and packs it into the in-RAM initial_ramdisk image —
|
||||
/// the volume's file structure is the single source of truth (no packed image
|
||||
/// artifact on disk).
|
||||
const system_directory_name = std.unicode.utf8ToUtf16LeStringLiteral("system");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
@@ -64,20 +65,14 @@ fn boot() !noreturn {
|
||||
|
||||
const entry = try loadKernel(bs, &boot_information);
|
||||
|
||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_information) catch |err| {
|
||||
log("EFI: no /system/services/init (");
|
||||
// Best effort: a volume without a /system tree of user binaries still boots
|
||||
// (kernel-only). The tree — init included — becomes the initial_ramdisk.
|
||||
loadSystemTree(bs, &boot_information) catch |err| {
|
||||
log("EFI: no /system binaries (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||
log("EFI: no initial_ramdisk (");
|
||||
logBytes(@errorName(err));
|
||||
log(")\r\n");
|
||||
};
|
||||
|
||||
// Build the page tables the kernel starts life on: identity + a physmap of
|
||||
// low RAM, plus the higher-half kernel image once it links high. Allocated
|
||||
// now, while boot services (and the memory map) are still stable — nothing
|
||||
@@ -97,7 +92,7 @@ fn boot() !noreturn {
|
||||
}
|
||||
|
||||
/// A display resolution in pixels.
|
||||
const Resolution = struct { width: u32, height: u32 };
|
||||
const Resolution = struct { width: u32, height: u32, refresh_hz: u32 };
|
||||
|
||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||
/// and read the resulting graphics mode into our own framebuffer description.
|
||||
@@ -128,6 +123,10 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||
// Each pixel is 32 bits, so the byte pitch is 4 * pixels-per-row.
|
||||
.pitch = info.pixels_per_scan_line * 4,
|
||||
.format = try pixelFormat(info.pixel_format),
|
||||
// The refresh rate rides the EDID preferred timing. If the firmware kept a
|
||||
// non-native mode it may not describe that mode exactly — but it is the panel's
|
||||
// own clock, a far better frame-clock seed than a hardcoded 60 Hz.
|
||||
.refresh_hz = if (native) |n| n.refresh_hz else 0,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -176,10 +175,12 @@ fn nativeResolution(bs: *uefi.tables.BootServices, handles: []uefi.Handle) ?Reso
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Parse the native resolution from a raw EDID block. The first Detailed Timing
|
||||
/// Descriptor (at byte 54) is the preferred — i.e. native — mode by convention;
|
||||
/// its active pixel counts are split across low bytes and the high nibbles of
|
||||
/// later bytes.
|
||||
/// Parse the native resolution and refresh rate from a raw EDID block. The first
|
||||
/// Detailed Timing Descriptor (at byte 54) is the preferred — i.e. native — mode by
|
||||
/// convention; its active pixel counts are split across low bytes and the high nibbles
|
||||
/// of later bytes. The refresh rate is derived, not stored: the descriptor carries the
|
||||
/// pixel clock (10 kHz units) and the active+blanking extents, and
|
||||
/// refresh = clock / (horizontal total × vertical total).
|
||||
fn edidNative(edid: []const u8) ?Resolution {
|
||||
if (edid.len < 128) return null;
|
||||
// Every EDID begins with this fixed 8-byte header.
|
||||
@@ -193,7 +194,13 @@ fn edidNative(edid: []const u8) ?Resolution {
|
||||
const w = @as(u32, dtd[2]) | (@as(u32, dtd[4] & 0xf0) << 4);
|
||||
const h = @as(u32, dtd[5]) | (@as(u32, dtd[7] & 0xf0) << 4);
|
||||
if (w == 0 or h == 0) return null;
|
||||
return .{ .width = w, .height = h };
|
||||
|
||||
const clock_hz = (@as(u64, dtd[0]) | (@as(u64, dtd[1]) << 8)) * 10_000;
|
||||
const h_blank = @as(u64, dtd[3]) | (@as(u64, dtd[4] & 0x0f) << 8);
|
||||
const v_blank = @as(u64, dtd[6]) | (@as(u64, dtd[7] & 0x0f) << 8);
|
||||
const total = (@as(u64, w) + h_blank) * (@as(u64, h) + v_blank);
|
||||
const refresh: u32 = if (total == 0) 0 else @intCast((clock_hz + total / 2) / total);
|
||||
return .{ .width = w, .height = h, .refresh_hz = refresh };
|
||||
}
|
||||
|
||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||
@@ -357,11 +364,28 @@ fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) nor
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Read a whole file off the boot volume into a pool buffer that outlives the
|
||||
/// loader. The buffer is deliberately NOT freed: it's LoaderData, which the
|
||||
/// memory-map conversion classifies as reserved, so the kernel identity-maps it
|
||||
/// and reads from there. Returns the buffer (pointer + length).
|
||||
fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
// --- the /system tree -> initial_ramdisk ------------------------------------
|
||||
|
||||
/// Cap on bundled binaries. Generous: the tree carries ~30 today.
|
||||
const maximum_bundled = 64;
|
||||
|
||||
/// How deep the walk goes below /system ("/system/services/x" is depth 1).
|
||||
const maximum_tree_depth = 3;
|
||||
|
||||
/// One binary discovered under /system: its FHS path (UTF-8, '/'-separated,
|
||||
/// NUL-free) and its contents in a transient pool buffer.
|
||||
const Bundled = struct {
|
||||
path: [initial_ramdisk.maximum_name]u8,
|
||||
path_len: usize,
|
||||
data: []align(8) u8,
|
||||
};
|
||||
|
||||
/// Walk the boot volume's /system tree and pack every regular file (except the
|
||||
/// kernel image itself — the only top-level file) into an in-RAM v2
|
||||
/// initial_ramdisk image, entries named by full FHS path. This is what makes the
|
||||
/// volume's file structure the single source of truth: there is no packed
|
||||
/// ramdisk artifact on disk, and init travels in the table like everything else.
|
||||
fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
@@ -371,40 +395,111 @@ fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
const root = try fs.openVolume();
|
||||
defer _ = root.close() catch {};
|
||||
|
||||
const file = try root.open(name, .read, .{});
|
||||
defer _ = file.close() catch {};
|
||||
const system_directory = try root.open(system_directory_name, .read, .{});
|
||||
defer _ = system_directory.close() catch {};
|
||||
|
||||
var list: [maximum_bundled]Bundled = undefined;
|
||||
var count: usize = 0;
|
||||
try walkDirectory(bs, system_directory, "/system", 0, &list, &count);
|
||||
if (count == 0) return error.NoBinaries;
|
||||
|
||||
// Assemble the v2 image: header, entry table, then the blobs.
|
||||
const table_end = @sizeOf(initial_ramdisk.Header) + count * @sizeOf(initial_ramdisk.Entry);
|
||||
var total: usize = table_end;
|
||||
for (list[0..count]) |e| total += e.data.len;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, total); // survives the handoff
|
||||
std.mem.bytesAsValue(initial_ramdisk.Header, image[0..@sizeOf(initial_ramdisk.Header)]).* = .{
|
||||
.magic = initial_ramdisk.magic,
|
||||
.count = @intCast(count),
|
||||
};
|
||||
var offset: usize = table_end;
|
||||
for (list[0..count], 0..) |e, i| {
|
||||
var record = initial_ramdisk.Entry{ .name = @splat(0), .offset = offset, .len = e.data.len };
|
||||
@memcpy(record.name[0..e.path_len], e.path[0..e.path_len]);
|
||||
const slot = image[@sizeOf(initial_ramdisk.Header) + i * @sizeOf(initial_ramdisk.Entry) ..][0..@sizeOf(initial_ramdisk.Entry)];
|
||||
std.mem.bytesAsValue(initial_ramdisk.Entry, slot).* = record;
|
||||
@memcpy(image[offset..][0..e.data.len], e.data);
|
||||
offset += e.data.len;
|
||||
_ = bs.freePool(e.data.ptr) catch {};
|
||||
}
|
||||
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = total;
|
||||
progress("EFI: /system tree loaded\r\n");
|
||||
}
|
||||
|
||||
/// Recursively collect the regular files below `directory` into `list`. Top-level
|
||||
/// files (depth 0) are skipped: the only one is /system/kernel, which loadKernel
|
||||
/// has already consumed and which is not a spawnable user binary.
|
||||
fn walkDirectory(
|
||||
bs: *uefi.tables.BootServices,
|
||||
directory: *uefi.protocol.File,
|
||||
prefix: []const u8,
|
||||
depth: usize,
|
||||
list: *[maximum_bundled]Bundled,
|
||||
count: *usize,
|
||||
) !void {
|
||||
// Each read() on a directory yields one EFI_FILE_INFO; zero bytes means done.
|
||||
var info_buffer: [1024]u8 align(8) = undefined;
|
||||
while (true) {
|
||||
const n = try directory.read(&info_buffer);
|
||||
if (n == 0) return;
|
||||
const info: *const uefi.protocol.File.Info.File = @ptrCast(@alignCast(&info_buffer));
|
||||
const name16 = info.getFileName();
|
||||
|
||||
// Convert the (ASCII in practice) UTF-16 name; skip "." and "..".
|
||||
var name_buffer: [initial_ramdisk.maximum_name]u8 = undefined;
|
||||
var name_length: usize = 0;
|
||||
while (name16[name_length] != 0) : (name_length += 1) {
|
||||
if (name_length == name_buffer.len) return error.NameTooLong;
|
||||
const c = name16[name_length];
|
||||
if (c > 0x7F) return error.UnsupportedName;
|
||||
name_buffer[name_length] = @intCast(c);
|
||||
}
|
||||
const name = name_buffer[0..name_length];
|
||||
if (std.mem.eql(u8, name, ".") or std.mem.eql(u8, name, "..")) continue;
|
||||
|
||||
if (info.attribute.directory) {
|
||||
if (depth == maximum_tree_depth) continue;
|
||||
var child_prefix: [initial_ramdisk.maximum_name]u8 = undefined;
|
||||
const child = try std.fmt.bufPrint(&child_prefix, "{s}/{s}", .{ prefix, name });
|
||||
const child_directory = try directory.open(name16, .read, .{});
|
||||
defer _ = child_directory.close() catch {};
|
||||
try walkDirectory(bs, child_directory, child, depth + 1, list, count);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (depth == 0) continue; // /system/kernel — already loaded, not bundled
|
||||
if (count.* == maximum_bundled) return error.TooManyBinaries;
|
||||
|
||||
var entry: *Bundled = &list[count.*];
|
||||
const path = try std.fmt.bufPrint(&entry.path, "{s}/{s}", .{ prefix, name });
|
||||
entry.path_len = path.len;
|
||||
|
||||
const file = try directory.open(name16, .read, .{});
|
||||
defer _ = file.close() catch {};
|
||||
entry.data = try readWholeFile(bs, file);
|
||||
count.* += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Read an open file completely into a fresh pool buffer that survives the
|
||||
/// handoff (LoaderData is classified reserved, so the kernel identity-maps it).
|
||||
fn readWholeFile(bs: *uefi.tables.BootServices, file: *uefi.protocol.File) ![]align(8) u8 {
|
||||
try file.setPosition(seek_end);
|
||||
const size: usize = @intCast(try file.getPosition());
|
||||
try file.setPosition(0);
|
||||
if (size == 0) return error.EmptyFile;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||
|
||||
const buffer = try bs.allocatePool(.loader_data, size);
|
||||
var read_total: usize = 0;
|
||||
while (read_total < size) {
|
||||
const n = try file.read(image[read_total..]);
|
||||
const n = try file.read(buffer[read_total..]);
|
||||
if (n == 0) return error.UnexpectedEof;
|
||||
read_total += n;
|
||||
}
|
||||
return image[0..size];
|
||||
}
|
||||
|
||||
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
progress("EFI: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
progress("EFI: initial_ramdisk loaded\r\n");
|
||||
return buffer[0..size];
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
|
||||
@@ -221,18 +221,21 @@ fn addKernel(
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||
/// bundle its own serial kernel while sharing the loader, init, and ramdisk — all
|
||||
/// built once per invocation (the loader's boot breadcrumbs and init's heartbeat
|
||||
/// both follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
/// One user binary and its FHS home on the boot volume (and in zig-out).
|
||||
const BundledBinary = struct { path: []const u8, binary: std.Build.LazyPath };
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding the
|
||||
/// EFI stub, the kernel, and every user binary at its FHS path — the volume's
|
||||
/// /system tree IS the system image; the EFI loader walks it at boot and builds
|
||||
/// the in-RAM initial_ramdisk from it. Factored so the serial-enabled
|
||||
/// `run-x86-64` variant can bundle its own serial kernel while sharing the
|
||||
/// loader and user tree (the loader's boot breadcrumbs and init's heartbeat both
|
||||
/// follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
init_bin: std.Build.LazyPath,
|
||||
initial_ramdisk_img: std.Build.LazyPath,
|
||||
bundled: []const BundledBinary,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
@@ -242,10 +245,10 @@ fn addBootImage(
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/services/init");
|
||||
mk_fat.addFileArg(init_bin);
|
||||
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||
mk_fat.addFileArg(initial_ramdisk_img);
|
||||
for (bundled) |item| {
|
||||
mk_fat.addArg(item.path);
|
||||
mk_fat.addFileArg(item.binary);
|
||||
}
|
||||
return fat_image;
|
||||
}
|
||||
|
||||
@@ -361,7 +364,7 @@ pub fn build(b: *std.Build) void {
|
||||
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||
// expose theirs the same way.
|
||||
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||
.root_source_file = b.path("system/vfs-protocol.zig"),
|
||||
});
|
||||
|
||||
// The input wire protocol: the input service's public interface, exposed as its own
|
||||
@@ -443,8 +446,9 @@ pub fn build(b: *std.Build) void {
|
||||
},
|
||||
});
|
||||
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and
|
||||
// the EFI loader (packs it in RAM from the boot volume's /system tree). No
|
||||
// dependencies.
|
||||
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||
});
|
||||
@@ -502,15 +506,12 @@ pub fn build(b: *std.Build) void {
|
||||
const init_options = b.addOptions();
|
||||
init_options.addOption(bool, "serial", serial);
|
||||
programModule(init_exe).addImport("build_options", init_options.createModule());
|
||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||
b.getInstallStep().dependOn(&init_install.step);
|
||||
|
||||
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||
// Each is built by the same user-binary recipe, then packed into one image by
|
||||
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
// --- the rest of the /system tree: services, drivers, test fixtures ---
|
||||
// Each is built by the same user-binary recipe and laid out at its FHS path on
|
||||
// the boot volume (see `bundled` below). The EFI loader walks the tree at boot
|
||||
// and hands the kernel an in-RAM initial_ramdisk of it (system/initial-ramdisk.zig).
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs-test/vfs-test.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
@@ -535,11 +536,13 @@ pub fn build(b: *std.Build) void {
|
||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||
const display_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||
// Threaded: the display runs a mouse-listener thread alongside its compositor loop
|
||||
// (docs/threading.md, docs/display.md), so it opts into real atomics/TLS.
|
||||
const display_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||
const shm_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-server", "system/services/shm-server/shm-server.zig");
|
||||
const shm_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-client", "system/services/shm-client/shm-client.zig");
|
||||
const shared_memory_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shared-memory-server", "system/services/shared-memory-server/shared-memory-server.zig");
|
||||
const shared_memory_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shared-memory-client", "system/services/shared-memory-client/shared-memory-client.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
// The PCI bus driver decodes each function's class triple to human names in its
|
||||
@@ -577,99 +580,55 @@ pub fn build(b: *std.Build) void {
|
||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||
const logger_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "logger", "system/services/logger/logger.zig");
|
||||
// The first multi-threaded binary: exercises runtime.Thread over the thread ABI
|
||||
// (docs/threading.md). Built threaded so its shared-memory poll is real.
|
||||
const thread_test_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "thread-test", "system/services/thread-test/thread-test.zig");
|
||||
|
||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||
mk_run.addArg("vfs");
|
||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||
mk_run.addArg("vfs-test");
|
||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-bus");
|
||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-keyboard");
|
||||
mk_run.addFileArg(ps2_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-mouse");
|
||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-xhci-bus");
|
||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-keyboard");
|
||||
mk_run.addFileArg(usb_hid_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-hid-mouse");
|
||||
mk_run.addFileArg(usb_hid_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-storage");
|
||||
mk_run.addFileArg(usb_storage_exe.getEmittedBin());
|
||||
mk_run.addArg("fat");
|
||||
mk_run.addFileArg(fat_exe.getEmittedBin());
|
||||
mk_run.addArg("fat-test");
|
||||
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
||||
mk_run.addArg("display");
|
||||
mk_run.addFileArg(display_exe.getEmittedBin());
|
||||
mk_run.addArg("display-demo");
|
||||
mk_run.addFileArg(display_demo_exe.getEmittedBin());
|
||||
mk_run.addArg("virtio-gpu");
|
||||
mk_run.addFileArg(virtio_gpu_exe.getEmittedBin());
|
||||
mk_run.addArg("shm-server");
|
||||
mk_run.addFileArg(shm_server_exe.getEmittedBin());
|
||||
mk_run.addArg("shm-client");
|
||||
mk_run.addFileArg(shm_client_exe.getEmittedBin());
|
||||
mk_run.addArg("pci-bus");
|
||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("crash-test");
|
||||
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||
mk_run.addArg("thread-test");
|
||||
mk_run.addFileArg(thread_test_exe.getEmittedBin());
|
||||
mk_run.addArg("device-list");
|
||||
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||
mk_run.addArg("discovery");
|
||||
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||
mk_run.addArg("device-manager");
|
||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||
mk_run.addArg("input");
|
||||
mk_run.addFileArg(input_exe.getEmittedBin());
|
||||
mk_run.addArg("input-source");
|
||||
mk_run.addFileArg(input_source_exe.getEmittedBin());
|
||||
mk_run.addArg("input-test");
|
||||
mk_run.addFileArg(input_test_exe.getEmittedBin());
|
||||
mk_run.addArg("args-echo");
|
||||
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||
mk_run.addArg("process-test");
|
||||
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||
mk_run.addArg("log-flush");
|
||||
mk_run.addFileArg(log_flush_exe.getEmittedBin());
|
||||
// Every user binary and its FHS home on the boot volume. There is no packed
|
||||
// ramdisk artifact any more: make-fat-image.py lays each binary out at this
|
||||
// path on the image, and the EFI loader walks /system at boot and builds the
|
||||
// in-RAM initial_ramdisk table from the tree — the volume's file structure is
|
||||
// the single source of truth. Entry names (and hence argv[0] and task names)
|
||||
// are these paths with a leading slash.
|
||||
const bundled = [_]BundledBinary{
|
||||
.{ .path = "system/services/init", .binary = init_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/fat", .binary = fat_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/display", .binary = display_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/display-demo", .binary = display_demo_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/device-manager", .binary = device_manager_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/input", .binary = input_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/discovery", .binary = discovery_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/logger", .binary = logger_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-bus", .binary = ps2_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-keyboard", .binary = ps2_keyboard_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/ps2-mouse", .binary = ps2_mouse_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-xhci-bus", .binary = usb_xhci_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-hid-keyboard", .binary = usb_hid_keyboard_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-hid-mouse", .binary = usb_hid_mouse_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/usb-storage", .binary = usb_storage_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/virtio-gpu", .binary = virtio_gpu_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/pci-bus", .binary = pci_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/vfs-test", .binary = vfstest_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/fat-test", .binary = fat_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/shared-memory-server", .binary = shared_memory_server_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/shared-memory-client", .binary = shared_memory_client_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/crash-test", .binary = crash_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/device-list", .binary = device_list_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/input-source", .binary = input_source_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/input-test", .binary = input_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/args-echo", .binary = args_echo_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/process-test", .binary = process_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/thread-test", .binary = thread_test_exe.getEmittedBin() },
|
||||
};
|
||||
|
||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||
.{ vfs_exe, "system/services" },
|
||||
.{ device_manager_exe, "system/services" },
|
||||
.{ input_exe, "system/services" },
|
||||
.{ ps2_bus_exe, "system/drivers" },
|
||||
.{ ps2_keyboard_exe, "system/drivers" },
|
||||
.{ ps2_mouse_exe, "system/drivers" },
|
||||
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||
.{ usb_hid_keyboard_exe, "system/drivers" },
|
||||
.{ usb_hid_mouse_exe, "system/drivers" },
|
||||
.{ usb_storage_exe, "system/drivers" },
|
||||
.{ fat_exe, "system/services" },
|
||||
.{ display_exe, "system/services" },
|
||||
.{ log_flush_exe, "system/services" },
|
||||
}) |entry| {
|
||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||
b.getInstallStep().dependOn(&step.step);
|
||||
// Install every bundled binary to its FHS home, so zig-out is a true image of
|
||||
// the filesystem — the same tree make-fat-image.py lays out on the boot volume.
|
||||
for (bundled) |item| {
|
||||
const install = b.addInstallFileWithDir(item.binary, .prefix, item.path);
|
||||
b.getInstallStep().dependOn(&install.step);
|
||||
}
|
||||
|
||||
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||
|
||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||
@@ -690,8 +649,10 @@ pub fn build(b: *std.Build) void {
|
||||
}),
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||
// The bootloader speaks the handoff contract and the ramdisk
|
||||
// container it packs the /system tree into — never the user ABI.
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||
.{ .name = "build_options", .module = loader_options_module },
|
||||
},
|
||||
}),
|
||||
@@ -704,11 +665,11 @@ pub fn build(b: *std.Build) void {
|
||||
|
||||
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||
// holding exactly what the firmware and bootloader need off the ESP: the EFI
|
||||
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||
// danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, exe.getEmittedBin(), efiexe.getEmittedBin(), &bundled);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
@@ -717,7 +678,7 @@ pub fn build(b: *std.Build) void {
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), init_exe.getEmittedBin(), initial_ramdisk_img);
|
||||
const fat_image_serial = addBootImage(b, exe_serial.getEmittedBin(), efiexe.getEmittedBin(), &bundled);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
@@ -848,6 +809,53 @@ pub fn build(b: *std.Build) void {
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// --- run-x86-64-gpu: the same boot plus a virtio-gpu adapter ---
|
||||
// The VGA device still supplies the boot (GOP) framebuffer the compositor starts
|
||||
// on; the virtio-gpu function is discovered by the device-manager stack, its
|
||||
// driver announces a shared scanout, and the compositor upgrades off the GOP
|
||||
// floor to fenced, tear-free native presents (docs/display-v2.md).
|
||||
// This is the interactive twin of the `display-native` test case, and 512M
|
||||
// matches it (the whole driver stack + the compositor's surfaces at once).
|
||||
// QEMU shows one head per adapter: pick the virtio-gpu head in the View menu
|
||||
// to watch the native output.
|
||||
const run_gpu = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"512M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_gpu.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
"-device",
|
||||
"virtio-gpu-pci",
|
||||
});
|
||||
const gpu_serial_log = b.fmt("{s}/run-x86-64-gpu-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_gpu.addArgs(&.{ "-serial", b.fmt("file:{s}", .{gpu_serial_log}) });
|
||||
run_gpu.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_gpu_step = b.step("run-x86-64-gpu", "Boot in QEMU with a virtio-gpu adapter: the compositor upgrades to fenced (tear-free) native presents; watch the virtio-gpu head in QEMU's View menu");
|
||||
run_gpu_step.dependOn(&run_gpu.step);
|
||||
|
||||
// const run_cmd = b.addRunArtifact(exe);
|
||||
// const run_step = b.step("run", "Run the app");
|
||||
// run_step.dependOn(&run_cmd.step);
|
||||
@@ -865,6 +873,7 @@ pub fn build(b: *std.Build) void {
|
||||
for ([_][]const u8{
|
||||
"system/boot-handoff.zig",
|
||||
"system/abi.zig",
|
||||
"system/initial-ramdisk.zig", // v2 path-named entries: find/basename/magic
|
||||
"system/devices/device-abi.zig",
|
||||
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||
@@ -877,8 +886,7 @@ pub fn build(b: *std.Build) void {
|
||||
"system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||
"system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||
"system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian)
|
||||
"system/services/vfs/path.zig", // mount-prefix path matching
|
||||
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"system/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||
"system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile
|
||||
@@ -911,6 +919,21 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
|
||||
// The tagged kernel log ring: append/wrap/reclaim/sequence-gap behavior over
|
||||
// a RAM buffer. Needs the `abi` module (record header layout), so it doesn't
|
||||
// fit the plain loop above.
|
||||
const log_ring_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/log-ring.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(log_ring_tests).step);
|
||||
|
||||
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||
// loop above.
|
||||
@@ -926,6 +949,22 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||
|
||||
// runtime.Thread's lock/condvar state machines (Mutex/Condition/RwLock/WaitGroup). Its
|
||||
// Futex seam falls back to std.Thread.Futex off the danos target, so the tests exercise
|
||||
// them with real host threads (docs/threading-plan.md M11). Like time.zig it pulls in
|
||||
// system.zig (syscall wrappers), which needs the `abi` module.
|
||||
const thread_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/thread.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(thread_tests).step);
|
||||
|
||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||
|
||||
+1
-1
@@ -107,7 +107,7 @@ Start with the north star:
|
||||
- **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
`runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||
a `thread_spawn` syscall, futex-backed blocking, aspace refcounting. Why it's the
|
||||
a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the
|
||||
native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](threading-plan.md).
|
||||
|
||||
@@ -134,7 +134,7 @@ Prove the pipeline end-to-end from a separate process.
|
||||
every IPC message at `MESSAGE_MAXIMUM` = 256 — so `replyWait` rejected the oversized
|
||||
receive buffer with `-E2BIG` and the serve loop had been *spinning* since D2 (unseen,
|
||||
as D2/D3 matched init-time heartbeats). Set it to 256; `blit_tile` is now explicitly
|
||||
a small-tile path (≤ 54 px inline), larger bitmaps being the deferred shm surface.
|
||||
a small-tile path (≤ 54 px inline), larger bitmaps being the deferred shared-memory surface.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display-demo` spawns the service + `display-demo`;
|
||||
the demo drives a run of frames of motion through the layer client API and logs
|
||||
@@ -172,7 +172,7 @@ documented (docs/display.md): no runtime mode-setting (native backend) and no tr
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Shared-memory surfaces** — generalize M13 capability passing to memory objects
|
||||
(`shm_create`/`shm_map`), so bitmap clients hand the compositor a rendered surface
|
||||
(`shared_memory_create`/`shared_memory_map`), so bitmap clients hand the compositor a rendered surface
|
||||
instead of drawing commands. The compositor's layer model already anticipates it.
|
||||
- **Native backend (Bochs DISPI, then virtio-gpu)** — behind the same internal backend
|
||||
interface as the dumb framebuffer: EDID mode list + runtime resolution/bpp change +
|
||||
|
||||
+25
-23
@@ -6,11 +6,11 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **First native backend = virtio-gpu** (VM standard: mode-set + present/flush + vsync).
|
||||
- **First native backend = virtio-gpu** (VM standard: mode-set + fenced present/flush).
|
||||
- **Dynamic hot-attach**: boot on GOP, upgrade to native when the driver **announces**
|
||||
(push, not polling); re-attach across driver restarts; GOP is the floor for "no driver
|
||||
ever," not a live fall-back after a reprogram.
|
||||
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||
- **v2 builds the shared-memory capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
- The compositor's layers/back-buffer/damage are **unchanged**; only scanout is pluggable.
|
||||
|
||||
@@ -26,7 +26,7 @@ kebab-case file names, no `Co-Authored-By` trailers. New user binaries go throug
|
||||
|
||||
**Every gate is serial-checkable — no screenshots** (this plan is built to run unattended).
|
||||
Where "does it actually display" would otherwise need a human eyeball, the code **reads its
|
||||
own pixels back**: the scanout resource is CPU-visible RAM (shm-backed) and the back buffer
|
||||
own pixels back**: the scanout resource is CPU-visible RAM (shared-memory-backed) and the back buffer
|
||||
is cacheable, so a driver/compositor can write a known value, read it back, and log a
|
||||
pass/fail — and a virtio `resource_flush` is confirmed by the device **acking it on the
|
||||
used ring**. Those two together (pixel-readback + flush-ack) are the automated stand-in for
|
||||
@@ -46,7 +46,7 @@ Extract scanout from the compositor so today's path becomes one backend among fu
|
||||
|
||||
- [x] `system/services/display/backend.zig`: a `Backend` tagged union with `info()`,
|
||||
`surface()` (the cacheable compose target), `present(damage)`, and capability flags
|
||||
(`canModeSet`/`hasVsync`, both false for GOP).
|
||||
(`canModeSet`/`hasFencedPresent`, both false for GOP).
|
||||
- [x] The v1 GOP path is now `backend.Gop` (claims the `display` node, WC-maps the LFB,
|
||||
keeps the cacheable back buffer, `present` = the damage-rect WC copy). display.zig
|
||||
composes into `backend.surface()` and calls `backend.present(damage)` — no LFB or
|
||||
@@ -57,25 +57,25 @@ Extract scanout from the compositor so today's path becomes one backend among fu
|
||||
**Gate (met):** `display-service` + `display-demo` pass **unchanged** (pure refactor; GOP
|
||||
is the only backend), and `zig build test` stays green.
|
||||
|
||||
## V2 — The `shm` cross-process memory capability (kernel) ✅
|
||||
## V2 — The shared-memory cross-process capability (kernel) ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a
|
||||
`shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous,
|
||||
- [x] [abi.zig](../system/abi.zig): `shared_memory_create` (34) / `shared_memory_map` (35) syscalls + a
|
||||
`shared_memory_test` service id. Handlers in process.zig: `shared_memory_create(len)` allocates contiguous,
|
||||
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||
handle, maps them into the caller's shm arena → returns vaddr + handle; `shm_map(cap)`
|
||||
handle, maps them into the caller's shared-memory arena → returns virtual_address + handle; `shared_memory_map(cap)`
|
||||
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
||||
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
||||
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
||||
dispatch by kind, so an `ShmObject` rides an `ipc_call` `send_cap` exactly like an
|
||||
dispatch by kind, so a `SharedMemoryObject` rides an `ipc_call` `send_cap` exactly like an
|
||||
endpoint and frees only when its last capability drops. `mapUserSharedInto` (paging)
|
||||
maps WB-cacheable + `device_grant`, so a sharer's teardown never frees the shared
|
||||
frames — the object owns them.
|
||||
- [x] `library/runtime/shm.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||
- [x] `library/runtime/shared-memory.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||
`map(handle) -> ptr`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py shm` — `shm-client` creates a region, writes a
|
||||
pattern, and passes its capability to `shm-server` as an `ipc_call` send_cap; the server
|
||||
`shm_map`s it and reads the **same bytes** back → `shm: shared 4096 bytes ok`. Guardrail:
|
||||
**Gate (met):** `python3 test/qemu_test.py shared-memory` — `shared-memory-client` creates a region, writes a
|
||||
pattern, and passes its capability to `shared-memory-server` as an `ipc_call` send_cap; the server
|
||||
`shared_memory_map`s it and reads the **same bytes** back → `shared-memory: shared 4096 bytes ok`. Guardrail:
|
||||
`ipc`/`ipc-call`/`ipc-cap`, `supervision`, `dma`, `usermem`, `display-service`, and host
|
||||
tests all still pass — the handle-table change broke no existing IPC.
|
||||
|
||||
@@ -88,7 +88,7 @@ tests all still pass — the handle-table change broke no existing IPC.
|
||||
VERSION_1, and stand up the control virtqueue in coherent DMA. `virtio-gpu-protocol.zig`
|
||||
+ `virtio-pci.zig` for the control/transport structs (host-tested sizes).
|
||||
- [x] Create a 2D scanout resource backed by a coherent DMA region (V4 swaps this for the
|
||||
shm-shared surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||
shared-memory surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||
+ `resource_flush` of a test pattern, and wait on the used ring.
|
||||
- [x] Register a `scanout` service (`ServiceId.scanout` = 11).
|
||||
|
||||
@@ -102,7 +102,7 @@ end without a screenshot (the used-ring ack is the device confirming it consumed
|
||||
|
||||
## V4 — The native backend + hot-attach ✅
|
||||
|
||||
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared `shm` scanout surface
|
||||
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared-memory scanout surface
|
||||
(the compositor composes straight into the device's resource backing; x86 DMA is
|
||||
coherent, so the cacheable shared pages need no flush), `present(damage)` = a `present`
|
||||
request over the driver's `.scanout` endpoint (→ transfer-to-host + resource flush).
|
||||
@@ -112,7 +112,7 @@ end without a screenshot (the used-ring ack is the device confirming it consumed
|
||||
driver registered it), switches backend, and re-composites the current frame full-screen.
|
||||
The present is deferred to a one-shot timer so it runs *after* the reply unblocks the
|
||||
driver and it serves `.scanout` — presenting inline would deadlock.
|
||||
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shm_physical` (a
|
||||
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shared_memory_physical` (a
|
||||
new syscall) gives the driver the guest-physical of the shared surface for `attach_backing`.
|
||||
|
||||
**Gate (met):** the `display-native` case (QEMU `-device virtio-gpu-pci`, `mem` bumped since it
|
||||
@@ -123,7 +123,7 @@ confirm the composited frame landed (`display: native present verified`), while
|
||||
ok` still fires — checked order-independently. Without `-device virtio-gpu-pci` nothing is
|
||||
announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pass unchanged.
|
||||
|
||||
## V5 — Mode-setting, EDID, and vsync ✅
|
||||
## V5 — Mode-setting, EDID, and fenced presents ✅
|
||||
|
||||
- [x] The driver negotiates `VIRTIO_GPU_F_EDID` (when offered) and reads the monitor's EDID,
|
||||
logging its preferred mode; it offers a small mode list over `.scanout` `get_modes`. The
|
||||
@@ -131,14 +131,16 @@ announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pas
|
||||
scanout rectangle (no resource/surface churn) — a runtime resolution change. `runtime.display`
|
||||
gains `modes()` / `setMode()` (display-protocol `get_modes`/`set_mode`, forwarded to the backend).
|
||||
- [x] Every `resource_flush` is issued fenced (`VIRTIO_GPU_FLAG_FENCE`); the device signals the
|
||||
fence when the frame is on screen, which the used-ring ack the synchronous present waits on
|
||||
already gates — a tear-free present.
|
||||
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasVsync` = true.
|
||||
fence when it has consumed the frame, which the used-ring ack the synchronous present waits
|
||||
on already gates — a tear-free present. (Completion feedback, **not vblank**: base
|
||||
virtio-gpu 2D has no display-refresh event, so nothing paces presents to the monitor —
|
||||
see the "Fenced is not vsync" note in [display-v2.md](display-v2.md).)
|
||||
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasFencedPresent` = true.
|
||||
|
||||
**Gate (met):** the `display-modeset` case (reusing the display-native boot) upgrades to
|
||||
virtio-gpu, queries the driver's modes, `setMode`s to a different resolution, and confirms the
|
||||
change by reading the backend's geometry back (`display: mode set to {w}x{h}, verified`); the
|
||||
fenced present path is exercised and confirmed (`display: vsync present ok`) — both from serial,
|
||||
fenced present path is exercised and confirmed (`display: fenced present ok`) — both from serial,
|
||||
passing 3/3. The driver also logs the EDID preferred mode (`virtio-gpu: EDID preferred mode …`).
|
||||
|
||||
## V6 — Resilience (restart + re-attach) + tests + docs ✅
|
||||
@@ -157,14 +159,14 @@ kills the virtio-gpu driver once after it hellos; the restart policy respawns it
|
||||
re-announces, and the compositor logs `display: scanout re-attached` after the initial
|
||||
`display: scanout upgraded to virtio-gpu`, with no CPU exception / panic (the compositor
|
||||
survives) — passing 3/3. All v1 + v2 cases (host tests, `ipc`/`ipc-call`/`ipc-cap`,
|
||||
`supervision`, `shm`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||
`supervision`, `shared-memory`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||
`display-modeset`) pass; default `zig build` is clean.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Client-rendered surfaces** — now unblocked by the `shm` capability (V2): an app renders
|
||||
- **Client-rendered surfaces** — now unblocked by the shared-memory capability (V2): an app renders
|
||||
its own bitmap and hands the compositor a reference. A natural follow-on.
|
||||
- **Bochs DISPI backend** — a simpler second native backend (mode-set only, dumb scanout);
|
||||
slots behind the same interface if wanted.
|
||||
|
||||
+23
-15
@@ -1,8 +1,8 @@
|
||||
# The display service v2: a pluggable scanout backend
|
||||
|
||||
**Status: complete (V1–V6).** The compositor boots on the GOP framebuffer and, when a
|
||||
virtio-gpu driver announces itself, hot-attaches a native backend over the shared `shm`
|
||||
scanout surface — with runtime mode-setting, EDID, and fenced (vsync) presents, and it
|
||||
virtio-gpu driver announces itself, hot-attaches a native backend over the shared-memory
|
||||
scanout surface — with runtime mode-setting, EDID, and fenced presents, and it
|
||||
re-attaches across driver restarts. All serial-gated (see [display-v2-plan.md](display-v2-plan.md)).
|
||||
|
||||
v1 ([display.md](display.md)) is a compositor that owns the **GOP framebuffer** — it
|
||||
@@ -25,10 +25,10 @@ The compositor itself (layers, back buffer, damage) does not change. Only the la
|
||||
scanout backend (selected at runtime — GOP by default, native when it appears)
|
||||
│
|
||||
├─ GopBackend the v1 path: WC copy back→front to the firmware LFB.
|
||||
│ Always available. No mode-set, no vsync. THE FLOOR.
|
||||
│ Always available. No mode-set, no present fence. THE FLOOR.
|
||||
│
|
||||
└─ VirtioGpuBackend talks to a virtio-gpu driver process over a `scanout`
|
||||
service: present via a shared resource + flush (real vsync),
|
||||
service: present via a shared resource + fenced flush,
|
||||
EDID mode list, runtime mode-set.
|
||||
```
|
||||
|
||||
@@ -37,8 +37,8 @@ A **backend** is a small interface the compositor calls:
|
||||
- `surface()` → the pixels to compose into and their geometry `{ptr, pitch, format, w, h}`
|
||||
(the LFB for GOP; a shared scanout resource for virtio-gpu),
|
||||
- `present(damage: Rect)` → make the damaged region visible (a no-op-ish WC copy for GOP;
|
||||
a virtio flush, optionally vsync-fenced, for the native path),
|
||||
- capability queries — `canModeSet`, `hasVsync` — and, when supported, `modes()` /
|
||||
a fenced virtio flush for the native path),
|
||||
- capability queries — `canModeSet`, `hasFencedPresent` — and, when supported, `modes()` /
|
||||
`setMode(m)`.
|
||||
|
||||
The compositor composes into `surface()` and calls `present(damage)` exactly as it does
|
||||
@@ -77,9 +77,9 @@ deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural gen
|
||||
of M13 capability-passing from *endpoints* to *memory objects* —
|
||||
|
||||
```
|
||||
shm_create(len) -> {handle, vaddr} // a shareable, page-aligned RAM region
|
||||
shared_memory_create(len) -> {handle, virtual_address} // a shareable, page-aligned RAM region
|
||||
… pass `handle` as the send_cap on an ipc_call …
|
||||
shm_map(cap) -> vaddr // the receiver maps the same physical pages
|
||||
shared_memory_map(cap) -> virtual_address // the receiver maps the same physical pages
|
||||
```
|
||||
|
||||
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
||||
@@ -92,35 +92,43 @@ A new ring-3 driver process (the topology v1 anticipated — "split the driver f
|
||||
compositor when a second backend arrives"). It claims the virtio-gpu PCI function, and:
|
||||
|
||||
- sets up the **virtqueues** (control + cursor) and the device's config space,
|
||||
- creates a **2D scanout resource** backed by an `shm` region, `attach_backing`s it,
|
||||
- creates a **2D scanout resource** backed by a shared-memory region, `attach_backing`s it,
|
||||
`set_scanout`s it to a CRTC, and `resource_flush`es damaged rectangles,
|
||||
- reads **EDID** (the `GET_EDID` control command) for the mode list, and `set_scanout`
|
||||
at a chosen mode for **runtime mode-setting**,
|
||||
- registers a `scanout` service and announces to the display service.
|
||||
|
||||
Its `resource_flush` is the real **present** — and gives a genuine **vsync/tear-free**
|
||||
path a dumb GOP framebuffer can't.
|
||||
Its `resource_flush` is the real **present** — and gives a **fenced, tear-free** path a
|
||||
dumb GOP framebuffer can't.
|
||||
|
||||
**Fenced is not vsync.** The fence completes when the device has *consumed* the frame:
|
||||
real completion feedback, and tear-freedom by snapshot semantics (the host displays
|
||||
discrete transferred frames, never a half-written surface). It is **not** a vblank —
|
||||
base virtio-gpu 2D has no display-refresh event at all (Linux's driver for this device
|
||||
fakes one with a software timer), so nothing paces presents to the monitor's refresh.
|
||||
Refresh-paced presents need either a native driver's vblank interrupt (delivered over
|
||||
the existing IRQ-as-IPC path) or the compositor's own frame clock.
|
||||
|
||||
## What v2 unlocks — and its honest scope
|
||||
|
||||
Behind the abstraction, a native backend gives runtime **mode-setting** (resolution /
|
||||
refresh / bpp), **EDID** enumeration, and **vsync**. But only on devices we have a driver
|
||||
refresh / bpp), **EDID** enumeration, and **fenced presents**. But only on devices we have a driver
|
||||
for — realistically **VMs** (virtio-gpu, and later maybe Bochs DISPI). Real discrete GPUs
|
||||
need per-vendor KMS-class drivers that aren't getting written, so they **stay on GOP** —
|
||||
which is genuinely fine (v1 on the NVIDIA box is smooth). So v2's real value is twofold:
|
||||
the **pluggable architecture** (a driver slots in when one exists) and a **rich, vsync'd
|
||||
the **pluggable architecture** (a driver slots in when one exists) and a **rich, fenced
|
||||
path in VMs**, where danos development happens. The framebuffer floor never goes away.
|
||||
|
||||
## Locked decisions
|
||||
|
||||
- **First native backend: virtio-gpu** — the VM standard; gives mode-set + a real
|
||||
present/flush (and vsync), and exercises the whole pluggable design. Tested with QEMU
|
||||
present/flush (fenced), and exercises the whole pluggable design. Tested with QEMU
|
||||
`-device virtio-gpu`.
|
||||
- **Dynamic hot-attach** — boot on GOP, upgrade to native on the driver's announce,
|
||||
re-attach across driver restarts; GOP is the floor for "no driver ever," not a live
|
||||
fall-back after a reprogram.
|
||||
- **Detection = push** (the driver announces to `.display`), not compositor polling.
|
||||
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||
- **v2 builds the shared-memory capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
|
||||
## See also
|
||||
|
||||
+58
-9
@@ -95,7 +95,7 @@ rest of the system hasn't had to face:
|
||||
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||
drawing clients (v1) surface clients (deferred)
|
||||
runtime.display commands: runtime.display surfaces:
|
||||
create_layer / configure_layer shm_create → pass as a capability →
|
||||
create_layer / configure_layer shared_memory_create → pass as a capability →
|
||||
fill_rect / blit_tile / damage the compositor maps & composites the
|
||||
present client-rendered bitmap directly
|
||||
```
|
||||
@@ -196,13 +196,21 @@ shell, a terminal, a cursor, and a wallpaper:
|
||||
| `fill_rect` | fill a rectangle of a layer with a colour |
|
||||
| `blit_tile` | copy a small client-supplied pixel tile into a layer (inline) |
|
||||
| `damage` | mark a region of a layer dirty |
|
||||
| `present` | composite dirty layers and flush to the screen |
|
||||
| `present` | request a repaint: composited at the next frame-clock tick |
|
||||
|
||||
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||
(the [PSF font](../system/kernel/font.psf) path the console already uses can move into a
|
||||
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||
policy in the client.
|
||||
|
||||
`present` is a *request*, not an immediate flush: the compositor runs a ~60 Hz **frame
|
||||
clock** (a one-shot kernel timer re-armed on demand), and each tick composites all the
|
||||
damage accumulated since the last one. Any number of client presents and cursor moves
|
||||
inside one interval coalesce into a single repaint — the software stand-in for vblank
|
||||
pacing on backends that have none (all of them today; see
|
||||
[display-v2.md](display-v2.md), "Fenced is not vsync"). Bring-up paths that must put
|
||||
pixels on screen synchronously (initialisation, the self-checks) bypass the clock.
|
||||
|
||||
## `runtime.display`
|
||||
|
||||
Clients speak the protocol through a new [`library/runtime/display.zig`](../library/runtime/runtime.zig),
|
||||
@@ -211,6 +219,38 @@ with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitT
|
||||
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||
runtime, as with every other danos service.
|
||||
|
||||
## The cursor: a mouse-listener thread feeding the compositor
|
||||
|
||||
The compositor is the single owner of the framebuffer — only the main `service.run` loop
|
||||
touches the backend and the layer stack. Tracking the mouse without breaking that
|
||||
ownership is the display's first use of [threads](threading.md): the service is built
|
||||
multi-threaded (`addThreadedUserBinary`) and, at startup, spawns a **mouse-listener
|
||||
thread** beside the compositor loop.
|
||||
|
||||
- **Listener thread.** Blocks on the input service's mouse stream
|
||||
(`input.subscribeMouse()`), accumulates the relative `dx`/`dy` motion into an absolute
|
||||
cursor position clamped to the screen, and hands it to the compositor. It never touches
|
||||
the compositor — so no lock guards the framebuffer. A parked `next()` leaves its core
|
||||
free to halt ([halting.md](halting.md)).
|
||||
- **The channel.** A single-slot *latest-value* cell (`CursorChannel`) guarded by a
|
||||
`runtime.Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
every delta, so a new position overwrites the old. The listener also **pokes** the
|
||||
compositor awake — the main loop is parked in `replyWait`, so the listener posts a
|
||||
zero-payload `ipc.send` to the compositor's endpoint, which arrives as a
|
||||
message-notification ([ipc.md](ipc.md)). The poke is *coalesced*: at most one is queued
|
||||
while the main loop has not drained the last, so a fast mouse cannot flood the endpoint.
|
||||
- **Render.** On the poke, the main loop takes the latest position and moves the cursor —
|
||||
which is just a top-z compositor layer — with the existing `configure` + `present` path
|
||||
(it damages the old and new footprints, so only those two rectangles repaint).
|
||||
|
||||
Two threading facts shape this (both in [threading.md](threading.md)). IPC **handles do
|
||||
not cross threads**, so the listener can't reuse the main loop's endpoint handle — it
|
||||
`ipc.lookup(.display)`s its *own* handle to the same endpoint to poke through. And a
|
||||
multi-threaded service doing concurrent IPC is why the kernel's endpoint-create / register
|
||||
/ lookup syscalls now serialize under the big kernel lock. Shared fate applies: a fault in
|
||||
the listener takes the whole display down, and the supervisor restarts the process
|
||||
([resilience.md](resilience.md)).
|
||||
|
||||
## What v1 does not do (and why that's fine)
|
||||
|
||||
Two capabilities are deliberately out of the first cut. Neither reshapes anything above;
|
||||
@@ -220,8 +260,8 @@ both are clean additions behind the interfaces v1 establishes.
|
||||
to render into its *own* buffer and hand the compositor a *reference*, not a stream of
|
||||
commands. That needs the missing cross-process shared-memory primitive — best built as
|
||||
the natural generalization of the existing M13 [capability passing](driver-model.md)
|
||||
from *endpoints* to *memory objects* (`shm_create(len) → {cap, vaddr}`, pass `cap` on
|
||||
an `ipc_call`, receiver `shm_map(cap) → vaddr`). v1 avoids it because server-owned
|
||||
from *endpoints* to *memory objects* (`shared_memory_create(len) → {cap, virtual_address}`, pass `cap` on
|
||||
an `ipc_call`, receiver `shared_memory_map(cap) → virtual_address`). v1 avoids it because server-owned
|
||||
surfaces already prove the whole pipeline.
|
||||
|
||||
- **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing
|
||||
@@ -232,7 +272,7 @@ both are clean additions behind the interfaces v1 establishes.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Three QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||
Four QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||
test/qemu_test.py <case>`), each layering on the last:
|
||||
|
||||
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||
@@ -245,11 +285,20 @@ test/qemu_test.py <case>`), each layering on the last:
|
||||
layer — logging `display: compositor self-check ok`.
|
||||
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||
[`display-demo`](../system/services/display-demo/) client (the
|
||||
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper, a
|
||||
sliding rectangle, a cursor — through the layer client API and heartbeats
|
||||
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper and
|
||||
a sliding rectangle — through the layer client API and heartbeats
|
||||
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||
the [input test](input.md) proves an event travels source → service → subscriber. The
|
||||
visible motion itself is a screenshot away via `zig build run-x86-64`.
|
||||
the [input test](input.md) proves an event travels source → service → subscriber. It draws
|
||||
no cursor and reads no input — the cursor is the service's own (below), and the demo
|
||||
animates on its own frame timer, independent of the mouse (the test spawns `input`
|
||||
alongside it to keep that independence honest). The visible motion itself is a screenshot
|
||||
away via `zig build run-x86-64`.
|
||||
- **`display-cursor`** — the mouse-listener thread end to end: with the `input` service up,
|
||||
`input-source mouse` publishes pure motion, and the display's listener thread accumulates
|
||||
it into a cursor position handed to the render loop over the `CursorChannel`. Once the
|
||||
cursor has tracked a run of that motion, the service logs
|
||||
`display: cursor tracking mouse ok`. Runs `smp: 4` — the compositor and listener threads
|
||||
execute on different cores, which is what surfaced the IPC-under-lock requirement above.
|
||||
|
||||
The compositor's pixel math (rectangle clipping, fill, composite, tile blit) and colour
|
||||
packing are additionally covered by pure host unit tests under `zig build test`.
|
||||
|
||||
@@ -240,8 +240,8 @@ once per page, maps writeback-cached, and never reveals a physical address.
|
||||
**The fix.**
|
||||
|
||||
```
|
||||
dma_alloc(len, flags) -> vaddr (rax), paddr (rdx)
|
||||
dma_free(vaddr, len) -> 0
|
||||
dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx)
|
||||
dma_free(virtual_address, len) -> 0
|
||||
|
||||
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||
|
||||
+1
-1
@@ -74,7 +74,7 @@ The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||
|---|------|---------|
|
||||
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||
| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window |
|
||||
| 13 | `mmio_map(id, res_idx) -> virtual_address` | Map a claimed device's register window |
|
||||
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||
|
||||
+78
-86
@@ -1,104 +1,96 @@
|
||||
# Logging: the diagnostic log vs. the display
|
||||
# Logging
|
||||
|
||||
danos separates two things that are easy to conflate: the **diagnostic log** — the
|
||||
machine-readable stream of *what the kernel is doing* — and the **display**, the
|
||||
framebuffer surface the OS draws on. They are different concerns with different
|
||||
lifetimes, so they're different code paths.
|
||||
Output is a *diagnostic convenience, never a correctness dependency*: the kernel
|
||||
and every service must run correctly with zero output channels. On top of that
|
||||
rule, danos has **per-process logging** — every process's output is attributed
|
||||
by the kernel and lands in its own file on the flash volume, which is what makes
|
||||
a headless real machine (no serial port) debuggable. The display (the
|
||||
framebuffer surface) is a separate concern and deliberately not a log sink;
|
||||
`main.zig` mirrors a few user-facing status lines and panics to it explicitly.
|
||||
|
||||
The guiding rule: **output is a diagnostic convenience, never a correctness
|
||||
dependency.** The kernel must boot and run correctly with *zero* output channels —
|
||||
no serial, no screen. Logging that can take the kernel down isn't robust; it's a
|
||||
liability. This is the same [resilience](resilience.md) posture the rest of the
|
||||
kernel follows.
|
||||
|
||||
## The log is multi-sink
|
||||
|
||||
`system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||
registered **sinks**, each best-effort and self-guarding:
|
||||
|
||||
```zig
|
||||
log.addSink(arch.serialWrite); // the serial UART
|
||||
if (arch.debugconPresent()) log.addSink(arch.debugconWrite); // 0xE9 debug console
|
||||
// later: log.addSink(fs.logWrite); // a file on a ramdisk / USB / SSD
|
||||
log.write("…"); log.print("x={d}\n", .{x});
|
||||
```
|
||||
|
||||
Properties that matter:
|
||||
|
||||
- **No allocation.** The sink table is a fixed array, so the log works before the
|
||||
heap is up and inside a panic.
|
||||
- **Best-effort.** A sink whose device is absent is a no-op (e.g. writing to a
|
||||
missing UART just goes nowhere — the TX-wait is bounded so it can't hang). A
|
||||
message reaches whatever channels exist; if none do, the kernel runs on, silent.
|
||||
- **Order-independent.** Every registered sink gets every message. Adding the file
|
||||
logger later is one `addSink` call and **zero** changes to call sites.
|
||||
|
||||
## The framebuffer is *not* a log sink
|
||||
|
||||
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
||||
Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
||||
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
||||
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
||||
|
||||
Instead the two paths are explicit:
|
||||
## The pipeline
|
||||
|
||||
```
|
||||
verbose diagnostics ──► log ──► serial, debugcon, (file later)
|
||||
user status / panics ──► status() ──► log (above) + framebuffer (if present)
|
||||
process std.log ──▶ debug_write(level) ──▶ tagged kernel ring ──▶ logger service ──▶ /var/log/<boot-stamp>/<binary-path>.log
|
||||
kernel log.print ─┘ │
|
||||
└▶ serial / 0xE9 sinks (QEMU, -Dserial)
|
||||
```
|
||||
|
||||
A handful of user-facing lines (`kernel initialised`, a panic) go through
|
||||
`main.zig`'s `status()` / `statusPrint()`, which write to the log **and** paint the
|
||||
framebuffer when one is present. Everything else uses `log.*` and never touches the
|
||||
screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
||||
1. **Emit.** A program calls `std.log.info("mounted {s}", .{path})` — the
|
||||
runtime's `logFn` (installed for every binary by the root shim,
|
||||
`library/runtime/log.zig`) formats one line and issues one `debug_write`
|
||||
carrying the level. The payload does NOT contain the process's name.
|
||||
`runtime.system.write` remains as the raw/bring-up path (panics, test
|
||||
fixtures); raw bytes ride the same ring, attributed all the same.
|
||||
|
||||
## Optional framebuffer (headless machines)
|
||||
2. **Stamp.** The kernel wraps every payload LINE in a record stamped with the
|
||||
sender's pid, task name (its binary path, e.g. `/system/services/fat`),
|
||||
level, a per-boot sequence number, and a monotonic timestamp
|
||||
(`system/kernel/log.zig` + `log-ring.zig`). Attribution is structural — a
|
||||
payload cannot forge another sender's tag, and an embedded newline just ends
|
||||
the record, so the forged "prefix" lands inside the forger's own next line.
|
||||
|
||||
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||
serial-less machine boots and runs correctly — it just goes quiet.
|
||||
3. **Retain.** The 512 KiB ring overwrites oldest-first; sequence gaps make any
|
||||
loss countable. `klog_read` (#32) copies stream bytes from a free-running
|
||||
offset; `klog_status` (#45) returns the cursors plus the wall-clock time of
|
||||
boot. The framing (`abi.KlogRecordHeader`) is 32 bytes + name + payload,
|
||||
8-byte aligned.
|
||||
|
||||
## Last-resort channels (no text output at all)
|
||||
4. **Render.** Registered sinks (serial under `-Dserial`, the 0xE9 debug
|
||||
console) get a live transcript: kernel/raw output verbatim, leveled records
|
||||
as `<binary path>: message` — one composed write per line, under the log's
|
||||
own spinlock (never the big kernel lock; panic paths try-acquire with a
|
||||
bound and fall back to sinks-only). Sinks are best-effort and self-guarding;
|
||||
a serial-less machine just goes quiet.
|
||||
|
||||
Two signals bypass the sink list, because they must survive even a total
|
||||
output-channel failure:
|
||||
5. **Persist.** The **logger service** (`system/services/logger`) drains the
|
||||
ring every 250 ms and demultiplexes records into one file per source under
|
||||
`/var/log/<boot-stamp>/`, e.g.
|
||||
|
||||
- **`log.checkpoint(code)`** — a one-byte **POST code** to I/O port `0x80` (a POST
|
||||
card or BMC shows it). `main.zig` emits one at each boot milestone (`cp_paging`,
|
||||
`cp_heap`, …) and on a fault/panic, so "where did it hang?" is answerable with no
|
||||
text output whatsoever. Writing `0x80` is universally safe.
|
||||
- **`log.recordPanic(msg)`** — stamps the panic message into a fixed record
|
||||
(`log.panic_record`, with a `magic` written last). A post-mortem — an attached
|
||||
debugger, a RAM dump, or a future file/pstore reader — recovers *what killed it*
|
||||
even though nothing was on screen.
|
||||
```
|
||||
/var/log/2026-07-21T150434Z/kernel.log
|
||||
/var/log/2026-07-21T150434Z/system/services/fat.log
|
||||
/var/log/2026-07-21T150434Z/system/drivers/usb-storage.log
|
||||
```
|
||||
|
||||
The panic and CPU-exception handlers fan out to every sink, emit a POST code, and
|
||||
drop the breadcrumb — they never assume a console.
|
||||
The boot stamp is the RTC anchor from `klog_status` (FAT-safe: no colons; a
|
||||
dead RTC yields the 1970 directory rather than no logs). Each line carries
|
||||
the record's monotonic timestamp and level. Storage is best-effort and late:
|
||||
the ring buffers a whole boot many times over, and the first successful
|
||||
`makePath` of the per-boot directory (also the readiness probe) triggers a
|
||||
full backlog write. Files close — which is the fat server's SCSI cache
|
||||
flush — after a ~2 s quiet period, bounding data-at-risk without per-record
|
||||
flush thrash. At shutdown init stops the logger FIRST (it is last in the
|
||||
boot order), so its final drain runs over a live storage chain.
|
||||
|
||||
## The 0xE9 debug console
|
||||
## Why a ring in the kernel, not a logging server
|
||||
|
||||
Port `0xE9` is the Bochs/QEMU debug console. It's detected safely: the port returns
|
||||
`0xE9` when read if present, and `0xFF` on real hardware, so `debugconPresent()`
|
||||
only enables the sink when it's really there. Under QEMU it's captured with
|
||||
`-debugcon file:…`, giving CI a log channel independent of `-serial`.
|
||||
The storage stack must be able to log. If the fat server wrote its own log file
|
||||
through the VFS it would rendezvous-deadlock on itself; if processes sent
|
||||
records to a logging server over IPC, early boot would need a buffer that is —
|
||||
a ring, one hop later. The kernel ring is that buffer, placed where every
|
||||
process (and the kernel itself) can reach it with one syscall, before any
|
||||
service exists. The logger service is a *reader*, not a hop.
|
||||
|
||||
## The robustness spectrum
|
||||
Two disciplines keep it honest:
|
||||
|
||||
The result handles every combination — framebuffer only, serial only, both, or
|
||||
**neither**. With no channels at all the kernel still boots and runs; port-`0x80`
|
||||
checkpoints track progress and the panic breadcrumb captures failures. *Runs blind
|
||||
but correct* is the goal, not *always has output*.
|
||||
- the logger announces itself **once** — a periodic status line would feed the
|
||||
very stream it drains;
|
||||
- lost records surface as an explicit `-- N records lost --` line, computed
|
||||
from sequence gaps, never silently.
|
||||
|
||||
## Related
|
||||
## Last-resort channels
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the display surface itself (pitch, format), the
|
||||
thing that becomes a graphics device driver.
|
||||
- [efi.md](efi.md) — where the loader captures (or, headless, doesn't capture) the
|
||||
framebuffer before `ExitBootServices`.
|
||||
- [device-interrupts.md](device-interrupts.md) — the serial UART bring-up the log's
|
||||
primary sink rides on.
|
||||
- [resilience.md](resilience.md) — why "never let a missing peripheral take the
|
||||
kernel down" is a core design stance.
|
||||
Unchanged, and independent of the sink list so they survive a total output
|
||||
failure: `checkpoint` (a one-byte POST code on port 0x80) and `recordPanic`
|
||||
(a fixed breadcrumb record, `log.panic_record`, findable in a RAM dump; magic
|
||||
written last so a reader only trusts a complete record).
|
||||
|
||||
## Accepted gaps
|
||||
|
||||
- A write-spamming process can evict other processes' unread records from the
|
||||
ring (a per-process quota is future work); the loss is at least visible via
|
||||
sequence gaps in every affected file.
|
||||
- `/var/log` files have no privacy until the VFS grows permissions.
|
||||
- Records emitted after the logger's final shutdown drain reach serial and the
|
||||
ring but not the files.
|
||||
|
||||
+10
-7
@@ -1,15 +1,18 @@
|
||||
# System Calls
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
|
||||
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||
> **Status:** danos has real user processes. User programs enter the kernel
|
||||
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
|
||||
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||
> dispatcher); the `int 0x80` gate is kept alongside as a minimal test path.
|
||||
> The live table is `system/abi.zig` (private, renumberable — see
|
||||
> [vdso.md](vdso.md) for the public boundary): process lifecycle + threads,
|
||||
> memory (mmap/dma/shared-memory), synchronous + async IPC with capability passing,
|
||||
> device access, time, the tagged-log diagnostics (`debug_write` with a level,
|
||||
> `klog_read`/`klog_status`), and filesystem NAMING (`fs_resolve`/`fs_node`/
|
||||
> `fs_mount`/`fs_unmount` — the kernel VFS root routes paths and serves the
|
||||
> read-only /system initrd mount; file DATA stays with userspace filesystem
|
||||
> servers over the vfs-protocol, docs/vfs-protocol.md).
|
||||
|
||||
## The Mechanism of a Syscall
|
||||
|
||||
|
||||
+214
-35
@@ -14,7 +14,7 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
- **Blocking is futex-backed, never spin-backed** — waiters park in the kernel so an
|
||||
idle core still halts ([halting.md](halting.md)).
|
||||
- **New syscalls are private**: extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shm_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
`shared_memory_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
`futex_wait = 40`, `futex_wake = 41`) + a `library/runtime` wrapper; user code never names a number.
|
||||
- **Restart granularity stays the process** — a faulting thread kills its process; the
|
||||
supervisor restarts the process, which respawns its threads.
|
||||
@@ -52,8 +52,11 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
|
||||
1. **Resume** at the first milestone that still has an unchecked `- [ ]`. (All earlier
|
||||
milestones are done — do not revisit them.)
|
||||
2. **Work on a branch.** On the first iteration, branch off `main` (e.g. `threading`);
|
||||
never commit threading work to `main`. All work stays local — **do not push**.
|
||||
2. **Work on a branch.** On the first iteration, branch off the current `main` into a new
|
||||
branch (e.g. `threading-phase2` — Phase 1's `threading` is already merged); never
|
||||
commit to `main` directly. Push that **branch** to `origin` after each milestone (step
|
||||
5) so progress is backed up remotely; **do not push `main`** — merging Phase 2 into
|
||||
`main` stays a human step.
|
||||
3. **Implement** every unchecked item in that milestone, including adding its
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../test/qemu_test.py) (with
|
||||
`smp: true` / a `mem` bump where noted) so the gate is runnable.
|
||||
@@ -64,8 +67,9 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
the whole guardrail set passes, `zig build` is clean, and host tests are green.
|
||||
→ tick this milestone's boxes **and** its `**Gate:**`-referenced case, `git commit`
|
||||
(`threads(M<n>): <summary>`, no `Co-Authored-By` trailer per
|
||||
[coding-standards.md](coding-standards.md)), and continue to the next milestone in
|
||||
the same iteration if budget remains; otherwise let the loop re-fire.
|
||||
[coding-standards.md](coding-standards.md)), then **`git push` the working branch to
|
||||
`origin`** (use `-u` on the first push to set upstream). Continue to the next
|
||||
milestone in the same iteration if budget remains; otherwise let the loop re-fire.
|
||||
- **Red** = anything above fails. Diagnose from the captured serial log
|
||||
(`zig-out/qemu-test/<case>-failed-serial.log`) and fix in place, then re-run — up to
|
||||
**3 fix attempts** for that gate. A concurrency case that fails then passes on a
|
||||
@@ -79,14 +83,18 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
|
||||
**The only stop conditions:**
|
||||
|
||||
- **Done** — every milestone box is checked (M1–M6), `zig build` clean, whole
|
||||
`thread-*` suite + guardrail green. Update threading.md's status line to "built" (that
|
||||
is M6's own task) and stop.
|
||||
- **Done** — every milestone box **in this plan** is checked (M1 through M11), `zig build`
|
||||
clean, the whole `thread-*` suite + guardrail green. Phase 1 (M1–M6) is *already*
|
||||
checked, so do **not** read that as Done: the loop's real work is the first plan section
|
||||
that still has unchecked boxes — Phase 2 (M7–M11). Only stop when M7–M11 are all checked
|
||||
too. Update threading.md's status line, push the final branch state to `origin`, and
|
||||
stop. The branch is on `origin` for review; **merging Phase 2 into `main` is the user's
|
||||
step**, not the loop's.
|
||||
- **Blocked** — a gate is still red after 3 fix attempts, or a step needs something
|
||||
outside the repo (a toolchain change, new hardware, a decision no locked decision
|
||||
covers). Append `> **BLOCKED (M<n>):** <what failed, what was tried, the serial
|
||||
marker missing>` under that milestone, commit the WIP on the branch, and stop. Do not
|
||||
thrash further and do not silently skip the milestone.
|
||||
marker missing>` under that milestone, commit **and push** the WIP on the branch, and
|
||||
stop. Do not thrash further and do not silently skip the milestone.
|
||||
|
||||
Nothing else warrants stopping — not "should I proceed?", not "is this right?". The
|
||||
checkboxes + git history are the resumable record; the next iteration picks up from the
|
||||
@@ -97,28 +105,28 @@ first unchecked box.
|
||||
## M1 — Address-space refcount (kernel foundation, no API, no behaviour change) ✅
|
||||
|
||||
The one invariant change threads require, landed and proven **before** anything shares
|
||||
an address space. Today aspace is 1:1 with a task and teardown destroys it on any user
|
||||
an address space. Today address space is 1:1 with a task and teardown destroys it on any user
|
||||
task's exit; make destruction happen on the **last** exit.
|
||||
|
||||
- [x] A refcount keyed by the address-space root, held in `scheduler.zig`
|
||||
(`aspace_refs`): `retainAspace` takes a reference in `spawnUserLocked` (on the
|
||||
(`address_space_refs`): `retainAddressSpace` takes a reference in `spawnUserLocked` (on the
|
||||
success path, after the slot + stack are secured), all under the big kernel lock.
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig):
|
||||
`exitUserLocked` and `destroyTaskLocked`) call `releaseAspace`, which decrements
|
||||
and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test
|
||||
spaces) is destroyed directly, preserving prior behaviour.
|
||||
- [x] `-Dtest-case=aspace-refcount`: spawn and reap several ring-3 processes in sequence
|
||||
and assert (via test-observable `liveAspaceCount`/`aspaceDestroyCount`) that the
|
||||
- [x] `-Dtest-case=address-space-refcount`: spawn and reap several ring-3 processes in sequence
|
||||
and assert (via test-observable `liveAddressSpaceCount`/`addressSpaceDestroyCount`) that the
|
||||
live-space count returns to **baseline** and destructions advance by exactly that
|
||||
many — each space destroyed exactly once, no leak, no double-free. (Refcount
|
||||
observables, not raw frame counts, since kernel stacks are still leaked on exit.)
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py aspace-refcount` passes
|
||||
(`aspace-refcount: spaces released to baseline ok` → `DANOS-TEST-RESULT: PASS`), and the
|
||||
**Gate (met):** `python3 test/qemu_test.py address-space-refcount` passes
|
||||
(`address-space-refcount: spaces released to baseline ok` → `DANOS-TEST-RESULT: PASS`), and the
|
||||
full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `smp`,
|
||||
`affinity`, `process`, `process-kill`, `supervision`, `fault-recovery`,
|
||||
`vfs-client-death`, `ipc`, `ipc-cap`, `display-service`); default `zig build` clean,
|
||||
`zig build test` green. The reframing is invisible until an aspace is actually shared.
|
||||
`zig build test` green. The reframing is invisible until an address space is actually shared.
|
||||
|
||||
## M2 — `thread_spawn` + `thread_exit`: a thread runs in the shared address space ✅
|
||||
|
||||
@@ -127,7 +135,7 @@ space and exits cleanly.
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
process.zig; `thread_spawn` calls `scheduler.spawnThread` (shares the caller's
|
||||
aspace, `retainAspace`); `thread_exit` ends the task like a process `exit(0)`
|
||||
address space, `retainAddressSpace`); `thread_exit` ends the task like a process `exit(0)`
|
||||
(`terminateCurrent` → `releaseAspace`). The closure pointer is delivered in the new
|
||||
thread's **rdi** via a new `jump_to_user_arg` asm path (`t.user_arg`, 0 for a
|
||||
process) — no naked runtime asm.
|
||||
@@ -144,14 +152,14 @@ space and exits cleanly.
|
||||
address space.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-spawn` passes
|
||||
(`thread-test: child ran in shared aspace ok` → `DANOS-TEST-RESULT: PASS`); guardrail set
|
||||
(`thread-test: child ran in shared address space ok` → `DANOS-TEST-RESULT: PASS`); guardrail set
|
||||
16/16 green (incl. `args`/`init`/`process`, which exercise the new `jump_to_user_arg`
|
||||
process path with arg 0) plus `aspace-refcount`; `zig build` clean, `zig build test`
|
||||
process path with arg 0) plus `address-space-refcount`; `zig build` clean, `zig build test`
|
||||
green.
|
||||
|
||||
> **Note (deferred to M3+):** the mmap arena is per-*task* (`heap_next`), so two threads
|
||||
> in one aspace that both `mmap` would collide. Fine for M2 (only the parent maps, for the
|
||||
> child's stack); make the arena per-aspace and the runtime heap thread-safe alongside the
|
||||
> in one address space that both `mmap` would collide. Fine for M2 (only the parent maps, for the
|
||||
> child's stack); make the arena per-address-space and the runtime heap thread-safe alongside the
|
||||
> `Mutex` work (M5).
|
||||
|
||||
## M3 — `join` + `detach` + real parallelism ✅
|
||||
@@ -176,7 +184,7 @@ green.
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-join` passes (`thread-test: join ok` →
|
||||
`DANOS-TEST-RESULT: PASS`), robust across 4 runs; guardrail 17/17 green (incl. `smp`,
|
||||
`affinity`, `process-kill`, and `args`/`init`/`process` on the exit-endpoint spawn path)
|
||||
plus `aspace-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green.
|
||||
plus `address-space-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green.
|
||||
|
||||
> **Note (deferred):** a detached thread's stack is freed only at process exit (not by the
|
||||
> reaper on thread exit) — kernel user-stack tracking + reclaim is a later refinement. And
|
||||
@@ -204,7 +212,7 @@ plus `aspace-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-futex` passes, robust across 3 runs —
|
||||
the case's **ordered** regex asserts `waiting → waking → woke → PASS` on the serial
|
||||
stream (the handoff proof), and `thread-futex: timeout ok` confirms the timeout.
|
||||
Guardrail 18/18 green (incl. `sleep`/`event`/`ipc` blocking paths) + `aspace-refcount`,
|
||||
Guardrail 18/18 green (incl. `sleep`/`event`/`ipc` blocking paths) + `address-space-refcount`,
|
||||
`thread-spawn`, `thread-join`; `zig build` clean, `zig build test` green.
|
||||
|
||||
> **Note:** the kernel test checks only the freshest verdict marker via `bufferHas` (the
|
||||
@@ -244,11 +252,11 @@ green.
|
||||
|
||||
- [x] `getCurrentId` via a small `thread_self = 42` syscall (`runtime.Thread.getCurrentId`
|
||||
returns the kernel task id). **Per-thread `threadlocal` TLS is deferred** — no
|
||||
consumer needs it, and it would require context-switching `fs.base` per task (real
|
||||
kernel + per-switch cost) for an unused feature; threaded binaries have run fine
|
||||
consumer needs it, and it would require context-switching the thread pointer per task
|
||||
(real kernel + per-switch cost) for an unused feature; threaded binaries have run fine
|
||||
without it through M2–M5. threading.md's TLS reasoning already scoped it as
|
||||
deferred-unless-needed. When a consumer appears, the shape is: `thread_spawn`
|
||||
allocates a per-thread TLS block, sets `fs.base`, and the context switch saves/
|
||||
allocates a per-thread TLS block, sets the thread pointer, and the context switch saves/
|
||||
restores it.
|
||||
- [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same
|
||||
`Futex`/`Mutex`/`Condition` when wanted.
|
||||
@@ -265,20 +273,191 @@ green.
|
||||
|
||||
---
|
||||
|
||||
## Status: built
|
||||
## Status
|
||||
|
||||
M1–M6 complete. danos has `runtime.Thread` — `spawn`/`join`/`detach`, cross-core
|
||||
parallelism, futex, and `Mutex`/`Condition`/`Semaphore`, all over a private thread ABI
|
||||
behind the runtime. Deferred (with rationale, no consumer yet): `threadlocal` TLS,
|
||||
`RwLock`/`WaitGroup`, kernel clear-on-exit for a futex-completion `join`, a per-aspace
|
||||
mmap arena / thread-safe runtime heap, and host-side unit tests via a mockable `Futex`.
|
||||
**Phase 1 (M1–M6): built.** danos has `runtime.Thread` — `spawn`/`join`/`detach`,
|
||||
cross-core parallelism, futex, and `Mutex`/`Condition`/`Semaphore`, all over a private
|
||||
thread ABI behind the runtime.
|
||||
|
||||
**Phase 2 (M7–M11): built.** Thread-safe allocation (M7), a task reaper that reclaims dead
|
||||
tasks' kernel stacks (M8), endpoint-free `thread_join` (M9), the per-thread thread pointer (M10),
|
||||
and `RwLock`/`WaitGroup` + host-testable sync (M11). Two things stay deferred by design
|
||||
(no consumer): the Zig `threadlocal` *compiler* layer (M10) and detached-thread user-stack
|
||||
reclaim (M9) — both noted in place.
|
||||
|
||||
---
|
||||
|
||||
## Phase 2 — hardening (M7–M11)
|
||||
|
||||
The organising principle, so Phase 2 reinforces danos's goals rather than eroding them:
|
||||
|
||||
- **Everything a thread owns is reclaimed on process death.** Thread stacks, TLS blocks,
|
||||
and futex words live in the process's **address space**, and the kernel's per-process
|
||||
state is keyed by the address-space root — so the M1 refcount + `destroyAddressSpace` already
|
||||
free all of it when the last thread exits. A crashed or killed threaded process leaves
|
||||
**nothing** behind. Phase 2 closes the one thing that is *not* address-space-owned — the
|
||||
per-task **kernel** stack (kernel heap) — with a reaper (M8). This is the
|
||||
[resilience](resilience.md) restart guarantee, extended to threads.
|
||||
- **Kernel owns mechanism; the runtime owns policy.** The kernel maps pages, saves/
|
||||
restores the thread pointer, and reaps dead tasks; the runtime decides allocation, TLS layout,
|
||||
and lock algorithms. Every new kernel entry stays a private syscall behind the runtime
|
||||
([syscall.md](syscall.md)) — the ABI stays renumberable.
|
||||
- **The process is still the isolation and restart boundary.** Threads share fate within
|
||||
one process; Phase 2 never adds a way for one process to reach into another (the
|
||||
cross-process futex stays explicitly out of scope, below).
|
||||
|
||||
### M7 — Thread-safe allocation (the correctness gap) ✅
|
||||
|
||||
Today the mmap arena cursor is per-*task* and the runtime heap is unlocked, so two
|
||||
threads in one process that both allocate corrupt each other. The thread *machinery*
|
||||
avoids this (closure on the stack, stacks mmap'd only by the spawner), but real
|
||||
multi-threaded code would hit it. Closed it:
|
||||
|
||||
- [x] **Kernel — per-address-space mmap arena.** Grew M1's `address_space_refs` entry into the
|
||||
per-address-space object holding the `mmap`/`mmio` arena cursors (moved off `Task`);
|
||||
`scheduler.addressSpaceMmapNextPtr`/`addressSpaceDeviceMapNextPtr` expose them. `systemMmap`
|
||||
reserves a disjoint range under a *brief* lock, then maps **per page** under a
|
||||
short-held lock — not the whole grant — because the big lock is held with interrupts
|
||||
disabled, so pinning it across a multi-MiB memset+map froze other cores (it timed
|
||||
the `affinity` scenario out mid-bring-up). Freed at refcount zero, so the cursors
|
||||
vanish with the process.
|
||||
- [x] **Runtime — thread-safe heap.** The allocator's two free-list mutators
|
||||
(`rawAlloc`/`rawFree`) take a `Thread.Mutex`, gated on
|
||||
`!@import("builtin").single_threaded` so single-threaded binaries compile it out and
|
||||
pay nothing. Uncontended acquisition is a single CAS (no syscall).
|
||||
- [x] `-Dtest-case=thread-alloc` (`smp: 4`): 4 threads each do 500 `alloc`/fill/verify/
|
||||
`free` cycles of varied sizes; each block is filled with a per-thread pattern and
|
||||
verified before free, so any overlap between concurrent allocations is caught.
|
||||
|
||||
**Gate (met):** `thread-alloc` passes (3× non-flaky); full guardrail 23/23 green,
|
||||
`zig build`/`zig build test` clean.
|
||||
|
||||
> **Also fixed here:** the `affinity` guardrail's fixed-count busy-loop (`while (spins <
|
||||
> 3e9)`) had codegen-dependent wall-time — adding a function to `tests.zig` flipped how
|
||||
> the optimiser compiled it, swinging affinity from ~4 s to ~63 s and timing it out.
|
||||
> Reworked it (and the settle loop) to wait on the wall clock instead, so its duration is
|
||||
> independent of unrelated code changes.
|
||||
|
||||
### M8 — The task reaper (cleanup + resilience) ✅
|
||||
|
||||
A dead task's **kernel** stack was leaked ("no reaper yet") — every process *and* thread
|
||||
death lost one, so a crash loop bled kernel memory. The reaper fixes it and serves the
|
||||
[resilience](resilience.md) restart goal directly:
|
||||
|
||||
- [x] A dying task cannot free the kernel stack it runs on, so `exit()`/`exitUserLocked`
|
||||
record it in a **per-core `reap_after_switch` slot** and switch away; the task that
|
||||
resumes on that core frees the stack in `switchTo`'s tail (it's on its own stack, the
|
||||
big lock is still held so the slot can't have been reused). A **tick-time drain**
|
||||
(`reapKillPendingLocked`) is the safety net for the case where the next task is
|
||||
*fresh* (enters via the trampoline, bypassing `switchTo`'s tail). A task killed while
|
||||
*not* running is freed immediately in `destroyTaskLocked`. A `live_stack_bytes`
|
||||
counter is the observable. *(Detached-thread user-stack reclaim moves to M9, which
|
||||
adds the joinable/detached flag.)*
|
||||
- [x] `-Dtest-case=task-reap` (`smp: 4`): spawn and kill 12 processes; poll the
|
||||
test-observable `scheduler.liveStackBytes()` until it returns to **baseline** (a
|
||||
correct reaper gets there in a few ms; a genuine leak times out) — every kernel
|
||||
stack reclaimed, no leak. Threads exit through the same `exitUserLocked`, so covered.
|
||||
|
||||
**Gate (met):** `task-reap` passes (5× isolated + 2× in the full batch); `fault-recovery`,
|
||||
`supervision`, `process-kill`, `address-space-refcount`, `smp`, `affinity` all still green (24/24
|
||||
full guardrail); `zig build`/`zig build test` clean.
|
||||
|
||||
> **Bug found + fixed here (touches every context switch):** the post-`switchContext` reap
|
||||
> first read the `pc` **parameter**, but a task that migrated cores carries a *stale* `pc`
|
||||
> in its saved `switchTo` frame — so it read the wrong core's slot and freed a live stack
|
||||
> (a #GP under SMP). Fixed to re-fetch `thisCpu()` after the switch (the switch only swaps
|
||||
> stacks on the current core).
|
||||
|
||||
### M9 — Futex-completion join (retire the per-thread endpoint)
|
||||
|
||||
With the reaper (M8) able to act *after* a thread is fully off its stack, migrate `join`
|
||||
to the std shape and drop M3's per-thread exit endpoint:
|
||||
|
||||
- [x] A **`thread_join(tid)` syscall** (not a user futex word): it blocks the caller until
|
||||
the task with id `tid` exits, and the exit paths call `wakeJoinersLocked`. `join`
|
||||
only reclaims the joined thread's **user** stack, which the thread vacates the moment
|
||||
it enters the kernel to exit — so waking at *exit* time (not reap time) is safe, and
|
||||
no reaper/address-space juggling or user-memory write is needed. This is equally
|
||||
std-shaped (like `pthread_join`) and much simpler/safer than the planned
|
||||
reaper-written completion word. `thread_spawn` no longer takes an exit endpoint (the
|
||||
runtime passes `no_cap`); the per-thread IPC endpoint is gone.
|
||||
- [x] `thread-join` passes on the new path, and its join mode now runs **40 spawn+join
|
||||
cycles** — under the old per-thread-endpoint scheme those leaked handles would
|
||||
exhaust the 16-slot handle table; here they all succeed, proving join is endpoint-free.
|
||||
|
||||
**Gate (met):** `thread-join` passes (3× isolated) on the `thread_join` path; full
|
||||
guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-reap`);
|
||||
`zig build`/`zig build test` clean.
|
||||
|
||||
> **Reaper hardened here (fixes an M8 flake).** M8's single per-core reap slot could be
|
||||
> *overwritten* by a second death on that core before the first drained (a fresh-task/SMP
|
||||
> timing window) — an intermittent one-stack leak (`task-reap` flaked ~20%). Replaced it
|
||||
> with a per-core reap **list** plus a `.reaping` task state so a pending slot can't be
|
||||
> reused before its stack is freed. `task-reap` now 11/11 isolated + 2× in the batch.
|
||||
|
||||
> **Deferred:** detached-thread **user-stack** reclaim (still freed at process exit, as in
|
||||
> M3). Doing it in the reaper needs the saved address space + stack range and a
|
||||
> translate/unmap in a not-currently-loaded address space — real complexity for a bounded leak.
|
||||
> A follow-up when a consumer needs it.
|
||||
|
||||
### M10 — Per-thread TLS: the thread-pointer mechanism ✅
|
||||
|
||||
Give each thread its own thread pointer and private TLS storage — the foundation
|
||||
self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on.
|
||||
|
||||
- [x] **Kernel** stores `thread_pointer` on `Task` and restores it on every context switch
|
||||
**only when it changes** (the same conditional-load discipline as CR3;
|
||||
`architecture.setThreadPointer` → `wrmsr IA32_FS_BASE` on x86_64). A
|
||||
`set_thread_pointer(addr)` = 44 syscall sets the caller's `thread_pointer` and loads it
|
||||
now. The kernel never touches FS, so there is no swapgs complication.
|
||||
- [x] **Runtime** lays a small per-thread TLS block at the top of each thread's stack
|
||||
(self-pointer at `%fs:0` + scratch slots) and the thread trampoline calls
|
||||
`set_thread_pointer` before any user code — so every spawned thread has a private,
|
||||
switch-stable thread pointer. Reclaimed with the stack.
|
||||
- [x] `-Dtest-case=thread-tls` (`smp: 4`): two threads each write a unique marker to their
|
||||
own `%fs:8` slot and — after both have written — read it back; a shared (non-per-thread)
|
||||
FS base would clobber one and cause cross-talk. Both read their own marker → pass.
|
||||
|
||||
**Gate (met):** `thread-tls` passes (3×); full guardrail 25/25 (the switch-time thread-pointer
|
||||
restore touches every context switch); `zig build`/`zig build test` clean.
|
||||
|
||||
> **Deferred: the Zig `threadlocal` *compiler* layer.** Real `threadlocal` variables need
|
||||
> the ELF **variant-II TLS** surface — `.tdata`/`.tbss` sections + a `PT_TLS` program header
|
||||
> in `user.ld`, a runtime that copies the template with exact negative-offset layout, and
|
||||
> the `.large`-code-model TLS section names — a high-uncertainty lift for a feature with
|
||||
> **no consumer today** (threading.md scopes it "only if a consumer needs it"). What lands
|
||||
> here is the load-bearing piece — the per-thread thread pointer, context-switched — so adding the
|
||||
> compiler layer later is purely runtime+linker work on top, no kernel change. `getCurrentId`
|
||||
> stays the `thread_self` syscall (M6) rather than an fs self-slot (which would need the
|
||||
> main thread's TLS set up in `_start` too).
|
||||
|
||||
**Gate:** `thread-tls` passes; full `thread-*` suite + guardrail green.
|
||||
|
||||
### M11 — `RwLock`, `WaitGroup`, and host-testable sync ✅
|
||||
|
||||
- [x] `runtime.Thread.RwLock` (reader-preferring: `>0` readers / `-1` writer / `0` free,
|
||||
with `lock`/`tryLock`/`unlock` + `lockShared`/`tryLockShared`/`unlockShared`) and
|
||||
`WaitGroup` (`start`/`finish`/`wait`), both on the existing `Mutex`/`Condition`.
|
||||
- [x] A compile-time `Futex` seam gated on `builtin.os.tag == .freestanding`: the futex
|
||||
syscalls on danos, a spin+yield mock off-target (Zig 0.16 has no `std.Thread.Futex`;
|
||||
`wake` is a no-op since the state machines re-check). `thread.zig` is wired into
|
||||
`zig build test`, so `Mutex`/`RwLock`/`WaitGroup` run as **host unit tests** with real
|
||||
`std.Thread` threads (`test` blocks only compile under test).
|
||||
- [x] `-Dtest-case=thread-rwlock` (`smp: 4`): 2 writers set both halves of a value under
|
||||
the exclusive lock while 3 readers check the halves match under the shared lock —
|
||||
zero half-write observations across ~150k reads. Host tests cover the Mutex,
|
||||
RwLock, and WaitGroup state machines.
|
||||
|
||||
**Gate (met):** `zig build test` covers the sync primitives (host threads); `thread-rwlock`
|
||||
passes (3×); full Done gate **26/26** (whole `thread-*` suite + guardrail); `zig build`
|
||||
clean.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Cross-process shared-memory futex** — the `(aspace, vaddr)` key can become a
|
||||
physical-address key so two processes share a futex through an [shm](display-v2.md)
|
||||
- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a
|
||||
physical-address key so two processes share a futex through a [shared-memory](display-v2.md)
|
||||
region. Not needed for intra-process threads.
|
||||
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||
|
||||
+44
-28
@@ -2,12 +2,13 @@
|
||||
|
||||
A note on danos **threads** — several tasks sharing one address space — provided by a
|
||||
`runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M6, see
|
||||
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core
|
||||
parallelism, a futex (`futex_wait`/`futex_wake`), and a futex-backed
|
||||
`Mutex`/`Condition`/`Semaphore`, plus `getCurrentId`/`currentCore`. Deferred by design
|
||||
(no consumer yet): per-thread `threadlocal` TLS, `RwLock`/`WaitGroup`, and migrating
|
||||
`join` to a futex completion word — see the plan's M5/M6 notes. The analysis is against
|
||||
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M11, see
|
||||
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core parallelism,
|
||||
a futex, `Mutex`/`Condition`/`Semaphore`/`RwLock`/`WaitGroup`, `getCurrentId`/`currentCore`,
|
||||
per-thread thread-pointer TLS, thread-safe allocation, and a task reaper that reclaims dead
|
||||
tasks' kernel stacks. Deferred by design (no consumer yet): the Zig `threadlocal`
|
||||
*compiler* layer (the per-thread thread pointer is in place, so it's runtime+linker work on top) and
|
||||
detached-thread user-stack reclaim — see the plan's M9/M10 notes. The analysis is against
|
||||
**Zig 0.16** (the pinned toolchain); `std.Thread`'s internals move between releases, so
|
||||
treat upstream shapes as "0.16.x."
|
||||
|
||||
@@ -132,7 +133,7 @@ Deviations from `std.Thread`, called out honestly:
|
||||
## Kernel primitives (new private syscalls)
|
||||
|
||||
Four new entries extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shm_physical = 36`, each with a `library/runtime` wrapper:
|
||||
`shared_memory_physical = 36`, each with a `library/runtime` wrapper:
|
||||
|
||||
| Syscall | Signature | Purpose |
|
||||
|---|---|---|
|
||||
@@ -147,10 +148,10 @@ Plus one invariant change with no new syscall: **address-space reference countin
|
||||
|
||||
### Address-space reference counting
|
||||
|
||||
Today an address space is 1:1 with a task: `spawnUserLocked` records `aspace` on the
|
||||
Task, and teardown does `destroyAddressSpace(t.aspace)` when **any** user task exits
|
||||
Today an address space is 1:1 with a task: `spawnUserLocked` records `address_space` on the
|
||||
Task, and teardown does `destroyAddressSpace(t.address_space)` when **any** user task exits
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
one `aspace`, so the first to exit would rip the address space out from under its
|
||||
one `address_space`, so the first to exit would rip the address space out from under its
|
||||
siblings.
|
||||
|
||||
Fix: a small refcount keyed by the address-space root (`createAddressSpace` in
|
||||
@@ -161,7 +162,7 @@ that must land and be proven before anything shares an address space.
|
||||
|
||||
### `thread_spawn` and the trampoline
|
||||
|
||||
The scheduler already accepts an arbitrary `aspace` and does **not** smuggle values
|
||||
The scheduler already accepts an arbitrary `address_space` and does **not** smuggle values
|
||||
through registers — `startUserTask` reads the entry/stack from the Task and
|
||||
`jumpToUser`s ([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread
|
||||
path clean:
|
||||
@@ -170,7 +171,7 @@ path clean:
|
||||
`{ fn_ptr, args_tuple, completion }`, the std "Instance" pattern — and writes the
|
||||
closure pointer to the **top word of the new stack**.
|
||||
2. It calls `thread_spawn(entry = &threadTrampoline, stack_top, arg = closure_ptr)`.
|
||||
The kernel calls the same `spawnUserLocked` path with the **caller's aspace**
|
||||
The kernel calls the same `spawnUserLocked` path with the **caller's address space**
|
||||
(refcount++), `entry`, and `user_sp = stack_top`.
|
||||
3. `threadTrampoline` (a small runtime shim) reads the closure off its stack, calls
|
||||
the user function, then calls `thread_exit`. No new register ABI — the closure
|
||||
@@ -184,7 +185,7 @@ Unlike a process start, there is **no** System V argc/argv/auxv block
|
||||
|
||||
- **`thread_exit`** marks the task dead and hands the kernel the thread's user-stack
|
||||
range. The kernel reaps the task on the scheduler (already running on a *kernel*
|
||||
stack, so it can safely unmap the user stack), decrements the aspace refcount, and
|
||||
stack, so it can safely unmap the user stack), decrements the address-space refcount, and
|
||||
frees the task slot.
|
||||
- **`join` — Stage 1** reuses the existing exit-notification machinery
|
||||
([process-lifecycle.md](process-lifecycle.md)): `spawn` passes a per-thread
|
||||
@@ -205,12 +206,12 @@ Unlike a process start, there is **no** System V argc/argv/auxv block
|
||||
call the futex wrappers on the slow path — the same construction `std.Thread` uses,
|
||||
so the algorithms port directly.
|
||||
|
||||
Keying: threads share an address space, so a **virtual address within that aspace**
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(aspace_root, vaddr)`.
|
||||
Keying by the **physical** address instead (translate `vaddr -> paddr` on entry) is a
|
||||
Keying: threads share an address space, so a **virtual address within that address space**
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`.
|
||||
Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a
|
||||
deliberate forward door: it lets two *processes* share a futex through an
|
||||
[shm](display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-aspace key and note the physical-key upgrade.
|
||||
[shared-memory](display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-address-space key and note the physical-key upgrade.
|
||||
|
||||
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||
other work or `hlt` ([halting.md](halting.md)). This is why futex is a locked
|
||||
@@ -218,12 +219,12 @@ decision, not a "maybe later."
|
||||
|
||||
### TLS and `getCurrentId`
|
||||
|
||||
danos sets up no `fs.base` TLS today (fine under `single_threaded`). Two scoped needs:
|
||||
danos sets up no thread-pointer TLS today (fine under `single_threaded`). Two scoped needs:
|
||||
|
||||
- **`getCurrentId`** returns the kernel task id — either a trivial syscall or, better,
|
||||
a value the runtime stashes in a per-thread control block.
|
||||
- **`threadlocal` variables** need a real per-thread TLS block and `fs.base` set per
|
||||
thread. `thread_spawn` sets `fs.base` to a runtime-allocated per-thread block; full
|
||||
- **`threadlocal` variables** need a real per-thread TLS block and the thread pointer set per
|
||||
thread. `thread_spawn` sets the thread pointer to a runtime-allocated per-thread block; full
|
||||
`threadlocal` support is Stage 3, only if a consumer needs it. Nothing in the core
|
||||
spawn/join/mutex path requires `threadlocal`.
|
||||
|
||||
@@ -237,17 +238,32 @@ it may call `runtime.Thread.spawn`. Everyone else stays single-threaded and lean
|
||||
## Interaction with the rest of the kernel
|
||||
|
||||
- **Scheduler / SMP** ([scheduling.md](scheduling.md), [smp.md](smp.md)): a thread is
|
||||
just another `Task` with an `aspace` shared with its siblings; the existing
|
||||
just another `Task` with an `address_space` shared with its siblings; the existing
|
||||
per-core ready queues, priorities, and affinity apply unchanged. Threads of one
|
||||
process can run on different cores simultaneously — that is the point.
|
||||
- **Halting** ([halting.md](halting.md)): futex-parked waiters keep the "idle core
|
||||
halts" property intact under lock contention — no busy-wait.
|
||||
- **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): killing a process
|
||||
must kill *all* its threads and only then drop the last aspace ref. The kill path
|
||||
already targets a process; it fans out to every task on that aspace.
|
||||
must kill *all* its threads and only then drop the last address-space ref. The kill path
|
||||
already targets a process; it fans out to every task on that address space.
|
||||
- **Resilience** ([resilience.md](resilience.md)): a faulting thread kills its whole
|
||||
process (shared fate). The supervisor restarts the **process**, which respawns its
|
||||
threads from a known-good state — restart granularity stays the process.
|
||||
- **IPC — two consequences threads forced ([ipc.md](ipc.md)):**
|
||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||
A thread that needs to reach an endpoint another thread owns looks it up
|
||||
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
||||
This is how the display's mouse-listener thread reaches the compositor loop's endpoint
|
||||
to poke it awake (docs/display.md).
|
||||
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
||||
`create_ipc_endpoint`/`ipc_register`/`ipc_lookup` allocate from the kernel heap and
|
||||
mutate the global service registry, endpoint refcounts, and handle tables. Those paths
|
||||
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
||||
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
||||
`send` already did — the kernel heap has no lock of its own yet (heap.zig: "a lock comes
|
||||
with threads/SMP"), so the big lock is what keeps its callers serialized.
|
||||
|
||||
## Build-out plan (staged, each gate serial-checkable)
|
||||
|
||||
@@ -257,8 +273,8 @@ The ordered, `/loop`-runnable milestones live in
|
||||
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||
|
||||
- **Stage 0 — address-space refcount.** Refcount on the aspace root; teardown destroys
|
||||
at zero. No API yet; nothing shares an aspace, so refcount is 1 everywhere.
|
||||
- **Stage 0 — address-space refcount.** Refcount on the address-space root; teardown destroys
|
||||
at zero. No API yet; nothing shares an address space, so refcount is 1 everywhere.
|
||||
*Gate:* the full QEMU suite stays green (no regression) — proves the reframing is
|
||||
invisible until used.
|
||||
- **Stage 1 — spawn / join / detach.** `thread_spawn` + `thread_exit`, the trampoline,
|
||||
@@ -272,7 +288,7 @@ a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
word. *Gate:* `-Dtest-case=thread-mutex` — a bounded producer/consumer over a
|
||||
`Mutex` + `Condition` moves K items with no lost wakeups and no busy-wait (assert
|
||||
the consumer blocked, e.g. via a low idle tick count).
|
||||
- **Stage 3 — polish.** Per-thread TLS / `fs.base` and `threadlocal` (only if a
|
||||
- **Stage 3 — polish.** Per-thread TLS / thread pointer and `threadlocal` (only if a
|
||||
consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired
|
||||
into [test/qemu_test.py](../test/qemu_test.py).
|
||||
|
||||
@@ -292,7 +308,7 @@ are — user code never names a syscall.
|
||||
- **No thread priorities distinct from the process.** Threads inherit the process
|
||||
priority; per-thread priority is a later question if it ever earns its keep.
|
||||
- **No cross-process shared-memory futex yet** — the physical-address key leaves the
|
||||
door open, but the first cut is private-per-aspace.
|
||||
door open, but the first cut is private-per-address-space.
|
||||
- **No `pthread`/POSIX surface.** The API is `std.Thread`-shaped Zig, nothing more.
|
||||
|
||||
## The self-hosting endgame
|
||||
|
||||
+8
-5
@@ -119,7 +119,7 @@ One table entry per kernel call, C ABI (System V AMD64), names prefixed
|
||||
returns are `u64`, errors return as negative values exactly as today.
|
||||
|
||||
The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||
(vaddr + paddr), `msi_bind` (address + data), `shm_create` (vaddr + handle) —
|
||||
(virtual_address + physical_address), `msi_bind` (address + data), `shared_memory_create` (virtual_address + handle) —
|
||||
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||
C-ABI spelling of the existing convention, at zero cost.
|
||||
@@ -129,11 +129,12 @@ Grouped as `abi.zig` groups them:
|
||||
| Group | Functions |
|
||||
|-------|-----------|
|
||||
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shm_create`, `danos_shm_map`, `danos_shm_physical` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shared_memory_create`, `danos_shared_memory_map`, `danos_shared_memory_physical` |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
||||
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||
| diagnostics | `danos_debug_write`, `danos_klog_read` |
|
||||
| diagnostics | `danos_debug_write` (leveled, kernel-stamped records), `danos_klog_read`, `danos_klog_status` |
|
||||
| filesystem naming | `danos_fs_resolve`, `danos_fs_node`, `danos_fs_mount`, `danos_fs_unmount` (naming only — file DATA still crosses the vfs-protocol IPC, see below) |
|
||||
|
||||
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
||||
@@ -197,8 +198,10 @@ Phased so every step ships alone (the M-milestone discipline):
|
||||
across processes stays what it is today: a service behind IPC, or source
|
||||
compiled into each binary.
|
||||
- **No file/device I/O in the vDSO.** The microkernel line doesn't move: the
|
||||
vDSO wraps the same deliberately tiny table (docs/syscall.md); files are
|
||||
still the VFS server's business over IPC.
|
||||
vDSO wraps the same deliberately tiny table (docs/syscall.md). The kernel
|
||||
resolves file NAMES (`fs_resolve` — the mount table moved in-kernel), but
|
||||
file data is still the filesystem server's business over the vfs-protocol
|
||||
IPC; the kernel never blocks on a userspace filesystem.
|
||||
- **No fast-path user-mode implementations yet.** Linux's vDSO exists mostly
|
||||
to answer `gettimeofday` without a kernel entry. `danos_clock` could one
|
||||
day read the calibrated TSC in user mode the same way — the blob is where
|
||||
|
||||
+26
-18
@@ -1,20 +1,24 @@
|
||||
# The VFS wire protocol
|
||||
|
||||
> **Status:** built and spoken today between `runtime.fs` (the client) and the
|
||||
> VFS server (`system/services/vfs`), with mounted backends (the FAT server)
|
||||
> speaking the same protocol behind the router. The Zig source of truth is
|
||||
> `system/services/vfs/protocol.zig` (the `vfs-protocol` module), whose unit
|
||||
> tests pin the sizes and values below. This page is the **language-neutral
|
||||
> wire specification** of that contract — what a Rust or C client implements
|
||||
> ([vdso.md](vdso.md) explains why the IPC protocols, not the syscall
|
||||
> numbers, are danos's public ABI).
|
||||
> filesystem BACKENDS (the FAT server). The mount router lives in the
|
||||
> **kernel** (`system/kernel/vfs.zig`): `fs_resolve` routes a path and either
|
||||
> serves it directly (the read-only /system initrd mount, via `fs_node`) or
|
||||
> redirects the caller to the owning backend's endpoint plus the rewritten
|
||||
> mount-relative path — after which the client speaks THIS protocol to the
|
||||
> backend, unchanged. The Zig source of truth is `system/vfs-protocol.zig`
|
||||
> (the `vfs-protocol` module), whose unit tests pin the sizes and values
|
||||
> below. This page is the **language-neutral wire specification** of that
|
||||
> contract — what a Rust or C client implements ([vdso.md](vdso.md) explains
|
||||
> why the IPC protocols, not the syscall numbers, are danos's public ABI).
|
||||
|
||||
## Transport
|
||||
|
||||
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
|
||||
docs/ipc.md): the client sends one message and blocks; the server replies
|
||||
with one message. The endpoint is found by well-known service id
|
||||
(`ipc_lookup`, service id **1** = vfs).
|
||||
with one message. The endpoint comes from the kernel's `fs_resolve` — which
|
||||
also hands back the path rewritten relative to the mount — not from a
|
||||
registry lookup. (Service id 1, the old userspace router, is retired.)
|
||||
|
||||
- A message is at most **256 bytes** (`message_maximum`).
|
||||
- A request is a fixed 32-byte **Request** header followed by an inline
|
||||
@@ -26,8 +30,11 @@ with one message. The endpoint is found by well-known service id
|
||||
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||
(`extern struct`), offsets given below so nothing need be inferred.
|
||||
|
||||
The kernel never parses any of this — it only moves the bytes
|
||||
(docs/syscall.md); files are entirely a user-space affair.
|
||||
The kernel resolves NAMES (the mount table) but never parses these
|
||||
messages — it moves the bytes; file state is entirely the backend's affair.
|
||||
With clients holding backend node ids directly, a backend records each open
|
||||
handle's owner and sweeps a dead client's handles via the published process
|
||||
exit events.
|
||||
|
||||
## Request header — 32 bytes
|
||||
|
||||
@@ -90,13 +97,14 @@ Notes per operation:
|
||||
position. Each call returns exactly one entry; the client increments the
|
||||
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
|
||||
have been opened with the `directory` flag.
|
||||
- **mount** — the one operation that passes a **capability**: the caller
|
||||
(a filesystem server, e.g. FAT) sends its own request endpoint as the
|
||||
`ipc_call` capability argument, and the router forwards everything under
|
||||
the mount point to it — speaking this same protocol, with paths rewritten
|
||||
relative to the mount. Prefixes match at path boundaries only
|
||||
(`/mnt/usb` never captures `/mnt/usbextra`); the longest matching prefix
|
||||
wins.
|
||||
- **mount / unmount** — RETIRED from the wire: mounting is the `fs_mount`
|
||||
syscall now (a filesystem server passes its endpoint handle; possession is
|
||||
the capability, exactly the trust of the old cap-passing op). The op
|
||||
numbers stay reserved. Mount-prefix semantics are unchanged: prefixes
|
||||
match at path boundaries only (`/mnt/usb` never captures `/mnt/usbextra`),
|
||||
the longest matching prefix wins, and an optional backend-side rewrite
|
||||
prefix maps a mount into the backend's namespace (fat serves `/mnt/usb`
|
||||
from its volume root and `/var` from its `/var` subtree).
|
||||
- **rename** — same-directory rename only (the router requires old and new to
|
||||
resolve under one mount).
|
||||
|
||||
|
||||
+129
-49
@@ -1,5 +1,5 @@
|
||||
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
|
||||
//! lists files served by the user-space VFS (system/services/vfs), each call
|
||||
//! lists files through the kernel VFS root (resolve + redirect), each call
|
||||
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
|
||||
//! danos programs use directly; it is also where the file operations that later
|
||||
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
|
||||
@@ -12,6 +12,7 @@
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("vfs-protocol");
|
||||
|
||||
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||
@@ -60,23 +61,37 @@ pub const OpenOptions = struct {
|
||||
}
|
||||
};
|
||||
|
||||
// The VFS server endpoint, looked up once by well-known id and cached.
|
||||
var vfs_handle: ipc.Handle = 0;
|
||||
var vfs_resolved = false;
|
||||
fn vfs() ?ipc.Handle {
|
||||
if (!vfs_resolved) {
|
||||
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||
vfs_resolved = true;
|
||||
// The route to a path: the kernel resolves (fs_resolve) and either serves the
|
||||
// node itself (the initrd at /system — a permanent token) or redirects us to
|
||||
// the owning filesystem backend's endpoint, to which we speak the vfs-protocol
|
||||
// rendezvous directly with the rewritten mount-relative path.
|
||||
const Route = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: ipc.Handle, path: [224]u8, path_len: usize },
|
||||
|
||||
fn backendPath(self: *const Route) []const u8 {
|
||||
return self.backend.path[0..self.backend.path_len];
|
||||
}
|
||||
};
|
||||
|
||||
fn resolve(path: []const u8, flags: usize) ?Route {
|
||||
var out: [224]u8 = undefined;
|
||||
const route = system.fsResolve(path, flags, &out) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .kernel = token },
|
||||
.backend => |b| {
|
||||
var r: Route = .{ .backend = .{ .handle = b.handle, .path = undefined, .path_len = b.path_len } };
|
||||
@memcpy(r.backend.path[0..b.path_len], out[0..b.path_len]);
|
||||
return r;
|
||||
},
|
||||
}
|
||||
return vfs_handle;
|
||||
}
|
||||
|
||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||
|
||||
// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||
// One request/reply round trip: [Request header][send payload] -> backend ->
|
||||
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
const h = vfs() orelse return null;
|
||||
fn transact(h: ipc.Handle, request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, protocol.maximum_payload);
|
||||
@@ -95,13 +110,21 @@ fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
pub const File = struct {
|
||||
node: u64,
|
||||
offset: u64 = 0,
|
||||
/// The owning backend's endpoint, or null for a kernel-served node (the
|
||||
/// read-only /system tree), whose `node` is a permanent fs_node token.
|
||||
backend: ?ipc.Handle = null,
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
|
||||
/// null on error.
|
||||
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||
const h = self.backend orelse {
|
||||
const n = system.fsNodeRead(self.node, self.offset, buffer) orelse return null;
|
||||
self.offset += n;
|
||||
return n;
|
||||
};
|
||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, &.{}, buffer) orelse return null;
|
||||
const r = transact(h, request, &.{}, buffer) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
@@ -109,11 +132,13 @@ pub const File = struct {
|
||||
|
||||
/// Write `data` at the current offset; returns the count written. A single
|
||||
/// call is capped at the VFS payload size, so the return may be short — use
|
||||
/// `writeAll` to write the whole slice. Null on error.
|
||||
/// `writeAll` to write the whole slice. Null on error (kernel-served nodes
|
||||
/// are read-only).
|
||||
pub fn write(self: *File, data: []const u8) ?usize {
|
||||
const h = self.backend orelse return null;
|
||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, data[0..want], &.{}) orelse return null;
|
||||
const r = transact(h, request, data[0..want], &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
@@ -138,27 +163,40 @@ pub const File = struct {
|
||||
|
||||
/// This file's metadata.
|
||||
pub fn attributes(self: *File) ?Attributes {
|
||||
const h = self.backend orelse {
|
||||
const a = system.fsNodeStatus(self.node) orelse return null;
|
||||
return .{ .size = a.size, .kind = if (a.kind == system.file_kind_directory) .directory else .regular, .mtime = a.mtime };
|
||||
};
|
||||
const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &buffer) orelse return null;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return null;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]);
|
||||
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||
}
|
||||
|
||||
/// Release the VFS's open handle for this file.
|
||||
/// Release the backend's open handle for this file. Kernel-served node
|
||||
/// tokens are permanent — nothing to release.
|
||||
pub fn close(self: *File) void {
|
||||
const h = self.backend orelse return;
|
||||
const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(request, &.{}, &.{});
|
||||
_ = transact(h, request, &.{}, &.{});
|
||||
}
|
||||
};
|
||||
|
||||
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
|
||||
pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = options.wireFlags() };
|
||||
const r = transact(request, path, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node };
|
||||
const route = resolve(path, options.wireFlags()) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .node = token, .backend = null },
|
||||
.backend => |b| {
|
||||
const relative = route.backendPath();
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() };
|
||||
const r = transact(b.handle, request, relative, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node, .backend = b.handle };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// A path's metadata without keeping it open (open -> status -> close).
|
||||
@@ -190,13 +228,27 @@ pub const Entry = struct {
|
||||
pub const Directory = struct {
|
||||
node: u64,
|
||||
cursor: u64 = 0,
|
||||
backend: ?ipc.Handle = null,
|
||||
|
||||
/// Fill `entry` with the next directory entry; false at end of directory or
|
||||
/// on error.
|
||||
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||
const h = self.backend orelse {
|
||||
var buffer: [@sizeOf(system.DirectoryEntryHeader) + 64]u8 = undefined;
|
||||
const n = system.fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false;
|
||||
if (n < @sizeOf(system.DirectoryEntryHeader)) return false; // end
|
||||
const header = std.mem.bytesToValue(system.DirectoryEntryHeader, buffer[0..@sizeOf(system.DirectoryEntryHeader)]);
|
||||
entry.kind = if (header.kind == system.file_kind_directory) .directory else .regular;
|
||||
entry.size = header.size;
|
||||
const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(system.DirectoryEntryHeader)..][0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
};
|
||||
const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
const r = transact(request, &.{}, &buffer) orelse return false;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return false;
|
||||
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||
if (r.payload.len < protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]);
|
||||
@@ -210,9 +262,9 @@ pub const Directory = struct {
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release the VFS's open handle for this directory.
|
||||
/// Release the backend's open handle for this directory.
|
||||
pub fn close(self: *Directory) void {
|
||||
var f = File{ .node = self.node };
|
||||
var f = File{ .node = self.node, .backend = self.backend };
|
||||
f.close();
|
||||
}
|
||||
};
|
||||
@@ -220,13 +272,18 @@ pub const Directory = struct {
|
||||
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
|
||||
pub fn openDirectory(path: []const u8) ?Directory {
|
||||
const file = open(path, .{ .directory = true }) orelse return null;
|
||||
return .{ .node = file.node };
|
||||
return .{ .node = file.node, .backend = file.backend };
|
||||
}
|
||||
|
||||
// A path-based request that returns only a status (mkdir, unlink).
|
||||
// A path-based request that returns only a status (mkdir, unlink). Kernel-served
|
||||
// paths (the read-only /system) refuse mutation by construction: the resolve
|
||||
// must land on a backend.
|
||||
fn pathOperation(operation: protocol.Operation, path: []const u8) bool {
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = 0 };
|
||||
const r = transact(request, path, &.{}) orelse return false;
|
||||
const route = resolve(path, 0) orelse return false;
|
||||
if (route != .backend) return false;
|
||||
const relative = route.backendPath();
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
const r = transact(route.backend.handle, request, relative, &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
@@ -236,39 +293,62 @@ pub fn makeDirectory(path: []const u8) bool {
|
||||
return pathOperation(.mkdir, path);
|
||||
}
|
||||
|
||||
/// Create every missing directory along `path` (mkdir -p). Probes each prefix
|
||||
/// with `exists` first — a FAT mkdir of an existing name is refused, and the
|
||||
/// probe keeps the common "already there" case cheap. Returns true when the
|
||||
/// whole path exists afterwards.
|
||||
pub fn makePath(path: []const u8) bool {
|
||||
var end: usize = 0;
|
||||
while (end < path.len) {
|
||||
end += 1;
|
||||
while (end < path.len and path[end] != '/') end += 1;
|
||||
const prefix = path[0..end];
|
||||
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
||||
// Best-effort per prefix: components at or above a mount point ("/mnt")
|
||||
// are router names, not filesystem nodes — they neither exist as nodes
|
||||
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
||||
if (!exists(prefix)) _ = makeDirectory(prefix);
|
||||
}
|
||||
return exists(path);
|
||||
}
|
||||
|
||||
/// Remove the file at `path`. Returns true on success. Directories are refused
|
||||
/// (a separate directory-removal would have to check emptiness).
|
||||
pub fn remove(path: []const u8) bool {
|
||||
return pathOperation(.unlink, path);
|
||||
}
|
||||
|
||||
/// Rename `old_path` to `new_path`. Both must be in the same directory (same-
|
||||
/// directory, 8.3-name rename only for now). Returns true on success.
|
||||
/// Rename `old_path` to `new_path`. Both must resolve to the SAME filesystem
|
||||
/// backend (same-directory, 8.3-name rename only for now). Returns true on
|
||||
/// success.
|
||||
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
const total = old_path.len + 1 + new_path.len;
|
||||
const old_route = resolve(old_path, 0) orelse return false;
|
||||
const new_route = resolve(new_path, 0) orelse return false;
|
||||
if (old_route != .backend or new_route != .backend) return false;
|
||||
if (old_route.backend.handle != new_route.backend.handle) return false; // cross-filesystem
|
||||
const old_relative = old_route.backendPath();
|
||||
const new_relative = new_route.backendPath();
|
||||
const total = old_relative.len + 1 + new_relative.len;
|
||||
if (total > protocol.maximum_payload) return false;
|
||||
var payload: [protocol.maximum_payload]u8 = undefined;
|
||||
@memcpy(payload[0..old_path.len], old_path);
|
||||
payload[old_path.len] = 0;
|
||||
@memcpy(payload[old_path.len + 1 ..][0..new_path.len], new_path);
|
||||
@memcpy(payload[0..old_relative.len], old_relative);
|
||||
payload[old_relative.len] = 0;
|
||||
@memcpy(payload[old_relative.len + 1 ..][0..new_relative.len], new_relative);
|
||||
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const r = transact(request, payload[0..total], &.{}) orelse return false;
|
||||
const r = transact(old_route.backend.handle, request, payload[0..total], &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||
/// the VFS then routes everything under `target` to that backend. This is the one
|
||||
/// call that hands the VFS a capability (the backend endpoint). Returns true on
|
||||
/// success.
|
||||
/// the kernel VFS then routes everything under `target` to that backend.
|
||||
/// Possession of the endpoint handle is the capability. Returns true on success.
|
||||
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||
const h = vfs() orelse return false;
|
||||
const request = protocol.Request{ .operation = .mount, .node = 0, .offset = 0, .len = @intCast(target.len), .flags = 0 };
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const tlen = @min(target.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..tlen], target[0..tlen]);
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const result = ipc.callCap(h, message[0 .. protocol.request_size + tlen], &rbuf, backend) catch return false;
|
||||
if (result.len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]).status == 0;
|
||||
return system.fsMount(target, backend, "");
|
||||
}
|
||||
|
||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||
return system.fsMount(target, backend, rewrite);
|
||||
}
|
||||
|
||||
@@ -9,15 +9,32 @@
|
||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||
//! itself, and the kernel picks the base address.
|
||||
//!
|
||||
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
||||
//! lock and larger alignments come when user programs gain threads.
|
||||
//! 16-byte maximum alignment, exactly like the kernel heap. The free list is guarded by
|
||||
//! a `Thread.Mutex` **only in multi-threaded binaries** (`addThreadedUserBinary`): the
|
||||
//! guard is gated on `builtin.single_threaded`, so an ordinary single-threaded binary
|
||||
//! compiles it out and pays nothing, while a threaded one can allocate safely from
|
||||
//! several threads at once (docs/threading-plan.md M7). The lock lives at the two
|
||||
//! free-list mutators — `rawAlloc`/`rawFree` — which every entry point funnels through.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const system_calls = @import("system.zig");
|
||||
const Mutex = @import("thread.zig").Thread.Mutex;
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// Guards `free_list`. A no-op in single-threaded builds (compiled out); a real futex
|
||||
/// mutex in threaded ones. Uncontended acquisition is a single CAS — no syscall.
|
||||
var heap_mutex: Mutex = .{};
|
||||
|
||||
inline fn lockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.lock();
|
||||
}
|
||||
inline fn unlockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.unlock();
|
||||
}
|
||||
|
||||
/// A block header, at the start of every block; while free it also links the
|
||||
/// free list via `next`.
|
||||
const Block = extern struct {
|
||||
@@ -84,8 +101,11 @@ fn insertFree(block: *Block) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory. Holds the heap lock
|
||||
/// across the free-list search and any `grow` (which also touches the free list).
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
@@ -119,6 +139,8 @@ fn rawAlloc(len: usize) ?[*]u8 {
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
//! The per-process logger: std.log wired to the tagged kernel log ring.
|
||||
//!
|
||||
//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped
|
||||
//! logger); this backend formats the line into a fixed buffer and emits ONE
|
||||
//! `debug_write` record carrying the level. The kernel stamps the record with
|
||||
//! the sender's pid and task name (its binary path) — the process does NOT put
|
||||
//! its own name in the payload; attribution is the kernel's, structural and
|
||||
//! unforgeable. Serial shows the kernel-rendered `<path>: message` line, and
|
||||
//! the logger service demultiplexes the ring into one file per process.
|
||||
//!
|
||||
//! Installed for every user binary by the root shim (library/runtime/root.zig)
|
||||
//! via `std_options`; a program can override by declaring its own
|
||||
//! `pub const std_options`.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
fn levelOf(comptime level: std.log.Level) system.KlogLevel {
|
||||
return switch (level) {
|
||||
.err => .err,
|
||||
.warn => .warn,
|
||||
.info => .info,
|
||||
.debug => .debug,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn logFn(
|
||||
comptime level: std.log.Level,
|
||||
comptime scope: @EnumLiteral(),
|
||||
comptime format: []const u8,
|
||||
args: anytype,
|
||||
) void {
|
||||
// One record = one line = at most klog_maximum_message bytes of payload.
|
||||
// On overflow keep what fits and end with "~" so the record is still a
|
||||
// whole line (the kernel would split an embedded rest anyway).
|
||||
var buffer: [256]u8 = undefined;
|
||||
const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") ";
|
||||
const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: {
|
||||
buffer[buffer.len - 1] = '~';
|
||||
break :truncated buffer[0..];
|
||||
};
|
||||
_ = system.writeRecord(levelOf(level), line);
|
||||
}
|
||||
|
||||
/// The std.Options the root shim installs unless the program overrides it.
|
||||
/// Debug level: filtering is the log *reader's* job here — the ring is cheap,
|
||||
/// serial is a dev convenience, and the logger service keeps everything.
|
||||
pub const default_options: std.Options = .{
|
||||
.log_level = .debug,
|
||||
.logFn = logFn,
|
||||
};
|
||||
@@ -14,6 +14,12 @@ pub const main = program.main;
|
||||
/// The panic handler for every safety check in the image (runtime.start.panic).
|
||||
pub const panic = runtime.panic;
|
||||
|
||||
/// std.log for every user binary goes to the tagged kernel log ring (the kernel
|
||||
/// stamps the sender; see runtime.log). A program overrides by declaring its
|
||||
/// own `pub const std_options`.
|
||||
pub const std_options: @import("std").Options =
|
||||
if (@hasDecl(program, "std_options")) program.std_options else runtime.log.default_options;
|
||||
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
//! nothing to declare per source file.
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
pub const log = @import("log.zig");
|
||||
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||
pub const time = @import("time.zig");
|
||||
@@ -38,9 +39,9 @@ pub const device = @import("device.zig");
|
||||
pub const dma = @import("dma.zig");
|
||||
|
||||
/// Shared cacheable memory: create a region + capability, pass the capability to another
|
||||
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shm.zig
|
||||
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shared-memory.zig
|
||||
/// and docs/display-v2.md.
|
||||
pub const shm = @import("shm.zig");
|
||||
pub const shared_memory = @import("shared-memory.zig");
|
||||
|
||||
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
//! User-space shared memory: `shm_create` / `shm_map`. A process creates a shareable,
|
||||
//! User-space shared memory: `shared_memory_create` / `shared_memory_map`. A process creates a shareable,
|
||||
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||
//! `shm_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! `shared_memory_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||
//! generalization of capability passing from endpoints to memory objects.
|
||||
|
||||
@@ -22,7 +22,7 @@ pub const Region = struct {
|
||||
};
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||
/// Returns the region or null on failure. Two return values — vaddr in rax, handle in rdx —
|
||||
/// Returns the region or null on failure. Two return values — virtual_address in rax, handle in rdx —
|
||||
/// so this is a hand-written stub like `dma.alloc`.
|
||||
pub fn create(len: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
@@ -30,7 +30,7 @@ pub fn create(len: usize) ?Region {
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shm_create)),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shared_memory_create)),
|
||||
[a0] "{rdi}" (len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
@@ -41,7 +41,7 @@ pub fn create(len: usize) ?Region {
|
||||
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||
/// Returns the pointer, or null on failure.
|
||||
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
const r = sc.systemCall1(.shm_map, handle);
|
||||
const r = sc.systemCall1(.shared_memory_map, handle);
|
||||
if (failed(r)) return null;
|
||||
return @ptrFromInt(r);
|
||||
}
|
||||
@@ -51,7 +51,7 @@ pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||
/// `attach_backing`. Returns null on failure.
|
||||
pub fn physical(handle: ipc.Handle) ?usize {
|
||||
const r = sc.systemCall1(.shm_physical, handle);
|
||||
const r = sc.systemCall1(.shared_memory_physical, handle);
|
||||
if (failed(r)) return null;
|
||||
return r;
|
||||
}
|
||||
+112
-11
@@ -21,10 +21,34 @@ pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||
/// The tagged-log level of a record — re-exported so runtime.log and the logger
|
||||
/// service don't import `abi` themselves.
|
||||
pub const KlogLevel = abi.KlogLevel;
|
||||
pub const KlogStatus = abi.KlogStatus;
|
||||
pub const KlogRecordHeader = abi.KlogRecordHeader;
|
||||
pub const klog_record_header_size = abi.klog_record_header_size;
|
||||
pub const klog_record_alignment = abi.klog_record_alignment;
|
||||
pub const klog_record_magic = abi.klog_record_magic;
|
||||
pub const klog_flag_truncated = abi.klog_flag_truncated;
|
||||
pub const klog_maximum_message = abi.klog_maximum_message;
|
||||
pub const maximum_process_name = abi.maximum_process_name;
|
||||
pub const FileAttributes = abi.FileAttributes;
|
||||
pub const DirectoryEntryHeader = abi.DirectoryEntryHeader;
|
||||
pub const file_kind_regular = abi.file_kind_regular;
|
||||
pub const file_kind_directory = abi.file_kind_directory;
|
||||
|
||||
/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary
|
||||
/// output goes through std.log -> writeRecord). The kernel stamps the record
|
||||
/// with this process's id and name. Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||
return writeRecord(.raw, message);
|
||||
}
|
||||
|
||||
/// Emit one leveled record into the tagged kernel log ring. The kernel stamps
|
||||
/// pid/name/sequence/timestamp; the payload should be a single line (embedded
|
||||
/// newlines split into further records).
|
||||
pub fn writeRecord(level: KlogLevel, message: []const u8) usize {
|
||||
return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level));
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds.
|
||||
@@ -59,14 +83,91 @@ pub fn wallClock() u64 {
|
||||
return @intCast(sc.systemCall0(.wall_clock));
|
||||
}
|
||||
|
||||
/// Copy bytes out of the kernel's in-memory diagnostic log — the accumulated
|
||||
/// stream of everything `write` (and the kernel itself) has emitted — starting at
|
||||
/// `offset`, into `out`. Returns the number of bytes copied (0 at end of buffer).
|
||||
/// A program reads the whole log by looping from offset 0, advancing by the return
|
||||
/// value, until it gets 0. This is how the boot log is persisted to disk on a
|
||||
/// headless/real machine where serial output is otherwise lost.
|
||||
pub fn klogRead(offset: usize, out: []u8) usize {
|
||||
return sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
/// Copy bytes out of the tagged kernel log ring — framed records of everything
|
||||
/// every process (and the kernel) has emitted — starting at stream offset
|
||||
/// `offset`, into `out`. Returns the byte count (0 = caught up), or null when
|
||||
/// `offset` fell behind the ring's tail (those records were overwritten) or
|
||||
/// lies past its head; re-sync via `klogStatus`. A reader parses
|
||||
/// [KlogRecordHeader][name][message] frames (8-byte aligned) from the bytes.
|
||||
pub fn klogRead(offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// The log ring's live cursors (oldest retained offset, end of stream, next
|
||||
/// sequence number) plus the wall-clock time of boot — how a log reader starts,
|
||||
/// detects loss, and names a per-boot log directory.
|
||||
pub fn klogStatus() ?KlogStatus {
|
||||
var status: KlogStatus = undefined;
|
||||
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null;
|
||||
return status;
|
||||
}
|
||||
|
||||
/// Where fs_resolve routed a path: served by the kernel (a permanent node
|
||||
/// token for fs_node) or by a userspace filesystem backend (an endpoint handle
|
||||
/// plus the rewritten mount-relative path, returned in the caller's buffer).
|
||||
pub const FsRoute = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: usize, path_len: usize },
|
||||
};
|
||||
|
||||
/// Route `path` through the kernel VFS. For a backend route the rewritten
|
||||
/// mount-relative path lands in `out` (behind a kernel-written length prefix,
|
||||
/// already stripped here: out[0..path_len] is the path).
|
||||
pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)),
|
||||
[a0] "{rdi}" (@intFromPtr(path.ptr)),
|
||||
[a1] "{rsi}" (path.len),
|
||||
[a3] "{r10}" (@intFromPtr(out.ptr)),
|
||||
[a4] "{r8}" (out.len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (@as(isize, @bitCast(rax)) < 0) return null;
|
||||
if (rax == abi.fs_route_kernel) return .{ .kernel = rdx };
|
||||
if (rax != abi.fs_route_backend) return null;
|
||||
const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8);
|
||||
if (path_len + 2 > out.len) return null;
|
||||
std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]);
|
||||
return .{ .backend = .{ .handle = rdx, .path_len = path_len } };
|
||||
}
|
||||
|
||||
/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read).
|
||||
pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// A kernel-served node's metadata (fs_node status).
|
||||
pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes {
|
||||
var attributes: abi.FileAttributes = undefined;
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attributes), @sizeOf(abi.FileAttributes));
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return attributes;
|
||||
}
|
||||
|
||||
/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills
|
||||
/// `out` with [DirectoryEntryHeader][name]; returns total bytes (0 = end).
|
||||
pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// Mount a userspace filesystem's endpoint at `prefix`, with an optional
|
||||
/// backend-side `rewrite` prefix ("" = none). Possession of the endpoint
|
||||
/// handle is the capability.
|
||||
pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool {
|
||||
return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0;
|
||||
}
|
||||
|
||||
pub fn fsUnmount(prefix: []const u8) bool {
|
||||
return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0;
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
|
||||
+251
-27
@@ -11,19 +11,27 @@
|
||||
//! A binary must be built multi-threaded (`addThreadedUserBinary`) before it may spawn.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const system = @import("system.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
|
||||
/// True in a real danos binary; false when this module is compiled for host unit tests.
|
||||
/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state
|
||||
/// machines can be exercised on the host against `std.Thread.Futex` (docs/threading-plan.md
|
||||
/// M11), while the danos build uses the futex syscalls.
|
||||
const on_danos = builtin.os.tag == .freestanding;
|
||||
|
||||
/// A thread stack, if the caller does not override it. 64 KiB of mmap'd, zeroed pages.
|
||||
pub const default_stack_size: usize = 64 * 1024;
|
||||
|
||||
/// Bytes reserved at the top of each thread's stack for its per-thread TLS block (the
|
||||
/// self-pointer plus scratch slots reachable via `%fs`). docs/threading-plan.md M10.
|
||||
const tls_block_size: usize = 64;
|
||||
|
||||
pub const Thread = struct {
|
||||
/// The kernel task id of the spawned thread.
|
||||
/// The kernel task id of the spawned thread — what `join` waits on.
|
||||
tid: u32,
|
||||
/// The endpoint the kernel notifies when this thread ends — what `join` blocks on.
|
||||
exit_endpoint: ipc.Handle,
|
||||
/// The mmap'd stack, reclaimed by `join` (or at process exit after `detach`).
|
||||
stack_base: usize,
|
||||
stack_size: usize,
|
||||
@@ -46,51 +54,52 @@ pub const Thread = struct {
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread {
|
||||
const Args = @TypeOf(args);
|
||||
const Closure = struct {
|
||||
tls_base: usize,
|
||||
args: Args,
|
||||
/// Entered directly by the kernel with `self` in rdi (C ABI). Runs the user
|
||||
/// function, then ends the thread — never returns.
|
||||
/// Entered directly by the kernel with `self` in rdi (C ABI). Establishes this
|
||||
/// thread's TLS pointer, runs the user function, then ends the thread.
|
||||
fn entry(self_addr: usize) callconv(.c) noreturn {
|
||||
const self: *@This() = @ptrFromInt(self_addr);
|
||||
setThreadPointer(self.tls_base); // per-thread thread pointer before any user code
|
||||
@call(.auto, function, self.args);
|
||||
exitThread();
|
||||
}
|
||||
};
|
||||
|
||||
// The endpoint the kernel posts this thread's exit notification to.
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return error.SystemResources;
|
||||
|
||||
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(base)) return error.SystemResources;
|
||||
|
||||
// Lay the closure at the very top of the thread's own stack, then start the
|
||||
// thread's rsp just below it (16-aligned minus 8, the alignment a `call` leaves
|
||||
// for a C-ABI entry) so the growing stack never overwrites the args.
|
||||
// Top of the thread's own stack, downward: the closure, then a small per-thread TLS
|
||||
// block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is
|
||||
// scratch for user TLS), then the stack proper (rsp starts below the TLS block, so
|
||||
// the growing stack never overwrites either).
|
||||
var closure_addr = (base + config.stack_size) - @sizeOf(Closure);
|
||||
closure_addr &= ~@as(usize, @alignOf(Closure) - 1); // align the closure down
|
||||
const closure: *Closure = @ptrFromInt(closure_addr);
|
||||
closure.* = .{ .args = args };
|
||||
|
||||
var stack_top = closure_addr & ~@as(usize, 15); // 16-align below the closure
|
||||
const tls_base = (closure_addr - tls_block_size) & ~@as(usize, 15);
|
||||
const tls: [*]usize = @ptrFromInt(tls_base);
|
||||
tls[0] = tls_base; // self-pointer (fs:0), as the x86_64 TLS ABI expects
|
||||
|
||||
const closure: *Closure = @ptrFromInt(closure_addr);
|
||||
closure.* = .{ .tls_base = tls_base, .args = args };
|
||||
|
||||
var stack_top = tls_base & ~@as(usize, 15); // 16-align below the TLS block
|
||||
stack_top -= 8; // ...then rsp % 16 == 8 at the C entry
|
||||
|
||||
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr, endpoint);
|
||||
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr);
|
||||
if (threadSpawnFailed(tid)) {
|
||||
_ = system.munmap(base, config.stack_size);
|
||||
return error.SystemResources;
|
||||
}
|
||||
return .{ .tid = @intCast(tid), .exit_endpoint = endpoint, .stack_base = base, .stack_size = config.stack_size };
|
||||
return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size };
|
||||
}
|
||||
|
||||
/// Block until this thread finishes, then reclaim its stack. Mirrors
|
||||
/// `std.Thread.join`. The exit endpoint is private to this thread, so the first
|
||||
/// child-exit notification on it is this thread's.
|
||||
pub fn join(self: Thread) void {
|
||||
var receive: [0]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(self.exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == self.tid) break;
|
||||
}
|
||||
_ = system.munmap(self.stack_base, self.stack_size);
|
||||
_ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited
|
||||
_ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||
}
|
||||
|
||||
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
|
||||
@@ -119,17 +128,37 @@ pub const Thread = struct {
|
||||
/// the value already differs (safe against spurious returns, as in std): the
|
||||
/// caller re-checks its condition in a loop.
|
||||
pub fn wait(ptr: *const std.atomic.Value(u32), expect: u32) void {
|
||||
_ = futexWait(@intFromPtr(ptr), expect, 0);
|
||||
if (comptime on_danos) {
|
||||
_ = futexWait(@intFromPtr(ptr), expect, 0);
|
||||
} else {
|
||||
// Host unit-test mock: spin+yield until the value changes (`wake` is a
|
||||
// no-op — the callers re-check their condition in a loop anyway). Correct,
|
||||
// if busy; fine for the state-machine tests.
|
||||
while (ptr.load(.acquire) == expect) std.Thread.yield() catch {};
|
||||
}
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first.
|
||||
pub fn timedWait(ptr: *const std.atomic.Value(u32), expect: u32, timeout_ns: u64) error{Timeout}!void {
|
||||
if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout;
|
||||
if (comptime on_danos) {
|
||||
if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout;
|
||||
} else {
|
||||
var spins: u64 = 0;
|
||||
const limit = timeout_ns / 1000 + 1;
|
||||
while (ptr.load(.acquire) == expect) : (spins += 1) {
|
||||
if (spins >= limit) return error.Timeout;
|
||||
std.Thread.yield() catch {};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Wake up to `max_waiters` threads blocked on `ptr`.
|
||||
pub fn wake(ptr: *const std.atomic.Value(u32), max_waiters: u32) void {
|
||||
_ = futexWake(@intFromPtr(ptr), max_waiters);
|
||||
if (comptime on_danos) {
|
||||
_ = futexWake(@intFromPtr(ptr), max_waiters);
|
||||
} else {
|
||||
// host mock: spin-waiters re-check their condition, so no wake is needed.
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -230,10 +259,101 @@ pub const Thread = struct {
|
||||
s.cond.signal();
|
||||
}
|
||||
};
|
||||
|
||||
/// A reader/writer lock, `std.Thread.RwLock`-shaped: many concurrent readers OR one
|
||||
/// exclusive writer. Reader-preferring (a steady stream of readers can delay a writer),
|
||||
/// built on `Mutex` + `Condition` over a signed state: `>0` = that many readers hold
|
||||
/// it, `-1` = a writer holds it, `0` = free.
|
||||
pub const RwLock = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
state: i64 = 0,
|
||||
|
||||
/// Acquire shared (read) access, blocking while a writer holds the lock.
|
||||
pub fn lockShared(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
while (rw.state < 0) rw.cond.wait(&rw.mutex);
|
||||
rw.state += 1;
|
||||
}
|
||||
|
||||
/// Try to acquire shared access without blocking.
|
||||
pub fn tryLockShared(rw: *RwLock) bool {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
if (rw.state < 0) return false;
|
||||
rw.state += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release shared access; wake a waiting writer once the last reader leaves.
|
||||
pub fn unlockShared(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
rw.state -= 1;
|
||||
if (rw.state == 0) rw.cond.broadcast();
|
||||
}
|
||||
|
||||
/// Acquire exclusive (write) access, blocking until no readers or writer remain.
|
||||
pub fn lock(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
while (rw.state != 0) rw.cond.wait(&rw.mutex);
|
||||
rw.state = -1;
|
||||
}
|
||||
|
||||
/// Try to acquire exclusive access without blocking.
|
||||
pub fn tryLock(rw: *RwLock) bool {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
if (rw.state != 0) return false;
|
||||
rw.state = -1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release exclusive access; wake all waiters (they re-check their condition).
|
||||
pub fn unlock(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
rw.state = 0;
|
||||
rw.cond.broadcast();
|
||||
}
|
||||
};
|
||||
|
||||
/// A `std.Thread.WaitGroup`-shaped counter: `start` before spawning work, `finish` as
|
||||
/// each unit completes, `wait` blocks until the count returns to zero.
|
||||
pub const WaitGroup = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
counter: usize = 0,
|
||||
|
||||
/// Register one pending unit of work.
|
||||
pub fn start(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
wg.counter += 1;
|
||||
}
|
||||
|
||||
/// Mark one unit done; wake waiters if that was the last.
|
||||
pub fn finish(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
wg.counter -= 1;
|
||||
if (wg.counter == 0) wg.cond.broadcast();
|
||||
}
|
||||
|
||||
/// Block until every started unit has finished.
|
||||
pub fn wait(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
while (wg.counter != 0) wg.cond.wait(&wg.mutex);
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
/// thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid, or a wrapped error.
|
||||
fn threadSpawn(entry: usize, stack_top: usize, arg: usize, exit_endpoint: ipc.Handle) usize {
|
||||
fn threadSpawn(entry: usize, stack_top: usize, arg: usize) usize {
|
||||
const exit_endpoint: usize = @intCast(abi.no_cap); // join uses thread_join, not an endpoint
|
||||
return sc.systemCall4(.thread_spawn, entry, stack_top, arg, exit_endpoint);
|
||||
}
|
||||
|
||||
@@ -249,6 +369,11 @@ fn exitThread() noreturn {
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Set the calling thread's FS base (its user TLS thread pointer).
|
||||
fn setThreadPointer(addr: usize) void {
|
||||
_ = sc.systemCall1(.set_thread_pointer, addr);
|
||||
}
|
||||
|
||||
/// futex_wait(addr, expect, timeout_ns) -> status (abi.futex_*).
|
||||
fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize {
|
||||
return sc.systemCall3(.futex_wait, addr, expect, timeout_ns);
|
||||
@@ -258,3 +383,102 @@ fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize {
|
||||
fn futexWake(addr: usize, count: u32) usize {
|
||||
return sc.systemCall2(.futex_wake, addr, count);
|
||||
}
|
||||
|
||||
// --- host unit tests (docs/threading-plan.md M11) ---------------------------
|
||||
//
|
||||
// These run under `zig build test` on the host: the `Futex` seam above uses
|
||||
// `std.Thread.Futex` off-danos, so the lock/condvar state machines can be exercised by
|
||||
// real host threads. They are never compiled into a danos binary (test blocks only build
|
||||
// under test), so their `std.Thread` use is fine even though `std.Thread` is unavailable
|
||||
// on the freestanding target.
|
||||
|
||||
test "Mutex serialises concurrent increments across host threads" {
|
||||
var m: Thread.Mutex = .{};
|
||||
var counter: u64 = 0;
|
||||
const workers = 8;
|
||||
const per = 20_000;
|
||||
const Ctx = struct {
|
||||
m: *Thread.Mutex,
|
||||
c: *u64,
|
||||
fn run(ctx: @This()) void {
|
||||
var i: usize = 0;
|
||||
while (i < per) : (i += 1) {
|
||||
ctx.m.lock();
|
||||
ctx.c.* += 1;
|
||||
ctx.m.unlock();
|
||||
}
|
||||
}
|
||||
};
|
||||
var handles: [workers]std.Thread = undefined;
|
||||
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .m = &m, .c = &counter }});
|
||||
for (handles) |h| h.join();
|
||||
try std.testing.expectEqual(@as(u64, workers * per), counter);
|
||||
}
|
||||
|
||||
test "RwLock never lets a reader observe a half-written pair" {
|
||||
var rw: Thread.RwLock = .{};
|
||||
var a: u64 = 0;
|
||||
var b: u64 = 0; // invariant while a lock is held: a == b
|
||||
var stop = std.atomic.Value(bool).init(false);
|
||||
var ok = std.atomic.Value(bool).init(true);
|
||||
|
||||
const Writer = struct {
|
||||
rw: *Thread.RwLock,
|
||||
a: *u64,
|
||||
b: *u64,
|
||||
stop: *std.atomic.Value(bool),
|
||||
fn run(w: @This()) void {
|
||||
var v: u64 = 1;
|
||||
while (!w.stop.load(.acquire)) : (v +%= 1) {
|
||||
w.rw.lock();
|
||||
w.a.* = v; // update both halves under the exclusive lock...
|
||||
w.b.* = v;
|
||||
w.rw.unlock();
|
||||
}
|
||||
}
|
||||
};
|
||||
const Reader = struct {
|
||||
rw: *Thread.RwLock,
|
||||
a: *u64,
|
||||
b: *u64,
|
||||
ok: *std.atomic.Value(bool),
|
||||
fn run(r: @This()) void {
|
||||
var i: usize = 0;
|
||||
while (i < 200_000) : (i += 1) {
|
||||
r.rw.lockShared();
|
||||
if (r.a.* != r.b.*) r.ok.store(false, .release); // ...so a reader must never see them differ
|
||||
r.rw.unlockShared();
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
var writers: [2]std.Thread = undefined;
|
||||
for (&writers) |*w| w.* = try std.Thread.spawn(.{}, Writer.run, .{Writer{ .rw = &rw, .a = &a, .b = &b, .stop = &stop }});
|
||||
var readers: [4]std.Thread = undefined;
|
||||
for (&readers) |*rd| rd.* = try std.Thread.spawn(.{}, Reader.run, .{Reader{ .rw = &rw, .a = &a, .b = &b, .ok = &ok }});
|
||||
for (readers) |rd| rd.join();
|
||||
stop.store(true, .release);
|
||||
for (writers) |w| w.join();
|
||||
try std.testing.expect(ok.load(.acquire));
|
||||
}
|
||||
|
||||
test "WaitGroup blocks until every started unit finishes" {
|
||||
var wg: Thread.WaitGroup = .{};
|
||||
var done = std.atomic.Value(u32).init(0);
|
||||
const n = 6;
|
||||
const Ctx = struct {
|
||||
wg: *Thread.WaitGroup,
|
||||
done: *std.atomic.Value(u32),
|
||||
fn run(c: @This()) void {
|
||||
_ = c.done.fetchAdd(1, .monotonic);
|
||||
c.wg.finish();
|
||||
}
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < n) : (i += 1) wg.start();
|
||||
var handles: [n]std.Thread = undefined;
|
||||
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .wg = &wg, .done = &done }});
|
||||
wg.wait(); // must not return until all n finished
|
||||
try std.testing.expectEqual(@as(u32, n), done.load(.acquire));
|
||||
for (handles) |h| h.join();
|
||||
}
|
||||
|
||||
+102
-9
@@ -39,13 +39,13 @@ pub const SystemCall = enum(u64) {
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> virtual_address: map a claimed device's MMIO into this address space
|
||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
||||
dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory
|
||||
dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc
|
||||
dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory
|
||||
dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc
|
||||
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||
io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource
|
||||
io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource
|
||||
@@ -58,17 +58,24 @@ pub const SystemCall = enum(u64) {
|
||||
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
||||
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy tagged log-ring stream bytes from `offset` out to a user buffer; fails once `offset` falls behind the ring's tail (re-sync via klog_status)
|
||||
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
||||
shm_create = 34, // shm_create(len) -> vaddr (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md)
|
||||
shm_map = 35, // shm_map(cap) -> vaddr: map the shared region named by a received capability into this AS (the same physical pages the creator sees)
|
||||
shm_physical = 36, // shm_physical(cap) -> paddr: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md)
|
||||
shared_memory_create = 34, // shared_memory_create(len) -> virtual_address (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md)
|
||||
shared_memory_map = 35, // shared_memory_map(cap) -> virtual_address: map the shared region named by a received capability into this address space (the same physical pages the creator sees)
|
||||
shared_memory_physical = 36, // shared_memory_physical(cap) -> physical_address: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md)
|
||||
thread_spawn = 37, // thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid: start a task sharing the caller's address space at `entry` on `stack_top`, `arg` in rdi; exit_endpoint (a handle, or no_cap) is notified when it ends — how join waits (docs/threading.md)
|
||||
thread_exit = 38, // thread_exit(): end the calling thread, dropping one reference to its address space (destroyed on the last)
|
||||
current_core = 39, // current_core() -> index: the dense 0-based index of the core the caller is running on (for parallelism/affinity introspection)
|
||||
futex_wait = 40, // futex_wait(addr, expected, timeout_ns) -> status: if *addr == expected, block until woken or the timeout; returns futex_woken/mismatch/timed_out (docs/threading.md)
|
||||
futex_wake = 41, // futex_wake(addr, count) -> woken: wake up to `count` tasks blocked in futex_wait on `addr` in this address space
|
||||
thread_self = 42, // thread_self() -> tid: the calling thread's kernel task id (runtime.Thread.getCurrentId)
|
||||
thread_join = 43, // thread_join(tid) -> 0: block until the thread with id `tid` has exited (runtime.Thread.join; no per-thread IPC endpoint) (docs/threading.md)
|
||||
set_thread_pointer = 44, // set_thread_pointer(addr) -> 0: set the caller's thread pointer (user-space TLS base; x86_64 IA32_FS_BASE, aarch64 TPIDR_EL0); restored per task across context switches (docs/threading-plan.md M10)
|
||||
klog_status = 45, // klog_status(ptr) -> 0: copy a KlogStatus (ring cursors + the boot wall-clock anchor) out to a user buffer
|
||||
fs_resolve = 46, // fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap) -> route tag (rax: fs_route_*) + node token or backend handle (rdx); a backend resolve writes the rewritten mount-relative path into out (length in r8 via third result)
|
||||
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
|
||||
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
|
||||
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
|
||||
_,
|
||||
};
|
||||
|
||||
@@ -185,11 +192,97 @@ pub const ProcessDescriptor = extern struct {
|
||||
name: [maximum_process_name]u8, // argv[0] at spawn; empty for kernel tasks
|
||||
};
|
||||
|
||||
// --- the tagged kernel log ring (klog) ---------------------------------------
|
||||
// Every `debug_write` becomes one RECORD per payload line, stamped by the kernel
|
||||
// with the sender's pid, task name (its binary path), level, a per-boot sequence
|
||||
// number, and a monotonic timestamp. `klog_read` copies raw stream bytes — a
|
||||
// reader parses [KlogRecordHeader][name][message] frames, each padded to
|
||||
// `klog_record_alignment`. Sequence gaps tell a reader exactly how many records
|
||||
// the ring overwrote while it wasn't looking.
|
||||
|
||||
/// Log level of a klog record — std.log's levels plus `raw` (untagged bytes:
|
||||
/// kernel prints and legacy runtime.system.write output).
|
||||
pub const KlogLevel = enum(u8) { err = 0, warn = 1, info = 2, debug = 3, raw = 4 };
|
||||
|
||||
/// "RK" — leads every record; a parser's resync/corruption guard.
|
||||
pub const klog_record_magic: u16 = 0x4B52;
|
||||
|
||||
/// KlogRecordHeader.flags bit: the emitter truncated the payload to fit.
|
||||
pub const klog_flag_truncated: u8 = 1;
|
||||
|
||||
/// Header of one ring record, followed by `name_len` bytes of task name and
|
||||
/// `message_len` bytes of payload; the whole record is padded to 8 bytes.
|
||||
pub const KlogRecordHeader = extern struct {
|
||||
magic: u16, // klog_record_magic
|
||||
level: KlogLevel,
|
||||
name_len: u8, // 0..maximum_process_name
|
||||
pid: u32, // sender process id; 0 = the kernel itself
|
||||
sequence: u64, // per-boot monotonic record number (gaps = lost records)
|
||||
timestamp_ns: u64, // monotonic ns since boot (the `clock` timebase)
|
||||
message_len: u16, // payload bytes (excludes the record's trailing pad)
|
||||
flags: u8, // klog_flag_* bits
|
||||
_reserved: [5]u8,
|
||||
};
|
||||
|
||||
pub const klog_record_header_size: usize = 32; // @sizeOf(KlogRecordHeader), pinned by a test
|
||||
pub const klog_record_alignment: usize = 8;
|
||||
/// Per-record payload cap (one line; longer emitter lines are truncated).
|
||||
pub const klog_maximum_message: usize = 256;
|
||||
|
||||
// --- the kernel VFS root (resolve + redirect) --------------------------------
|
||||
// fs_resolve routes a path through the kernel mount table. Kernel-backed mounts
|
||||
// (the initrd at /system) resolve to a permanent node TOKEN served by fs_node;
|
||||
// userspace mounts resolve to the backend's endpoint handle (installed in the
|
||||
// caller's table, deduplicated) plus the rewritten mount-relative path — the
|
||||
// caller then speaks the vfs-protocol to the backend directly. The kernel never
|
||||
// blocks on a userspace filesystem.
|
||||
|
||||
/// fs_resolve result tags (rax).
|
||||
pub const fs_route_kernel: u64 = 0; // rdx = node token; serve via fs_node
|
||||
pub const fs_route_backend: u64 = 1; // rdx = endpoint handle; speak vfs-protocol
|
||||
|
||||
/// fs_node operations — the same numbers as the vfs-protocol Operation enum, so
|
||||
/// client code shares one vocabulary.
|
||||
pub const fs_node_read: u64 = 2;
|
||||
pub const fs_node_status: u64 = 4;
|
||||
pub const fs_node_readdir: u64 = 5;
|
||||
|
||||
/// fs_resolve flags (same values as the vfs-protocol open flags).
|
||||
pub const fs_flag_create: u64 = 1;
|
||||
|
||||
/// FileStatus-shaped node metadata (matches the vfs-protocol payload layout).
|
||||
pub const file_kind_regular: u32 = 0;
|
||||
pub const file_kind_directory: u32 = 1;
|
||||
pub const FileAttributes = extern struct {
|
||||
size: u64,
|
||||
kind: u32,
|
||||
_pad: u32 = 0,
|
||||
mtime: u64 = 0,
|
||||
};
|
||||
|
||||
/// One fs_node readdir result: the header, followed by `name_len` name bytes in
|
||||
/// the caller's buffer (matches the vfs-protocol DirectoryEntry layout).
|
||||
pub const DirectoryEntryHeader = extern struct {
|
||||
kind: u32,
|
||||
name_len: u32,
|
||||
size: u64,
|
||||
};
|
||||
|
||||
/// The klog_status copy-out: the ring's live cursors plus the wall-clock time
|
||||
/// of boot — the anchor a log persister names its per-boot directory with and
|
||||
/// combines with record timestamps for wall-clock line stamps.
|
||||
pub const KlogStatus = extern struct {
|
||||
tail: u64, // oldest retained stream offset — always a record boundary
|
||||
head: u64, // next byte to be written (end of stream)
|
||||
next_sequence: u64, // the sequence the next record will get
|
||||
boot_unix_seconds: u64, // wall-clock time of boot (RTC anchor)
|
||||
};
|
||||
|
||||
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
|
||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||
pub const ServiceId = enum(u32) {
|
||||
vfs = 1,
|
||||
vfs = 1, // RETIRED: the router moved into the kernel (fs_resolve); the slot stays reserved
|
||||
input = 2,
|
||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||
@@ -198,7 +291,7 @@ pub const ServiceId = enum(u32) {
|
||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
||||
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
||||
shm_test = 10, // the shm test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
||||
_,
|
||||
};
|
||||
|
||||
+10
-9
@@ -37,6 +37,11 @@ pub const Framebuffer = extern struct {
|
||||
height: u32, // visible rows (e.g. 1080)
|
||||
pitch: u32, // bytes from the start of one row to the start of the next
|
||||
format: PixelFormat,
|
||||
/// The panel's refresh rate in Hz, computed from its EDID preferred timing (pixel
|
||||
/// clock / total pixels per frame) while GOP was still alive — the one moment it is
|
||||
/// readable (docs/gop.md). 0 = unknown (no EDID). The display service paces its
|
||||
/// frame clock by it; without vblank this fixes the *rate*, never the *phase*.
|
||||
refresh_hz: u32 = 0,
|
||||
|
||||
/// Whether a usable framebuffer was handed over.
|
||||
pub fn present(self: Framebuffer) bool {
|
||||
@@ -142,15 +147,11 @@ pub const BootInformation = extern struct {
|
||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||
acpi_rsdp: u64 = 0,
|
||||
/// The raw `/system/services/init` ELF image, read off the boot volume by the loader
|
||||
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||
/// kernel boots without user space. Grows into a full initial_ramdisk handoff later.
|
||||
init_base: u64 = 0,
|
||||
init_len: u64 = 0,
|
||||
/// The initial_ramdisk image (a bundle of extra user binaries — the VFS server and
|
||||
/// device drivers), read off the boot volume into memory that survives the
|
||||
/// handoff, same as `init` above. 0/0 = no initial_ramdisk. See system/initial-ramdisk.zig.
|
||||
/// The initial_ramdisk image: every user binary from the boot volume's /system
|
||||
/// tree (init included), packed by the loader into memory that survives the
|
||||
/// handoff (classified reserved, so the kernel identity-maps it and never
|
||||
/// allocates over it). Entries are named by full FHS path. 0/0 = no binaries
|
||||
/// found — the kernel boots without user space. See system/initial-ramdisk.zig.
|
||||
initial_ramdisk_base: u64 = 0,
|
||||
initial_ramdisk_len: u64 = 0,
|
||||
};
|
||||
|
||||
@@ -97,6 +97,7 @@ pub const DisplayInfo = extern struct {
|
||||
height: u32 = 0, // visible rows
|
||||
pitch: u32 = 0, // bytes from one row's start to the next
|
||||
format: u32 = 0, // a DisplayFormat value
|
||||
refresh_hz: u32 = 0, // panel refresh rate from EDID (0 = unknown); see boot-handoff
|
||||
};
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||
|
||||
@@ -16,11 +16,6 @@ const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Log a discovered function with its (class / subclass / prog-IF) triple decoded
|
||||
/// to human names — the boot-log breadcrumb that says *what* the hardware is, so
|
||||
/// "class 0x01 (Mass Storage Controller) subclass 0x06 (Serial ATA Controller)
|
||||
@@ -74,7 +69,7 @@ fn configWrite16(bus: u64, dev: u64, function: u64, offset: u64, value: u16) voi
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(bridge_id)) {
|
||||
writeLine("/system/drivers/pci-bus: unable to claim bridge device {d}\n", .{bridge_id});
|
||||
std.log.info("unable to claim bridge device {d}", .{bridge_id});
|
||||
return false;
|
||||
}
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
@@ -85,7 +80,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == bridge_id) break d;
|
||||
} else {
|
||||
writeLine("/system/drivers/pci-bus: device {d} not in the device tree\n", .{bridge_id});
|
||||
std.log.info("device {d} not in the device tree", .{bridge_id});
|
||||
return false;
|
||||
};
|
||||
// Resource 0 is the ECAM window (1 MiB of config space per bus); the bus
|
||||
@@ -159,7 +154,7 @@ fn scan() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
writeLine("/system/drivers/pci-bus: {d} functions found\n", .{found});
|
||||
std.log.info("{d} functions found", .{found});
|
||||
}
|
||||
|
||||
/// Register one function under the bridge and report it to the manager. The
|
||||
@@ -229,7 +224,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
}
|
||||
|
||||
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||
writeLine("/system/drivers/pci-bus: register refused for {d}:{d}.{d}\n", .{ bus, dev, function });
|
||||
std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function });
|
||||
return;
|
||||
};
|
||||
const report = protocol.ChildAdded{
|
||||
@@ -240,7 +235,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("/system/drivers/pci-bus: child report for {d}:{d}.{d} failed\n", .{ bus, dev, function });
|
||||
std.log.info("child report for {d}:{d}.{d} failed", .{ bus, dev, function });
|
||||
};
|
||||
}
|
||||
|
||||
@@ -255,7 +250,7 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent
|
||||
bridge_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/pci-bus: malformed bridge device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed bridge device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
|
||||
@@ -23,11 +23,6 @@ const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
const protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
|
||||
/// registering) is still coming up.
|
||||
fn lookupBus() ?ipc.Handle {
|
||||
@@ -75,14 +70,14 @@ pub fn main(init: runtime.process.Init) void {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: no HID argument\n");
|
||||
return;
|
||||
}
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: starting for hid {s}\n", .{hid});
|
||||
std.log.info("starting for hid {s}", .{hid});
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/keyboard: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (device.findDeviceDescriptorByHid(buffer, hid) == null) {
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: no device for hid {s}\n", .{hid});
|
||||
std.log.info("no device for hid {s}", .{hid});
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -90,7 +85,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
// absent (as today) it defaults to us.
|
||||
const layout_name = init.arguments.get(2) orelse "us";
|
||||
const layout = xkb.byName(layout_name) orelse xkb.us;
|
||||
writeLine("/system/drivers/ps2-bus/keyboard: layout {s}\n", .{layout.name});
|
||||
std.log.info("layout {s}", .{layout.name});
|
||||
|
||||
// Attach to the bus: hand it our endpoint, and it forwards every byte the
|
||||
// keyboard sends (it owns the controller; we own the decoding).
|
||||
|
||||
@@ -22,11 +22,6 @@ const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
const protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
|
||||
/// registering) is still coming up.
|
||||
fn lookupBus() ?ipc.Handle {
|
||||
@@ -54,14 +49,14 @@ pub fn main(init: runtime.process.Init) void {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: no HID argument\n");
|
||||
return;
|
||||
}
|
||||
writeLine("/system/drivers/ps2-bus/mouse: starting for hid {s}\n", .{hid});
|
||||
std.log.info("starting for hid {s}", .{hid});
|
||||
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (ps2.findMouseDescriptor(buffer) == null) {
|
||||
writeLine("/system/drivers/ps2-bus/mouse: no device for hid {s}\n", .{hid});
|
||||
std.log.info("no device for hid {s}", .{hid});
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
@@ -16,13 +16,6 @@ const ps2 = @import("ps2-library.zig");
|
||||
const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so output
|
||||
/// from the child drivers (which run concurrently) can never interleave with it.
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// Ask the device on `port` what it is, then spawn the matching driver from the
|
||||
/// initial-ramdisk, handing it the device's HID as argv[1]. The driver is chosen
|
||||
/// from what the device reports, not from the port number. Returns the identified
|
||||
@@ -30,19 +23,19 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
/// attaches, or null if nothing was spawned.
|
||||
fn spawnIdentifiedDriver(controller: ps2.Controller, port: ps2.Port) ?ps2.DeviceType {
|
||||
const device_type = controller.identifyDevice(port) orelse {
|
||||
writeLine("/system/drivers/ps2-bus: identify timed out on port {s}\n", .{@tagName(port)});
|
||||
std.log.info("identify timed out on port {s}", .{@tagName(port)});
|
||||
return null;
|
||||
};
|
||||
const driver_name = device_type.driverName() orelse {
|
||||
writeLine("/system/drivers/ps2-bus: unrecognized device on port {s}\n", .{@tagName(port)});
|
||||
std.log.info("unrecognized device on port {s}", .{@tagName(port)});
|
||||
return null;
|
||||
};
|
||||
const hid = device_type.hid() orelse "";
|
||||
if (runtime.system.spawnWithArguments(driver_name, &.{hid}) != null) {
|
||||
writeLine("/system/drivers/ps2-bus: port {s} is a {s}, spawned {s}\n", .{ @tagName(port), hid, driver_name });
|
||||
std.log.info("port {s} is a {s}, spawned {s}", .{ @tagName(port), hid, driver_name });
|
||||
return device_type;
|
||||
}
|
||||
writeLine("/system/drivers/ps2-bus: failed to spawn {s}\n", .{driver_name});
|
||||
std.log.info("failed to spawn {s}", .{driver_name});
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -83,7 +76,7 @@ fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
const device_type = maybe_type orelse continue;
|
||||
if (@intFromEnum(device_type) != request.device_type) continue;
|
||||
port_endpoints[port_index] = endpoint;
|
||||
writeLine("/system/drivers/ps2-bus: {s} driver attached\n", .{@tagName(device_type)});
|
||||
std.log.info("{s} driver attached", .{@tagName(device_type)});
|
||||
return reply.write(out, .ok);
|
||||
}
|
||||
return reply.write(out, .no_such_device);
|
||||
|
||||
@@ -23,11 +23,6 @@ const ipc = runtime.ipc;
|
||||
const process = runtime.process;
|
||||
const input_protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The modifier state a character lookup needs — derived from the report's
|
||||
// modifier byte, plus the driver-tracked caps-lock toggle.
|
||||
const ModifierSnapshot = struct {
|
||||
@@ -71,7 +66,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
const device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-hid/keyboard: malformed device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
const layout = xkb.byName(init.arguments.get(2) orelse "us") orelse xkb.us;
|
||||
@@ -82,7 +77,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
}
|
||||
var device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-hid/keyboard: could not open device {d}\n", .{device_id});
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return;
|
||||
};
|
||||
const endpoint = device.findEndpoint(runtime.usb.transfer_type_interrupt, true) orelse {
|
||||
@@ -104,7 +99,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
_ = process.bindSignals(device.endpoint);
|
||||
writeLine("/system/drivers/usb-hid/keyboard: ok (device {d}, interface {d}, layout {s})\n", .{ device_id, device.interface_number, layout.name });
|
||||
std.log.info("ok (device {d}, interface {d}, layout {s})", .{ device_id, device.interface_number, layout.name });
|
||||
|
||||
var decoder = hid.KeyboardDecoder{};
|
||||
var caps_lock = false;
|
||||
|
||||
@@ -18,11 +18,6 @@ const ipc = runtime.ipc;
|
||||
const process = runtime.process;
|
||||
const input_protocol = runtime.input_protocol;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The current pressed-button bitmask in input-protocol terms.
|
||||
fn buttonMask(buttons: u8) u32 {
|
||||
var mask: u32 = 0;
|
||||
@@ -38,7 +33,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
const device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-hid/mouse: malformed device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -47,7 +42,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
}
|
||||
var device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-hid/mouse: could not open device {d}\n", .{device_id});
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return;
|
||||
};
|
||||
const endpoint = device.findEndpoint(runtime.usb.transfer_type_interrupt, true) orelse {
|
||||
@@ -67,7 +62,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
_ = process.bindSignals(device.endpoint);
|
||||
writeLine("/system/drivers/usb-hid/mouse: ok (device {d}, interface {d})\n", .{ device_id, device.interface_number });
|
||||
std.log.info("ok (device {d}, interface {d})", .{ device_id, device.interface_number });
|
||||
|
||||
var previous_buttons: u8 = 0;
|
||||
var receive: [64]u8 = undefined;
|
||||
|
||||
@@ -18,11 +18,6 @@ const bot = @import("bulk-only-transport.zig");
|
||||
const block_protocol = @import("block-protocol");
|
||||
const dma = runtime.dma;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
var device_id: u64 = 0;
|
||||
var device: runtime.usb.Device = undefined;
|
||||
var bulk_in: runtime.usb.Endpoint = undefined;
|
||||
@@ -73,7 +68,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
return false;
|
||||
}
|
||||
device = runtime.usb.open(device_id) orelse {
|
||||
writeLine("/system/drivers/usb-storage: could not open device {d}\n", .{device_id});
|
||||
std.log.info("could not open device {d}", .{device_id});
|
||||
return false;
|
||||
};
|
||||
bulk_in = device.findEndpoint(runtime.usb.transfer_type_bulk, true) orelse {
|
||||
@@ -112,14 +107,14 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
const capacity = scsi.parseCapacity(capacity_bytes);
|
||||
block_size = capacity.block_size;
|
||||
block_count = @as(u64, capacity.last_lba) + 1;
|
||||
writeLine("/system/drivers/usb-storage: ready ({d} blocks x {d} bytes)\n", .{ block_count, block_size });
|
||||
std.log.info("ready ({d} blocks x {d} bytes)", .{ block_count, block_size });
|
||||
|
||||
// Self-check: read block 0 and log its trailing signature (0x55AA for a boot
|
||||
// sector) — proof READ(10) works end to end over the bulk path.
|
||||
const read0 = scsi.read10(0, 1);
|
||||
if (block_size <= 4096 and transact(&read0, true, command_data.physical, block_size)) {
|
||||
const sector: [*]const u8 = @ptrFromInt(command_data.virtual);
|
||||
writeLine("/system/drivers/usb-storage: block 0 signature 0x{x:0>2}{x:0>2}\n", .{ sector[510], sector[511] });
|
||||
std.log.info("block 0 signature 0x{x:0>2}{x:0>2}", .{ sector[510], sector[511] });
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -171,7 +166,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-storage: malformed device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(block_protocol.message_maximum, .{
|
||||
|
||||
@@ -64,13 +64,6 @@ fn reportEndpointFor(device_token: u64) ?usize {
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so
|
||||
/// concurrent instances (one per controller) can never interleave mid-line.
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
var controller_id: u64 = protocol.no_device;
|
||||
|
||||
/// Claim the assigned controller, find its register window, and hello the
|
||||
@@ -79,7 +72,7 @@ var controller_id: u64 = protocol.no_device;
|
||||
fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
if (!device.claim(controller_id)) {
|
||||
writeLine("/system/drivers/usb-xhci-bus: unable to claim controller device {d}\n", .{controller_id});
|
||||
std.log.info("unable to claim controller device {d}", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -92,7 +85,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
writeLine("/system/drivers/usb-xhci-bus: device {d} not in the device tree\n", .{controller_id});
|
||||
std.log.info("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
@@ -105,10 +98,10 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
break resource;
|
||||
}
|
||||
} else {
|
||||
writeLine("/system/drivers/usb-xhci-bus: controller device {d} has no register BAR\n", .{controller_id});
|
||||
std.log.info("controller device {d} has no register BAR", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: claimed controller device {d} (registers at 0x{x}, {d} bytes)\n", .{
|
||||
std.log.info("claimed controller device {d} (registers at 0x{x}, {d} bytes)", .{
|
||||
controller_id,
|
||||
register_window.start,
|
||||
register_window.len,
|
||||
@@ -124,7 +117,7 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: controller reset/bring-up failed\n");
|
||||
return false;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: controller running ({d} slots, {d}-byte contexts)\n", .{
|
||||
std.log.info("controller running ({d} slots, {d}-byte contexts)", .{
|
||||
controller.?.max_slots,
|
||||
controller.?.context_size,
|
||||
});
|
||||
@@ -198,7 +191,7 @@ fn scanPorts(manager: runtime.ipc.Handle) void {
|
||||
_ = runtime.system.write("/system/drivers/usb-xhci-bus: controller not initialised\n");
|
||||
return;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: {d} root-hub ports\n", .{engine.max_ports});
|
||||
std.log.info("{d} root-hub ports", .{engine.max_ports});
|
||||
|
||||
var port: u32 = 1;
|
||||
var connected: u32 = 0;
|
||||
@@ -207,17 +200,17 @@ fn scanPorts(manager: runtime.ipc.Handle) void {
|
||||
if (port_status & 1 == 0) continue; // CCS: nothing connected
|
||||
connected += 1;
|
||||
const speed = (port_status >> 10) & 0xF; // the PORTSC port-speed class
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} connected — {s} (speed class {d})\n", .{ port, speedName(speed), speed });
|
||||
std.log.info("port {d} connected — {s} (speed class {d})", .{ port, speedName(speed), speed });
|
||||
|
||||
const usb_device = engine.setupDevice(port, speed) orelse {
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} device setup failed\n", .{port});
|
||||
std.log.info("port {d} device setup failed", .{port});
|
||||
continue;
|
||||
};
|
||||
if (!engine.enumerate(usb_device)) {
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} enumeration failed\n", .{port});
|
||||
std.log.info("port {d} enumeration failed", .{port});
|
||||
continue;
|
||||
}
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} device vendor 0x{x:0>4} product 0x{x:0>4}, {d} interface(s)\n", .{
|
||||
std.log.info("port {d} device vendor 0x{x:0>4} product 0x{x:0>4}, {d} interface(s)", .{
|
||||
port,
|
||||
usb_device.device_descriptor.vendor_id,
|
||||
usb_device.device_descriptor.product_id,
|
||||
@@ -259,7 +252,7 @@ fn reportInterface(manager: runtime.ipc.Handle, port: u32, interface: library.In
|
||||
descriptor.hid_len = hid_text.len;
|
||||
@memcpy(descriptor.hid[0..hid_text.len], hid_text);
|
||||
const registered = device.register(controller_id, &descriptor) orelse {
|
||||
writeLine("/system/drivers/usb-xhci-bus: register refused for port {d} interface {d}\n", .{ port, interface.number });
|
||||
std.log.info("register refused for port {d} interface {d}", .{ port, interface.number });
|
||||
return null;
|
||||
};
|
||||
|
||||
@@ -271,10 +264,10 @@ fn reportInterface(manager: runtime.ipc.Handle, port: u32, interface: library.In
|
||||
};
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(manager, std.mem.asBytes(&report), &reply) catch {
|
||||
writeLine("/system/drivers/usb-xhci-bus: child report for port {d} interface {d} failed\n", .{ port, interface.number });
|
||||
std.log.info("child report for port {d} interface {d} failed", .{ port, interface.number });
|
||||
return null;
|
||||
};
|
||||
writeLine("/system/drivers/usb-xhci-bus: port {d} interface {d} class {d}/{d}/{d} registered as device {d}\n", .{
|
||||
std.log.info("port {d} interface {d} class {d}/{d}/{d} registered as device {d}", .{
|
||||
port,
|
||||
interface.number,
|
||||
interface.class,
|
||||
@@ -407,7 +400,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
writeLine("/system/drivers/usb-xhci-bus: malformed controller device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed controller device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(transfer.message_maximum, .{
|
||||
|
||||
@@ -18,7 +18,7 @@ const runtime = @import("runtime");
|
||||
const mmio = @import("mmio");
|
||||
const device = runtime.device;
|
||||
const dma = runtime.dma;
|
||||
const shm = runtime.shm;
|
||||
const shared_memory = runtime.shared_memory;
|
||||
const system = runtime.system;
|
||||
const ipc = runtime.ipc;
|
||||
const dp = runtime.display_protocol;
|
||||
@@ -53,13 +53,18 @@ const offered_modes = [_]Mode{ .{ .width = 640, .height = 480 }, .{ .width = 800
|
||||
var current_width: u32 = offered_modes[0].width;
|
||||
var current_height: u32 = offered_modes[0].height;
|
||||
|
||||
/// Monotonic fence id for fenced (vsync) flushes; the device signals the fence when the flush
|
||||
/// is complete, which its used-ring ack already gates our synchronous present on.
|
||||
/// Monotonic fence id for fenced flushes; the device signals the fence when the flush is
|
||||
/// complete, which its used-ring ack already gates our synchronous present on. Completion
|
||||
/// feedback, not vblank — nothing here is paced to the display's refresh.
|
||||
var fence_next: u64 = 1;
|
||||
|
||||
/// Whether the device offered VIRTIO_GPU_F_EDID, so `get_edid` is worth issuing.
|
||||
var edid_available = false;
|
||||
|
||||
/// The panel refresh rate parsed from the EDID preferred timing (0 = unknown). Carried to
|
||||
/// the compositor in the announce so its frame clock paces to the panel, not a guess.
|
||||
var edid_refresh_hz: u32 = 0;
|
||||
|
||||
/// The control virtqueue. We drive it synchronously — one command, notify, poll the used
|
||||
/// ring — so a depth of 16 is ample; we ask the device to shrink to it (virtio 1.0 lets the
|
||||
/// driver reduce queue_size), keeping the whole ring inside one page.
|
||||
@@ -88,23 +93,16 @@ var bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 };
|
||||
var ring: dma.Region = undefined;
|
||||
var command: dma.Region = undefined;
|
||||
|
||||
// The scanout backing is a **shared** (shm) region, not DMA: cacheable so the compositor
|
||||
// The scanout backing is a **shared** (shared-memory) region, not DMA: cacheable so the compositor
|
||||
// composites into it cheaply (x86 DMA is coherent, so the device still sees the writes), and
|
||||
// shareable so the same physical pages the device scans out of are the ones the compositor
|
||||
// paints. The driver keeps the capability to hand to the compositor in the announce.
|
||||
var surface: shm.Region = undefined;
|
||||
var surface: shared_memory.Region = undefined;
|
||||
|
||||
// Split-virtqueue producer/consumer shadows.
|
||||
var avail_shadow: u16 = 0;
|
||||
var used_shadow: u16 = 0;
|
||||
|
||||
/// Format one whole log line and emit it in a single `write`, so this driver's output can
|
||||
/// never interleave mid-line with the other drivers the manager runs concurrently.
|
||||
fn log(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [160]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// --- common-config register access (little-endian MMIO at `common_base`) ---------------
|
||||
|
||||
fn cfgRead(comptime T: type, comptime field: []const u8) T {
|
||||
@@ -149,7 +147,7 @@ fn mapBar(config: usize, descriptor: *const device.DeviceDescriptor, bar: u8) ?u
|
||||
return v;
|
||||
}
|
||||
}
|
||||
log("virtio-gpu: BAR {d} (physical 0x{x}) is not a mapped resource\n", .{ bar, base });
|
||||
std.log.info("BAR {d} (physical 0x{x}) is not a mapped resource", .{ bar, base });
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -157,7 +155,7 @@ fn mapBar(config: usize, descriptor: *const device.DeviceDescriptor, bar: u8) ?u
|
||||
/// notify structures (the only two V3 needs). Returns false if either is missing.
|
||||
fn walkCapabilities(config: usize, descriptor: *const device.DeviceDescriptor) bool {
|
||||
if (mmio.read(u16, config + 0x06) & 0x10 == 0) { // Status bit 4: capabilities list present
|
||||
log("virtio-gpu: device has no PCI capability list\n", .{});
|
||||
std.log.info("device has no PCI capability list", .{});
|
||||
return false;
|
||||
}
|
||||
var cap: u8 = @as(u8, @truncate(mmio.read(u8, config + 0x34))) & 0xFC;
|
||||
@@ -188,7 +186,7 @@ fn walkCapabilities(config: usize, descriptor: *const device.DeviceDescriptor) b
|
||||
cap = next;
|
||||
}
|
||||
if (common_base == 0 or notify_base == 0) {
|
||||
log("virtio-gpu: missing common-config or notify capability\n", .{});
|
||||
std.log.info("missing common-config or notify capability", .{});
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
@@ -270,7 +268,7 @@ fn testPixel(index: u32) u32 {
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(device_id)) {
|
||||
log("virtio-gpu: unable to claim device {d}\n", .{device_id});
|
||||
std.log.info("unable to claim device {d}", .{device_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -279,7 +277,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
const descriptor = for (descriptors[0..@min(total, descriptors.len)]) |*d| {
|
||||
if (d.id == device_id) break d;
|
||||
} else {
|
||||
log("virtio-gpu: device {d} not in the device tree\n", .{device_id});
|
||||
std.log.info("device {d} not in the device tree", .{device_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
@@ -287,13 +285,13 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
// decode + bus mastering (the device DMAs the ring and backing out of RAM); pci-bus only
|
||||
// preserves whatever the firmware left, and a secondary display is often left disabled.
|
||||
const config = device.mmioMap(device_id, 0) orelse {
|
||||
log("virtio-gpu: config-space map failed\n", .{});
|
||||
std.log.info("config-space map failed", .{});
|
||||
return false;
|
||||
};
|
||||
const vendor = mmio.read(u16, config + 0x00);
|
||||
const dev = mmio.read(u16, config + 0x02);
|
||||
if (vendor != virtio_vendor or dev != virtio_gpu_device) {
|
||||
log("virtio-gpu: not a virtio-gpu (vendor 0x{x} device 0x{x})\n", .{ vendor, dev });
|
||||
std.log.info("not a virtio-gpu (vendor 0x{x} device 0x{x})", .{ vendor, dev });
|
||||
return false;
|
||||
}
|
||||
mmio.write(u16, config + 0x04, mmio.read(u16, config + 0x04) | 0x06); // MEM + bus master
|
||||
@@ -312,7 +310,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
// High feature word: VERSION_1 (bit 32) is required for a modern device.
|
||||
cfgWrite(u32, "device_feature_select", vp.feature_version_1_word);
|
||||
if (cfgRead(u32, "device_feature") & vp.feature_version_1_bit == 0) {
|
||||
log("virtio-gpu: device does not offer VERSION_1 (not a modern device)\n", .{});
|
||||
std.log.info("device does not offer VERSION_1 (not a modern device)", .{});
|
||||
return false;
|
||||
}
|
||||
// Accept exactly VERSION_1, plus EDID when the device offered it (never a feature it didn't).
|
||||
@@ -322,7 +320,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
cfgWrite(u32, "driver_feature", vp.feature_version_1_bit);
|
||||
orStatus(vp.status_features_ok);
|
||||
if (cfgRead(u8, "device_status") & vp.status_features_ok == 0) {
|
||||
log("virtio-gpu: device rejected the negotiated features\n", .{});
|
||||
std.log.info("device rejected the negotiated features", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -330,15 +328,15 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
cfgWrite(u16, "queue_select", 0);
|
||||
const device_qsize = cfgRead(u16, "queue_size");
|
||||
if (device_qsize < queue_size) {
|
||||
log("virtio-gpu: control queue too small ({d})\n", .{device_qsize});
|
||||
std.log.info("control queue too small ({d})", .{device_qsize});
|
||||
return false;
|
||||
}
|
||||
ring = dma.alloc(4096, dma.coherent) orelse {
|
||||
log("virtio-gpu: virtqueue allocation failed\n", .{});
|
||||
std.log.info("virtqueue allocation failed", .{});
|
||||
return false;
|
||||
};
|
||||
command = dma.alloc(4096, dma.coherent) orelse {
|
||||
log("virtio-gpu: command-buffer allocation failed\n", .{});
|
||||
std.log.info("command-buffer allocation failed", .{});
|
||||
return false;
|
||||
};
|
||||
mmio.write(u16, ring.virtual + avail_offset, 1); // VIRTQ_AVAIL_F_NO_INTERRUPT: we poll
|
||||
@@ -366,19 +364,19 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
.height = max_height,
|
||||
};
|
||||
if (command_nodata(@sizeOf(vg.ResourceCreate2d)) != ok_nodata) {
|
||||
log("virtio-gpu: resource_create_2d failed\n", .{});
|
||||
std.log.info("resource_create_2d failed", .{});
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Back the resource with a shared (shm) surface, so the compositor and the device work
|
||||
// Back the resource with a shared (shared-memory) surface, so the compositor and the device work
|
||||
// the same physical pages. The device needs the guest-physical base for attach_backing.
|
||||
surface = shm.create(scanout_bytes) orelse {
|
||||
log("virtio-gpu: scanout surface allocation failed\n", .{});
|
||||
surface = shared_memory.create(scanout_bytes) orelse {
|
||||
std.log.info("scanout surface allocation failed", .{});
|
||||
return false;
|
||||
};
|
||||
const surface_physical = shm.physical(surface.handle) orelse {
|
||||
log("virtio-gpu: could not resolve the scanout surface physical address\n", .{});
|
||||
const surface_physical = shared_memory.physical(surface.handle) orelse {
|
||||
std.log.info("could not resolve the scanout surface physical address", .{});
|
||||
return false;
|
||||
};
|
||||
{
|
||||
@@ -391,15 +389,15 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
const entry: *vg.MemEntry = @ptrFromInt(command.virtual + request_offset + @sizeOf(vg.ResourceAttachBacking));
|
||||
entry.* = .{ .addr = surface_physical, .length = @intCast(scanout_bytes) };
|
||||
if (command_nodata(@sizeOf(vg.ResourceAttachBacking) + @sizeOf(vg.MemEntry)) != ok_nodata) {
|
||||
log("virtio-gpu: resource_attach_backing failed\n", .{});
|
||||
std.log.info("resource_attach_backing failed", .{});
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!setScanoutRect()) {
|
||||
log("virtio-gpu: set_scanout failed\n", .{});
|
||||
std.log.info("set_scanout failed", .{});
|
||||
return false;
|
||||
}
|
||||
log("virtio-gpu: scanout {d}x{d} online\n", .{ current_width, current_height });
|
||||
std.log.info("scanout {d}x{d} online", .{ current_width, current_height });
|
||||
|
||||
// Hello the device manager so it counts us as up (and does not stop us at the hello
|
||||
// deadline). A restarted instance re-hellos here and re-announces below — the compositor
|
||||
@@ -417,17 +415,17 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
for (0..pixel_count) |i| pixels[i] = testPixel(@intCast(i));
|
||||
|
||||
if (!presentFull()) {
|
||||
log("virtio-gpu: initial present failed\n", .{});
|
||||
std.log.info("initial present failed", .{});
|
||||
return false;
|
||||
}
|
||||
// The scanout surface is CPU-visible RAM: read the pattern back to prove the mapping,
|
||||
// which together with the flush ack above is the automated stand-in for "it's on screen".
|
||||
mmio.rmb();
|
||||
if (pixels[0] != testPixel(0) or pixels[pixel_count / 2] != testPixel(@intCast(pixel_count / 2))) {
|
||||
log("virtio-gpu: pixel read-back mismatch\n", .{});
|
||||
std.log.info("pixel read-back mismatch", .{});
|
||||
return false;
|
||||
}
|
||||
log("virtio-gpu: flush acked, pixel check ok\n", .{});
|
||||
std.log.info("flush acked, pixel check ok", .{});
|
||||
|
||||
// Offer the shared surface to the compositor so it upgrades off the GOP floor (V4).
|
||||
announce();
|
||||
@@ -451,26 +449,34 @@ fn setScanoutRect() bool {
|
||||
/// a device that doesn't offer EDID, or a missing/short block, is logged and ignored.
|
||||
fn readEdid() void {
|
||||
if (!edid_available) {
|
||||
log("virtio-gpu: EDID not offered by device\n", .{});
|
||||
std.log.info("EDID not offered by device", .{});
|
||||
return;
|
||||
}
|
||||
const request = requestAt(vg.GetEdid);
|
||||
request.* = .{ .hdr = .{ .type = @intFromEnum(vg.CmdType.get_edid) }, .scanout = 0 };
|
||||
if (!submit(@sizeOf(vg.GetEdid), @sizeOf(vg.RespEdid))) {
|
||||
log("virtio-gpu: EDID request not acked\n", .{});
|
||||
std.log.info("EDID request not acked", .{});
|
||||
return;
|
||||
}
|
||||
const response: *vg.RespEdid = @ptrFromInt(command.virtual + response_offset);
|
||||
if (response.hdr.type != @intFromEnum(vg.CmdType.resp_ok_edid) or response.size < 64) {
|
||||
log("virtio-gpu: EDID unavailable\n", .{});
|
||||
std.log.info("EDID unavailable", .{});
|
||||
return;
|
||||
}
|
||||
// The first detailed timing descriptor (EDID base-block offset 54) is the preferred mode:
|
||||
// active pixels are 12-bit, low byte + high nibble (bytes 2/4 horizontal, 5/7 vertical).
|
||||
// The refresh rate is derived from the same descriptor: pixel clock (bytes 0-1, 10 kHz
|
||||
// units) over total (active + blanking) pixels per frame — the loader does the identical
|
||||
// computation for the boot framebuffer (boot/efi.zig edidNative).
|
||||
const e = &response.edid;
|
||||
const h_active = @as(u32, e[56]) | (@as(u32, e[58] & 0xF0) << 4);
|
||||
const v_active = @as(u32, e[59]) | (@as(u32, e[61] & 0xF0) << 4);
|
||||
log("virtio-gpu: EDID preferred mode {d}x{d}\n", .{ h_active, v_active });
|
||||
const clock_hz = (@as(u64, e[54]) | (@as(u64, e[55]) << 8)) * 10_000;
|
||||
const h_blank = @as(u64, e[57]) | (@as(u64, e[58] & 0x0F) << 8);
|
||||
const v_blank = @as(u64, e[60]) | (@as(u64, e[61] & 0x0F) << 8);
|
||||
const total = (@as(u64, h_active) + h_blank) * (@as(u64, v_active) + v_blank);
|
||||
if (total != 0) edid_refresh_hz = @intCast((clock_hz + total / 2) / total);
|
||||
std.log.info("EDID preferred mode {d}x{d} @ {d} Hz", .{ h_active, v_active, edid_refresh_hz });
|
||||
}
|
||||
|
||||
/// Present the whole surface: copy the guest backing into the host resource, then flush it to
|
||||
@@ -492,8 +498,9 @@ fn presentFull() bool {
|
||||
if (command_nodata(@sizeOf(vg.TransferToHost2d)) != ok_nodata) return false;
|
||||
}
|
||||
{
|
||||
// A fenced flush (vsync): the device signals the fence when the frame is actually on
|
||||
// screen — which its used-ring ack, what our synchronous submit waits on, already gates.
|
||||
// A fenced flush: the device signals the fence once it has consumed the frame — which
|
||||
// its used-ring ack, what our synchronous submit waits on, already gates. Completion
|
||||
// feedback and a tear-free snapshot, not vblank pacing.
|
||||
const request = requestAt(vg.ResourceFlush);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_flush), .flags = vg.flag_fence, .fence_id = fence_next },
|
||||
@@ -515,20 +522,20 @@ fn helloManager() void {
|
||||
if (ipc.lookup(.device_manager)) |h| break h;
|
||||
system.sleep(20);
|
||||
} else {
|
||||
log("virtio-gpu: no device manager to hello\n", .{});
|
||||
std.log.info("no device manager to hello", .{});
|
||||
return;
|
||||
};
|
||||
const hello = dm.Hello{ .role = @intFromEnum(dm.Role.bus), .device_id = device_id };
|
||||
var reply: [dm.reply_size]u8 = undefined;
|
||||
const n = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch {
|
||||
log("virtio-gpu: hello call failed\n", .{});
|
||||
std.log.info("hello call failed", .{});
|
||||
return;
|
||||
};
|
||||
if (n < dm.reply_size or std.mem.bytesToValue(dm.HelloReply, reply[0..dm.reply_size]).status != 0) {
|
||||
log("virtio-gpu: hello refused\n", .{});
|
||||
std.log.info("hello refused", .{});
|
||||
return;
|
||||
}
|
||||
log("virtio-gpu: hello acknowledged\n", .{});
|
||||
std.log.info("hello acknowledged", .{});
|
||||
}
|
||||
|
||||
/// Announce the scanout to the display service so it upgrades off the GOP framebuffer: hand it
|
||||
@@ -542,22 +549,23 @@ fn announce() void {
|
||||
if (ipc.lookup(.display)) |h| break h;
|
||||
system.sleep(20);
|
||||
} else {
|
||||
log("virtio-gpu: no display service to announce to (scanout-only)\n", .{});
|
||||
std.log.info("no display service to announce to (scanout-only)", .{});
|
||||
return;
|
||||
};
|
||||
var request = dp.Request{
|
||||
.operation = @intFromEnum(dp.Operation.attach_scanout),
|
||||
.x = max_width, // the shared surface's row stride in pixels (it is sized to the max mode)
|
||||
.y = edid_refresh_hz, // the panel refresh from EDID (0 = unknown) — the frame-clock seed
|
||||
.width = current_width,
|
||||
.height = current_height,
|
||||
.colour = display_format_bgrx,
|
||||
};
|
||||
var reply: [dp.reply_size]u8 = undefined;
|
||||
_ = ipc.callCap(display, std.mem.asBytes(&request), &reply, surface.handle) catch {
|
||||
log("virtio-gpu: announce to display failed\n", .{});
|
||||
std.log.info("announce to display failed", .{});
|
||||
return;
|
||||
};
|
||||
log("virtio-gpu: announced scanout to display\n", .{});
|
||||
std.log.info("announced scanout to display", .{});
|
||||
}
|
||||
|
||||
/// A `sp.Reply{status}` written into `reply`.
|
||||
@@ -606,7 +614,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
};
|
||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
log("virtio-gpu: malformed device id '{s}'\n", .{argument});
|
||||
std.log.info("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(256, .{
|
||||
|
||||
@@ -1,7 +1,13 @@
|
||||
//! The initial_ramdisk (initial ramdisk) container format — shared by the build-time
|
||||
//! packer (tools/make-initial-ramdisk.py) and the kernel that unpacks it. Deliberately
|
||||
//! trivial: a header, a table of fixed-size entries, then the concatenated file
|
||||
//! blobs. We own both producer and consumer, so it need be no fancier.
|
||||
//! The initial_ramdisk (initial ramdisk) container format — built in RAM by the
|
||||
//! bootloader (boot/efi.zig walks the boot volume's /system tree) and unpacked by
|
||||
//! the kernel. Deliberately trivial: a header, a table of fixed-size entries, then
|
||||
//! the concatenated file blobs. We own both producer and consumer, so it need be
|
||||
//! no fancier.
|
||||
//!
|
||||
//! v2: entry names are full FHS paths ("/system/services/init"), 64 bytes — the
|
||||
//! same limit as a task name (abi.maximum_process_name), so a path-named task is
|
||||
//! never truncated. The boot volume's file tree is the single source of truth;
|
||||
//! this image is only the loader→kernel handoff snapshot of it.
|
||||
//!
|
||||
//! Layout:
|
||||
//! Header (magic, count)
|
||||
@@ -10,8 +16,14 @@
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// "DNRD" — identifies a danos initial_ramdisk image.
|
||||
pub const magic: u32 = 0x444E5244;
|
||||
/// "DNR2" — identifies a danos initial_ramdisk image, format v2 (path names).
|
||||
/// The v1 magic ("DNRD", basename entries) is rejected: a stale image should
|
||||
/// fail loudly at Reader.init, not misparse names.
|
||||
pub const magic: u32 = 0x32524E44;
|
||||
|
||||
/// Entry name capacity. Matches abi.maximum_process_name so a spawned task can
|
||||
/// always carry its full binary path as its name.
|
||||
pub const maximum_name = 64;
|
||||
|
||||
pub const Header = extern struct {
|
||||
magic: u32,
|
||||
@@ -19,11 +31,17 @@ pub const Header = extern struct {
|
||||
};
|
||||
|
||||
pub const Entry = extern struct {
|
||||
name: [32]u8, // NUL-padded file name (basename)
|
||||
name: [maximum_name]u8, // NUL-padded FHS path, e.g. "/system/services/init"
|
||||
offset: u64, // byte offset of the blob within the image
|
||||
len: u64, // blob length in bytes
|
||||
};
|
||||
|
||||
/// The basename of a path: the final component after the last '/'.
|
||||
pub fn basename(path: []const u8) []const u8 {
|
||||
const i = std.mem.lastIndexOfScalar(u8, path, '/') orelse return path;
|
||||
return path[i + 1 ..];
|
||||
}
|
||||
|
||||
/// A validated view over an initial_ramdisk image. `init` checks the magic and that the
|
||||
/// entry table fits; `entry` bounds-checks each blob against the image.
|
||||
pub const Reader = struct {
|
||||
@@ -48,11 +66,74 @@ pub const Reader = struct {
|
||||
if (e.offset > self.image.len or e.len > self.image.len - e.offset) return null;
|
||||
// The name is stored in the entry's fixed field; return a stable slice
|
||||
// into the image (not the value copy) up to the NUL terminator.
|
||||
const name_field = self.image[off .. off + 32];
|
||||
const name_field = self.image[off .. off + maximum_name];
|
||||
const nlen = std.mem.indexOfScalar(u8, name_field, 0) orelse name_field.len;
|
||||
return .{
|
||||
.name = name_field[0..nlen],
|
||||
.blob = self.image[@intCast(e.offset)..][0..@intCast(e.len)],
|
||||
};
|
||||
}
|
||||
|
||||
/// Look a binary up by name: an exact path match wins; otherwise a unique
|
||||
/// basename match ("fat" finds "/system/services/fat") keeps pre-path callers
|
||||
/// working. The returned Item's name is always the stored full path.
|
||||
pub fn find(self: Reader, name: []const u8) ?Item {
|
||||
var i: u32 = 0;
|
||||
while (i < self.count) : (i += 1) {
|
||||
const item = self.entry(i) orelse continue;
|
||||
if (std.mem.eql(u8, item.name, name)) return item;
|
||||
}
|
||||
i = 0;
|
||||
while (i < self.count) : (i += 1) {
|
||||
const item = self.entry(i) orelse continue;
|
||||
if (std.mem.eql(u8, basename(item.name), name)) return item;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
// --- tests (host) -----------------------------------------------------------
|
||||
|
||||
fn testImage(buffer: []u8, entries: []const struct { name: []const u8, blob: []const u8 }) []const u8 {
|
||||
const table_end = @sizeOf(Header) + entries.len * @sizeOf(Entry);
|
||||
var offset: usize = table_end;
|
||||
std.mem.bytesAsValue(Header, buffer[0..@sizeOf(Header)]).* = .{ .magic = magic, .count = @intCast(entries.len) };
|
||||
for (entries, 0..) |e, i| {
|
||||
var record = Entry{ .name = @splat(0), .offset = offset, .len = e.blob.len };
|
||||
@memcpy(record.name[0..e.name.len], e.name);
|
||||
std.mem.bytesAsValue(Entry, buffer[@sizeOf(Header) + i * @sizeOf(Entry) ..][0..@sizeOf(Entry)]).* = record;
|
||||
@memcpy(buffer[offset..][0..e.blob.len], e.blob);
|
||||
offset += e.blob.len;
|
||||
}
|
||||
return buffer[0..offset];
|
||||
}
|
||||
|
||||
test "find matches exact path, then unique basename; name is the stored path" {
|
||||
var buffer: [1024]u8 = undefined;
|
||||
const image = testImage(&buffer, &.{
|
||||
.{ .name = "/system/services/init", .blob = "INIT" },
|
||||
.{ .name = "/system/drivers/ps2-bus", .blob = "PS2" },
|
||||
});
|
||||
const rd = Reader.init(image).?;
|
||||
|
||||
const by_path = rd.find("/system/services/init").?;
|
||||
try std.testing.expectEqualStrings("/system/services/init", by_path.name);
|
||||
try std.testing.expectEqualStrings("INIT", by_path.blob);
|
||||
|
||||
const by_base = rd.find("ps2-bus").?;
|
||||
try std.testing.expectEqualStrings("/system/drivers/ps2-bus", by_base.name);
|
||||
try std.testing.expectEqualStrings("PS2", by_base.blob);
|
||||
|
||||
try std.testing.expect(rd.find("no-such-binary") == null);
|
||||
}
|
||||
|
||||
test "v1 magic is rejected" {
|
||||
var buffer: [64]u8 = @splat(0);
|
||||
std.mem.bytesAsValue(Header, buffer[0..@sizeOf(Header)]).* = .{ .magic = 0x444E5244, .count = 0 };
|
||||
try std.testing.expect(Reader.init(&buffer) == null);
|
||||
}
|
||||
|
||||
test "basename" {
|
||||
try std.testing.expectEqualStrings("fat", basename("/system/services/fat"));
|
||||
try std.testing.expectEqualStrings("fat", basename("fat"));
|
||||
}
|
||||
|
||||
@@ -189,8 +189,8 @@ pub fn mapUserDmaInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
}
|
||||
|
||||
/// Map shared cacheable RAM into address space `root`: write-back cacheable, RW+NX, and
|
||||
/// marked so teardown won't free the frames (they're owned by a refcounted shm object,
|
||||
/// freed when its last capability drops). For shm_create/shm_map.
|
||||
/// marked so teardown won't free the frames (they're owned by a refcounted shared-memory object,
|
||||
/// freed when its last capability drops). For shared_memory_create/shared_memory_map.
|
||||
pub fn mapUserSharedInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
paging.mapUserSharedInto(root, virtual, physical, len);
|
||||
}
|
||||
@@ -304,6 +304,17 @@ pub fn cpuLocal() usize {
|
||||
return pcpu.scheduler();
|
||||
}
|
||||
|
||||
const ia32_fs_base = 0xC000_0100;
|
||||
|
||||
/// Set the user-space TLS **thread pointer** — the arch-neutral name the generic scheduler
|
||||
/// calls (`architecture.setThreadPointer`). On x86_64 that is the FS-segment base
|
||||
/// (`IA32_FS_BASE`); an aarch64 port implements the same call against `TPIDR_EL0`. The
|
||||
/// kernel never touches FS, so this only affects the user task that runs next, which the
|
||||
/// scheduler restores per task across context switches (docs/threading-plan.md M10).
|
||||
pub fn setThreadPointer(base: u64) void {
|
||||
io.wrmsr(ia32_fs_base, base);
|
||||
}
|
||||
|
||||
// --- SMP: application-processor bring-up ----------------------------------
|
||||
|
||||
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
|
||||
|
||||
@@ -509,7 +509,7 @@ pub fn unmapInto(pml4: u64, virtual: u64) void {
|
||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
||||
/// resolving the offset within it. The foundation for cross-address-space copies
|
||||
/// and for munmap (which needs the frame behind a user vaddr to free it).
|
||||
/// and for munmap (which needs the frame behind a user virtual_address to free it).
|
||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return null;
|
||||
|
||||
@@ -61,7 +61,7 @@ pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||
/// [[boot-handoff]], not the device tree), so it is seeded explicitly, after `init`.
|
||||
/// Returns the new device id, or null when there is no framebuffer (headless) or the
|
||||
/// table is full. Idempotent-ish: only ever call once per boot.
|
||||
pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32) ?u64 {
|
||||
pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32) ?u64 {
|
||||
if (base == 0 or width == 0 or height == 0) return null; // headless
|
||||
if (count >= maximum_devices) {
|
||||
dropped += 1;
|
||||
@@ -79,7 +79,7 @@ pub fn seedDisplay(base: u64, width: u32, height: u32, pitch: u32, format: u32)
|
||||
.len = @as(u64, height) * pitch,
|
||||
.flags = device_abi.resource_flag_write_combining,
|
||||
};
|
||||
d.display = .{ .width = width, .height = height, .pitch = pitch, .format = format };
|
||||
d.display = .{ .width = width, .height = height, .pitch = pitch, .format = format, .refresh_hz = refresh_hz };
|
||||
devices[count] = d;
|
||||
display_device = d.id;
|
||||
count += 1;
|
||||
|
||||
@@ -153,35 +153,35 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
||||
/// The `kind` tag on a `scheduler.HandleObject` — which capability object a handle names.
|
||||
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
|
||||
pub const handle_kind_endpoint: u8 = 0;
|
||||
pub const handle_kind_shm: u8 = 1;
|
||||
pub const handle_kind_shared_memory: u8 = 1;
|
||||
|
||||
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
|
||||
/// capability handles across processes and freed when the last one drops. `phys` is its
|
||||
/// contiguous physical base, `pages` its length. A sharer's address-space teardown never
|
||||
/// reclaims these frames (the mapping carries `device_grant`); this object owns them.
|
||||
pub const ShmObject = struct {
|
||||
pub const SharedMemoryObject = struct {
|
||||
refcount: u32 = 1,
|
||||
phys: u64,
|
||||
pages: usize,
|
||||
};
|
||||
|
||||
/// Wrap `pages` contiguous frames at `phys` (already allocated + zeroed by the caller) in a
|
||||
/// refcounted shm object, or null if the heap is out of room.
|
||||
pub fn createShm(phys: u64, pages: usize) ?*ShmObject {
|
||||
const shm = heap.allocator().create(ShmObject) catch return null;
|
||||
shm.* = .{ .phys = phys, .pages = pages };
|
||||
return shm;
|
||||
/// refcounted shared-memory object, or null if the heap is out of room.
|
||||
pub fn createSharedMemory(phys: u64, pages: usize) ?*SharedMemoryObject {
|
||||
const shared_memory = heap.allocator().create(SharedMemoryObject) catch return null;
|
||||
shared_memory.* = .{ .phys = phys, .pages = pages };
|
||||
return shared_memory;
|
||||
}
|
||||
|
||||
/// Drop a shared-memory reference; when the last one goes, return its frames to the
|
||||
/// allocator and free the object. (The mappings themselves are torn down with each
|
||||
/// sharer's address space; `device_grant` keeps that from freeing the frames early.)
|
||||
pub fn dropShmRef(shm: *ShmObject) void {
|
||||
if (shm.refcount > 1) {
|
||||
shm.refcount -= 1;
|
||||
pub fn dropSharedMemoryReference(shared_memory: *SharedMemoryObject) void {
|
||||
if (shared_memory.refcount > 1) {
|
||||
shared_memory.refcount -= 1;
|
||||
} else {
|
||||
for (0..shm.pages) |i| pmm.free(shm.phys + i * page_size);
|
||||
heap.allocator().destroy(shm);
|
||||
for (0..shared_memory.pages) |i| pmm.free(shared_memory.phys + i * page_size);
|
||||
heap.allocator().destroy(shared_memory);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -294,8 +294,8 @@ fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
||||
const e: *Endpoint = @ptrCast(@alignCast(entry.ptr));
|
||||
e.refcount += 1;
|
||||
},
|
||||
handle_kind_shm => {
|
||||
const s: *ShmObject = @ptrCast(@alignCast(entry.ptr));
|
||||
handle_kind_shared_memory => {
|
||||
const s: *SharedMemoryObject = @ptrCast(@alignCast(entry.ptr));
|
||||
s.refcount += 1;
|
||||
},
|
||||
else => return -EBADF,
|
||||
@@ -355,7 +355,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
||||
me.ipc_client = null;
|
||||
const n = @min(reply_len, client.ipc_reply_cap);
|
||||
client.ipc_received_cap = abi.no_cap;
|
||||
if (!copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
|
||||
if (!copyAcross(me.address_space, reply_ptr, client.address_space, client.ipc_reply_ptr, n)) {
|
||||
client.ipc_status = -EFAULT;
|
||||
} else if (send_cap != abi.no_cap) {
|
||||
// Transfer the reply's capability into the client. A failure fails the
|
||||
@@ -383,8 +383,8 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
||||
}
|
||||
if (popPost(endpoint)) |slot| {
|
||||
const n = @min(@as(usize, slot.length), receive_cap);
|
||||
// Copy from the kernel-resident ring slot (source aspace 0) into the receiver.
|
||||
if (!copyAcross(0, @intFromPtr(&slot.bytes), me.aspace, receive_ptr, n)) {
|
||||
// Copy from the kernel-resident ring slot (source address_space 0) into the receiver.
|
||||
if (!copyAcross(0, @intFromPtr(&slot.bytes), me.address_space, receive_ptr, n)) {
|
||||
continue; // bad receive buffer: drop this message, keep serving
|
||||
}
|
||||
out_badge.* = slot.sender_id | notify_badge_bit | notify_message_bit;
|
||||
@@ -392,7 +392,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
||||
}
|
||||
if (dequeueSender(endpoint)) |caller| {
|
||||
const n = @min(caller.ipc_send_len, receive_cap);
|
||||
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) {
|
||||
if (!copyAcross(caller.address_space, caller.ipc_send_ptr, me.address_space, receive_ptr, n)) {
|
||||
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
||||
scheduler.readyLocked(caller);
|
||||
continue;
|
||||
@@ -512,12 +512,26 @@ pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||
}
|
||||
|
||||
/// Install a shared-memory handle.
|
||||
pub fn installShmHandle(t: *Task, shm: *ShmObject) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_shm, .ptr = @ptrCast(shm) });
|
||||
pub fn installSharedMemoryHandle(t: *Task, shared_memory: *SharedMemoryObject) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_shared_memory, .ptr = @ptrCast(shared_memory) });
|
||||
}
|
||||
|
||||
/// Install an endpoint handle, reusing an existing slot that already names this
|
||||
/// endpoint (no new reference taken in that case). For callers that install per
|
||||
/// operation — fs_resolve — so a 16-slot table can't be exhausted by repeats.
|
||||
/// Any subsystem installing handles per-call should come through here.
|
||||
pub fn installHandleDeduped(t: *Task, endpoint: *Endpoint) i64 {
|
||||
for (t.handles, 0..) |slot, i| {
|
||||
const entry = slot orelse continue;
|
||||
if (entry.kind == handle_kind_endpoint and entry.ptr == @as(*anyopaque, @ptrCast(endpoint))) return @intCast(i);
|
||||
}
|
||||
const h = installHandle(t, endpoint);
|
||||
if (h >= 0) endpoint.refcount += 1; // the table entry owns a reference
|
||||
return h;
|
||||
}
|
||||
|
||||
/// Resolve a handle to its endpoint, or null if out of range, unused, or a different kind
|
||||
/// (e.g. an shm handle used where an endpoint is expected).
|
||||
/// (e.g. a shared-memory handle used where an endpoint is expected).
|
||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
@@ -526,11 +540,11 @@ pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
}
|
||||
|
||||
/// Resolve a handle to its shared-memory object, or null if out of range, unused, or not
|
||||
/// an shm handle.
|
||||
pub fn resolveShm(t: *Task, h: u64) ?*ShmObject {
|
||||
/// a shared-memory handle.
|
||||
pub fn resolveSharedMemory(t: *Task, h: u64) ?*SharedMemoryObject {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
if (entry.kind != handle_kind_shm) return null;
|
||||
if (entry.kind != handle_kind_shared_memory) return null;
|
||||
return @ptrCast(@alignCast(entry.ptr));
|
||||
}
|
||||
|
||||
@@ -550,7 +564,7 @@ pub fn closeHandles(t: *Task) void {
|
||||
fn dropEntry(entry: scheduler.HandleObject) void {
|
||||
switch (entry.kind) {
|
||||
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_shm => dropShmRef(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_shared_memory => dropSharedMemoryReference(@ptrCast(@alignCast(entry.ptr))),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
+14
-18
@@ -75,7 +75,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// Retain the whole stream in a RAM buffer too, so a user program can later
|
||||
// read it back (klog_read) and persist the boot log to disk — the only way to
|
||||
// see it on a headless/real machine with no host capturing serial.
|
||||
log.addSink(log.ramSink);
|
||||
// (Retention is the tagged ring inside log.zig — not a sink.)
|
||||
|
||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||
@@ -200,8 +200,8 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// Publish the loader's framebuffer as a claimable `display` device, so a
|
||||
// user-space display service can take it over the same claim + mmio_map path as
|
||||
// any other hardware (it is not firmware-discovered; it rides the boot handoff).
|
||||
if (devices_broker.seedDisplay(fb.base, fb.width, fb.height, fb.pitch, @intFromEnum(fb.format))) |display_id| {
|
||||
log.print("/system/kernel: framebuffer device {d} seeded ({d}x{d}, pitch {d}, write-combining)\n", .{ display_id, fb.width, fb.height, fb.pitch });
|
||||
if (devices_broker.seedDisplay(fb.base, fb.width, fb.height, fb.pitch, @intFromEnum(fb.format), fb.refresh_hz)) |display_id| {
|
||||
log.print("/system/kernel: framebuffer device {d} seeded ({d}x{d}, pitch {d}, {d} Hz, write-combining)\n", .{ display_id, fb.width, fb.height, fb.pitch, fb.refresh_hz });
|
||||
}
|
||||
|
||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||
@@ -329,20 +329,16 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// service supervisor and the device manager spawns the drivers it discovers.
|
||||
publishInitialRamdisk(boot_information);
|
||||
|
||||
// Hand over to user space: load /system/services/init (read off the boot volume by
|
||||
// the loader) and spawn it as a real ring-3 process, PID 1. As the supervisor it
|
||||
// brings up the system services (the VFS server, the device manager); the device
|
||||
// manager then discovers the hardware and spawns each driver. init runs on its own
|
||||
// address space, preemptively — this boot context becomes the BSP's idle loop.
|
||||
if (boot_information.init_len != 0) {
|
||||
status("/system/kernel: starting /system/services/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
process.spawnProcess(image, 4, &.{"/system/services/init"}) catch |err| {
|
||||
statusPrint("/system/kernel: /system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
} else {
|
||||
status("no /system/services/init on the boot volume.\n");
|
||||
}
|
||||
// Hand over to user space: spawn /system/services/init out of the ramdisk as a
|
||||
// real ring-3 process, PID 1 — it rides the same table as every other binary.
|
||||
// As the supervisor it brings up the system services (the VFS server, the device
|
||||
// manager); the device manager then discovers the hardware and spawns each
|
||||
// driver. init runs on its own address space, preemptively — this boot context
|
||||
// becomes the BSP's idle loop.
|
||||
status("/system/kernel: starting /system/services/init...\n");
|
||||
process.spawnBundled("/system/services/init") catch |err| {
|
||||
statusPrint("/system/kernel: /system/services/init failed to start: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
|
||||
// Become the idle task: drop below every real task and halt until an
|
||||
// interrupt. The timer keeps preempting into init and any other work.
|
||||
@@ -430,7 +426,7 @@ fn status(message: []const u8) void {
|
||||
/// any display service holding the framebuffer. The console is otherwise silent in normal
|
||||
/// operation (see `status`); it exists now only for early-boot and fatal output.
|
||||
fn fatal(message: []const u8) void {
|
||||
log.write(message);
|
||||
log.appendPanic(message); // bounded lock wait: a panic never deadlocks on the log
|
||||
console.setSuppressed(false);
|
||||
console.write(message);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,234 @@
|
||||
//! The tagged kernel log ring — a circular byte buffer of framed records, each
|
||||
//! stamped by the writer (the kernel) with the sender's pid, task name, level,
|
||||
//! per-boot sequence number, and monotonic timestamp. Pure code over an
|
||||
//! embedded buffer — no architecture or lock imports — so it host-tests
|
||||
//! alongside the other pure kernel pieces (`zig build test`).
|
||||
//!
|
||||
//! `head` and `tail` are free-running u64 positions in a logical byte stream;
|
||||
//! the physical wrap is invisible to readers (all copies are modulo the
|
||||
//! buffer), so a record never splits logically and no padding records exist.
|
||||
//! Reclaim happens record by record: the writer parses the header at `tail`
|
||||
//! (which it wrote itself) and advances until the new record fits — `tail`
|
||||
//! always sits on a record boundary, and sequence-number gaps tell a reader
|
||||
//! exactly how many records it lost.
|
||||
//!
|
||||
//! Locking is the caller's job (log.zig holds its log lock around every call);
|
||||
//! the ring itself is single-writer, snapshot-reader.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
|
||||
pub fn Ring(comptime capacity: usize) type {
|
||||
comptime std.debug.assert(std.math.isPowerOfTwo(capacity));
|
||||
return struct {
|
||||
const Self = @This();
|
||||
|
||||
buffer: [capacity]u8 = undefined,
|
||||
head: u64 = 0,
|
||||
tail: u64 = 0,
|
||||
next_sequence: u64 = 0,
|
||||
|
||||
/// Append one record; returns its sequence number. `name` and `message`
|
||||
/// are clamped to their ABI caps (the syscall clamps earlier too — the
|
||||
/// clamp here makes the ring safe in isolation).
|
||||
pub fn append(
|
||||
self: *Self,
|
||||
pid: u32,
|
||||
name: []const u8,
|
||||
level: abi.KlogLevel,
|
||||
timestamp_ns: u64,
|
||||
message: []const u8,
|
||||
truncated: bool,
|
||||
) u64 {
|
||||
const name_len: usize = @min(name.len, abi.maximum_process_name);
|
||||
const message_len: usize = @min(message.len, abi.klog_maximum_message);
|
||||
const record_len = recordLength(name_len, message_len);
|
||||
|
||||
// Reclaim whole records until the new one fits.
|
||||
while (self.head + record_len - self.tail > capacity) self.reclaimOne();
|
||||
|
||||
const sequence = self.next_sequence;
|
||||
self.next_sequence += 1;
|
||||
|
||||
const header = abi.KlogRecordHeader{
|
||||
.magic = abi.klog_record_magic,
|
||||
.level = level,
|
||||
.name_len = @intCast(name_len),
|
||||
.pid = pid,
|
||||
.sequence = sequence,
|
||||
.timestamp_ns = timestamp_ns,
|
||||
.message_len = @intCast(message_len),
|
||||
.flags = if (truncated) abi.klog_flag_truncated else 0,
|
||||
._reserved = @splat(0),
|
||||
};
|
||||
self.put(self.head, std.mem.asBytes(&header));
|
||||
self.put(self.head + abi.klog_record_header_size, name[0..name_len]);
|
||||
self.put(self.head + abi.klog_record_header_size + name_len, message[0..message_len]);
|
||||
// The alignment pad is dead space; zero it so raw dumps stay tidy.
|
||||
var pad = abi.klog_record_header_size + name_len + message_len;
|
||||
while (pad < record_len) : (pad += 1)
|
||||
self.buffer[@intCast((self.head + pad) % capacity)] = 0;
|
||||
self.head += record_len;
|
||||
return sequence;
|
||||
}
|
||||
|
||||
/// Copy stream bytes beginning at `offset` into `out`. Returns null if
|
||||
/// `offset` fell behind `tail` (overwritten) or lies past `head` — the
|
||||
/// reader re-syncs from status(). 0 bytes means caught up.
|
||||
pub fn read(self: *const Self, offset: u64, out: []u8) ?usize {
|
||||
if (offset < self.tail or offset > self.head) return null;
|
||||
const n: usize = @intCast(@min(out.len, self.head - offset));
|
||||
self.get(offset, out[0..n]);
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Cursors for klog_status. boot_unix_seconds is the kernel wrapper's
|
||||
/// to fill — the ring knows nothing of wall clocks.
|
||||
pub fn status(self: *const Self) abi.KlogStatus {
|
||||
return .{
|
||||
.tail = self.tail,
|
||||
.head = self.head,
|
||||
.next_sequence = self.next_sequence,
|
||||
.boot_unix_seconds = 0,
|
||||
};
|
||||
}
|
||||
|
||||
fn reclaimOne(self: *Self) void {
|
||||
var header_bytes: [abi.klog_record_header_size]u8 = undefined;
|
||||
self.get(self.tail, &header_bytes);
|
||||
const header = std.mem.bytesToValue(abi.KlogRecordHeader, &header_bytes);
|
||||
// The writer wrote this header itself: the assert guards against
|
||||
// memory corruption, not bad input.
|
||||
std.debug.assert(header.magic == abi.klog_record_magic);
|
||||
self.tail += recordLength(header.name_len, header.message_len);
|
||||
}
|
||||
|
||||
fn recordLength(name_len: usize, message_len: usize) usize {
|
||||
return std.mem.alignForward(usize, abi.klog_record_header_size + name_len + message_len, abi.klog_record_alignment);
|
||||
}
|
||||
|
||||
// Byte-at-a-time modulo copies keep the wrap logic obviously correct;
|
||||
// if they ever show in a profile, split into two @memcpy spans.
|
||||
fn put(self: *Self, offset: u64, bytes: []const u8) void {
|
||||
for (bytes, 0..) |b, i| self.buffer[@intCast((offset + i) % capacity)] = b;
|
||||
}
|
||||
|
||||
fn get(self: *const Self, offset: u64, out: []u8) void {
|
||||
for (out, 0..) |*b, i| b.* = self.buffer[@intCast((offset + i) % capacity)];
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
// --- tests (host) -----------------------------------------------------------
|
||||
|
||||
const TestRing = Ring(4096);
|
||||
|
||||
/// Parse the record at `offset` out of `ring`, returning the header plus name
|
||||
/// and message copies — the same walk a userspace drainer performs.
|
||||
const Parsed = struct {
|
||||
header: abi.KlogRecordHeader,
|
||||
name: [abi.maximum_process_name]u8 = undefined,
|
||||
message: [abi.klog_maximum_message]u8 = undefined,
|
||||
|
||||
fn nameSlice(self: *const Parsed) []const u8 {
|
||||
return self.name[0..self.header.name_len];
|
||||
}
|
||||
fn messageSlice(self: *const Parsed) []const u8 {
|
||||
return self.message[0..self.header.message_len];
|
||||
}
|
||||
fn next(self: *const Parsed, offset: u64) u64 {
|
||||
return offset + std.mem.alignForward(usize, abi.klog_record_header_size + self.header.name_len + self.header.message_len, abi.klog_record_alignment);
|
||||
}
|
||||
};
|
||||
|
||||
fn parseAt(ring: *const TestRing, offset: u64) Parsed {
|
||||
var p: Parsed = undefined;
|
||||
var header_bytes: [abi.klog_record_header_size]u8 = undefined;
|
||||
std.debug.assert(ring.read(offset, &header_bytes).? == header_bytes.len);
|
||||
p.header = std.mem.bytesToValue(abi.KlogRecordHeader, &header_bytes);
|
||||
std.debug.assert(p.header.magic == abi.klog_record_magic);
|
||||
_ = ring.read(offset + abi.klog_record_header_size, p.name[0..p.header.name_len]);
|
||||
_ = ring.read(offset + abi.klog_record_header_size + p.header.name_len, p.message[0..p.header.message_len]);
|
||||
return p;
|
||||
}
|
||||
|
||||
test "header size is pinned" {
|
||||
try std.testing.expectEqual(abi.klog_record_header_size, @sizeOf(abi.KlogRecordHeader));
|
||||
}
|
||||
|
||||
test "append/read round trip" {
|
||||
var ring = std.testing.allocator.create(TestRing) catch unreachable;
|
||||
defer std.testing.allocator.destroy(ring);
|
||||
ring.* = .{};
|
||||
|
||||
_ = ring.append(7, "/system/services/fat", .info, 123, "mounted /mnt/usb", false);
|
||||
_ = ring.append(0, "kernel", .raw, 456, "wall clock online", false);
|
||||
|
||||
const first = parseAt(ring, ring.tail);
|
||||
try std.testing.expectEqual(@as(u32, 7), first.header.pid);
|
||||
try std.testing.expectEqual(abi.KlogLevel.info, first.header.level);
|
||||
try std.testing.expectEqual(@as(u64, 123), first.header.timestamp_ns);
|
||||
try std.testing.expectEqualStrings("/system/services/fat", first.nameSlice());
|
||||
try std.testing.expectEqualStrings("mounted /mnt/usb", first.messageSlice());
|
||||
|
||||
const second = parseAt(ring, first.next(ring.tail));
|
||||
try std.testing.expectEqual(@as(u32, 0), second.header.pid);
|
||||
try std.testing.expectEqualStrings("kernel", second.nameSlice());
|
||||
try std.testing.expectEqual(@as(u64, 1), second.header.sequence);
|
||||
}
|
||||
|
||||
test "wrap reclaims whole records and keeps tail on a boundary" {
|
||||
var ring = std.testing.allocator.create(TestRing) catch unreachable;
|
||||
defer std.testing.allocator.destroy(ring);
|
||||
ring.* = .{};
|
||||
|
||||
// Fill far past capacity so the ring wraps many times.
|
||||
var i: u32 = 0;
|
||||
while (i < 200) : (i += 1) {
|
||||
var message: [64]u8 = undefined;
|
||||
const m = std.fmt.bufPrint(&message, "line {d} padding padding padding", .{i}) catch unreachable;
|
||||
_ = ring.append(1, "/system/tests/writer", .info, i, m, false);
|
||||
}
|
||||
try std.testing.expect(ring.head - ring.tail <= 4096);
|
||||
|
||||
// The record at tail parses cleanly (boundary held), and walking to head
|
||||
// yields consecutive sequence numbers.
|
||||
var offset = ring.tail;
|
||||
var previous: ?u64 = null;
|
||||
while (offset < ring.head) {
|
||||
const p = parseAt(ring, offset);
|
||||
if (previous) |q| try std.testing.expectEqual(q + 1, p.header.sequence);
|
||||
previous = p.header.sequence;
|
||||
offset = p.next(offset);
|
||||
}
|
||||
try std.testing.expectEqual(ring.head, offset);
|
||||
// Records were lost (sequence at tail > 0), and the count is the gap.
|
||||
try std.testing.expect(parseAt(ring, ring.tail).header.sequence > 0);
|
||||
}
|
||||
|
||||
test "stale offset returns null; head offset reads zero bytes" {
|
||||
var ring = std.testing.allocator.create(TestRing) catch unreachable;
|
||||
defer std.testing.allocator.destroy(ring);
|
||||
ring.* = .{};
|
||||
|
||||
var i: u32 = 0;
|
||||
while (i < 300) : (i += 1)
|
||||
_ = ring.append(1, "w", .info, i, "0123456789abcdef0123456789abcdef", false);
|
||||
|
||||
var out: [16]u8 = undefined;
|
||||
try std.testing.expect(ring.read(0, &out) == null); // long overwritten
|
||||
try std.testing.expect(ring.read(ring.head + 1, &out) == null); // past the end
|
||||
try std.testing.expectEqual(@as(usize, 0), ring.read(ring.head, &out).?); // caught up
|
||||
}
|
||||
|
||||
test "truncation flag and clamping" {
|
||||
var ring = std.testing.allocator.create(TestRing) catch unreachable;
|
||||
defer std.testing.allocator.destroy(ring);
|
||||
ring.* = .{};
|
||||
|
||||
const long = "x" ** 300; // past klog_maximum_message
|
||||
_ = ring.append(2, "w", .warn, 0, long, true);
|
||||
const p = parseAt(ring, ring.tail);
|
||||
try std.testing.expectEqual(@as(u16, abi.klog_maximum_message), p.header.message_len);
|
||||
try std.testing.expect(p.header.flags & abi.klog_flag_truncated != 0);
|
||||
}
|
||||
+171
-39
@@ -4,22 +4,37 @@
|
||||
//! Output is a *diagnostic convenience, never a correctness dependency* — the
|
||||
//! kernel must boot and run correctly with zero output channels. So logging fans
|
||||
//! out to a set of registered **sinks**, each best-effort and self-guarding: the
|
||||
//! serial UART, the 0xE9 debug console, and — later — a file on a ramdisk/USB/SSD.
|
||||
//! A message reaches whatever channels exist; if none do, the kernel runs on,
|
||||
//! silent but correct.
|
||||
//! serial UART and the 0xE9 debug console. A message reaches whatever channels
|
||||
//! exist; if none do, the kernel runs on, silent but correct.
|
||||
//!
|
||||
//! Retention is the tagged RING (log-ring.zig): every emission becomes one
|
||||
//! record per line, stamped with the sender's pid, task name (its binary path),
|
||||
//! level, sequence number, and monotonic timestamp — attribution is structural,
|
||||
//! stamped by the kernel, not a naming convention a process could forge. The
|
||||
//! stamping is per LINE: an embedded '\n' ends the record, so a payload cannot
|
||||
//! imitate another sender on the line that follows. Oldest records are
|
||||
//! overwritten when the ring is full; sequence gaps make the loss countable.
|
||||
//! `klog_read`/`klog_status` expose the stream to userspace (the logger service
|
||||
//! drains it into per-process files once storage is up).
|
||||
//!
|
||||
//! Locking: a dedicated log spinlock, NOT the big kernel lock. `print` is
|
||||
//! called both inside and outside BKL sections (and from ISRs), so the log
|
||||
//! lock is taken with interrupts off and nothing inside it ever takes the BKL —
|
||||
//! lock order is strictly BKL -> log lock, never the reverse. Panic paths use a
|
||||
//! bounded try-acquire and fall back to sinks-only: a panic must never deadlock
|
||||
//! on its own diagnostics.
|
||||
//!
|
||||
//! The **framebuffer is deliberately not a sink here.** It's a separate output
|
||||
//! surface (a bootstrap text console today, a graphics device driver later), so
|
||||
//! the log never assumes the machine is text-based. `main.zig` mirrors a few
|
||||
//! user-facing status lines and panics to it explicitly; the verbose log does not.
|
||||
//!
|
||||
//! No allocation: the sink table is fixed, so the log works before the heap is up
|
||||
//! and inside a panic. Two channels don't go through the sink list because they
|
||||
//! must survive even a total-output failure: `checkpoint` (a one-byte POST code)
|
||||
//! and `recordPanic` (a breadcrumb in a fixed record).
|
||||
//! the log never assumes the machine is text-based. Two channels bypass the
|
||||
//! sink list because they must survive even a total-output failure:
|
||||
//! `checkpoint` (a one-byte POST code) and `recordPanic` (a fixed breadcrumb).
|
||||
|
||||
const std = @import("std");
|
||||
const architecture = @import("architecture");
|
||||
const abi = @import("abi");
|
||||
const log_ring = @import("log-ring.zig");
|
||||
const wall_clock = @import("wall-clock.zig");
|
||||
|
||||
pub const SinkFn = *const fn ([]const u8) void;
|
||||
|
||||
@@ -36,42 +51,141 @@ pub fn addSink(sink: SinkFn) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Fan `bytes` out to every registered sink.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
// --- the log lock ------------------------------------------------------------
|
||||
|
||||
var lock_held = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn lockAcquire() u64 {
|
||||
const flags = architecture.saveInterrupts();
|
||||
while (lock_held.cmpxchgWeak(0, 1, .acquire, .monotonic) != null) std.atomic.spinLoopHint();
|
||||
return flags;
|
||||
}
|
||||
|
||||
// --- the RAM sink: a retained copy of the whole diagnostic stream ------------
|
||||
//
|
||||
// A fixed in-image buffer that accumulates every logged byte, so a user program
|
||||
// (`log-flush`, and init at shutdown) can read it back through `klog_read` and
|
||||
// persist it to a file — the boot log survives on a headless/real machine that
|
||||
// has no host capturing serial. It is a *sink like any other*: register it with
|
||||
// `addSink(ramSink)` at boot. No allocation (works pre-heap and in a panic).
|
||||
//
|
||||
// It fills linearly and stops when full: the earliest output — the most valuable
|
||||
// for diagnosing a boot — is kept, and the tail is still on the live serial sink.
|
||||
// 256 KiB comfortably holds a full boot plus a long run (a boot is ~15 KiB).
|
||||
fn lockTryAcquire(spins: usize) ?u64 {
|
||||
const flags = architecture.saveInterrupts();
|
||||
var i: usize = 0;
|
||||
while (i < spins) : (i += 1) {
|
||||
if (lock_held.cmpxchgWeak(0, 1, .acquire, .monotonic) == null) return flags;
|
||||
std.atomic.spinLoopHint();
|
||||
}
|
||||
architecture.restoreInterrupts(flags);
|
||||
return null;
|
||||
}
|
||||
|
||||
const ram_capacity = 256 * 1024;
|
||||
var ram_buffer: [ram_capacity]u8 = undefined;
|
||||
var ram_len: usize = 0;
|
||||
fn lockRelease(flags: u64) void {
|
||||
lock_held.store(0, .release);
|
||||
architecture.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// The RAM sink. Best-effort and self-guarding like every sink: appends what fits
|
||||
/// and silently drops the rest once full. (Concurrency matches the other sinks —
|
||||
/// the dominant writer, debug_write, already holds the kernel lock; a rare torn
|
||||
/// append on a kernel-internal line is an accepted diagnostic imperfection.)
|
||||
pub fn ramSink(bytes: []const u8) void {
|
||||
const n = @min(ram_buffer.len - ram_len, bytes.len);
|
||||
if (n != 0) {
|
||||
@memcpy(ram_buffer[ram_len..][0..n], bytes[0..n]);
|
||||
ram_len += n;
|
||||
// --- the ring + renderer -----------------------------------------------------
|
||||
|
||||
/// 512 KiB: the tagged frames cost ~30% over the raw text, and the ring only
|
||||
/// needs to cover the pre-mount backlog (a boot is ~15 KiB of text) — the
|
||||
/// logger service tails it continuously once storage is up.
|
||||
const ring_capacity = 512 * 1024;
|
||||
var ring: log_ring.Ring(ring_capacity) = .{};
|
||||
|
||||
/// Renderer state: whether the sinks sit at a line start, and which pid's line
|
||||
/// is currently open — when a different sender interleaves mid-line, the
|
||||
/// renderer closes the line so serial output can't visually merge two senders.
|
||||
var at_line_start: bool = true;
|
||||
var open_line_pid: u32 = 0;
|
||||
|
||||
/// Append `bytes` as one tagged record per line and render them to the sinks.
|
||||
/// The core emission path: `debug_write` calls this with the sender's identity;
|
||||
/// kernel-internal `write`/`print` funnel here as pid 0 ("kernel", raw).
|
||||
pub fn append(pid: u32, name: []const u8, level: abi.KlogLevel, bytes: []const u8) void {
|
||||
if (bytes.len == 0) return;
|
||||
const now = architecture.nanos();
|
||||
const flags = lockAcquire();
|
||||
defer lockRelease(flags);
|
||||
appendLocked(pid, name, level, now, bytes);
|
||||
}
|
||||
|
||||
/// The panic-safe variant: bounded lock wait; on failure, sinks only — the ring
|
||||
/// entry is lost but the message still reaches serial, and the panic cannot
|
||||
/// deadlock on a core that died holding the log lock.
|
||||
pub fn appendPanic(bytes: []const u8) void {
|
||||
if (lockTryAcquire(100_000)) |flags| {
|
||||
defer lockRelease(flags);
|
||||
appendLocked(0, "kernel", .raw, architecture.nanos(), bytes);
|
||||
} else {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
}
|
||||
}
|
||||
|
||||
/// The accumulated log so far — what `klog_read` copies out.
|
||||
pub fn ramSnapshot() []const u8 {
|
||||
return ram_buffer[0..ram_len];
|
||||
fn appendLocked(pid: u32, name: []const u8, level: abi.KlogLevel, now: u64, bytes: []const u8) void {
|
||||
var rest = bytes;
|
||||
while (rest.len != 0) {
|
||||
const newline = std.mem.indexOfScalar(u8, rest, '\n');
|
||||
// The record payload excludes the newline: a record IS a line. Raw
|
||||
// emissions may leave a line open (kernel boot tables build lines from
|
||||
// pieces); a LEVELED record is a complete line by contract — std.log
|
||||
// payloads carry no trailing newline.
|
||||
const line = if (newline) |i| rest[0..i] else rest;
|
||||
const line_complete = newline != null or level != .raw;
|
||||
if (line.len != 0 or line_complete)
|
||||
_ = ring.append(pid, name, level, now, line, line.len > abi.klog_maximum_message);
|
||||
render(pid, name, level, line, line_complete);
|
||||
rest = if (newline) |i| rest[i + 1 ..] else rest[rest.len..];
|
||||
}
|
||||
}
|
||||
|
||||
/// Serial/debugcon rendering. Kernel output and legacy raw user output pass
|
||||
/// through byte-identical to the historical stream (services still write their
|
||||
/// own "name: " prefixes until the std.log migration). Leveled (std.log)
|
||||
/// records get a kernel-rendered "<name>: " prefix at line start — err/warn/
|
||||
/// debug also get their level spelled out.
|
||||
fn render(pid: u32, name: []const u8, level: abi.KlogLevel, line: []const u8, line_complete: bool) void {
|
||||
if (sink_count == 0) return;
|
||||
if (line.len == 0 and !line_complete) return;
|
||||
// Compose the whole rendered piece first and emit it in ONE sink call per
|
||||
// sink: fewer, larger UART writes, and no partial-line window should any
|
||||
// path ever reach a sink without the log lock.
|
||||
var buffer: [render_buffer_size]u8 = undefined;
|
||||
var used: usize = 0;
|
||||
if (!at_line_start and open_line_pid != pid) {
|
||||
buffer[used] = '\n';
|
||||
used += 1;
|
||||
at_line_start = true;
|
||||
}
|
||||
if (at_line_start and level != .raw) {
|
||||
used += place(buffer[used..], name);
|
||||
used += place(buffer[used..], ": ");
|
||||
used += place(buffer[used..], switch (level) {
|
||||
.err => "error: ",
|
||||
.warn => "warning: ",
|
||||
.debug => "debug: ",
|
||||
.info, .raw => "",
|
||||
});
|
||||
}
|
||||
used += place(buffer[used..], line);
|
||||
if (line_complete and used < buffer.len) {
|
||||
buffer[used] = '\n';
|
||||
used += 1;
|
||||
}
|
||||
fanOut(buffer[0..used]);
|
||||
at_line_start = line_complete;
|
||||
open_line_pid = pid;
|
||||
}
|
||||
|
||||
/// newline + name + ": warning: " + a full payload line + newline.
|
||||
const render_buffer_size = 1 + abi.maximum_process_name + 11 + abi.klog_maximum_message + 1;
|
||||
|
||||
fn place(destination: []u8, bytes: []const u8) usize {
|
||||
const n = @min(destination.len, bytes.len);
|
||||
@memcpy(destination[0..n], bytes[0..n]);
|
||||
return n;
|
||||
}
|
||||
|
||||
fn fanOut(bytes: []const u8) void {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
}
|
||||
|
||||
/// Kernel-internal write — a raw record from "kernel" (pid 0). The signature is
|
||||
/// unchanged so every existing kernel call site stays as it is.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
append(0, "kernel", .raw, bytes);
|
||||
}
|
||||
|
||||
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
||||
@@ -81,6 +195,24 @@ pub fn print(comptime fmt: []const u8, args: anytype) void {
|
||||
write(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// klog_read: copy ring stream bytes from `offset` into `out`. Null when the
|
||||
/// cursor was overwritten or lies past the end — the reader re-syncs via
|
||||
/// status(). Zero bytes means caught up.
|
||||
pub fn readAt(offset: u64, out: []u8) ?usize {
|
||||
const flags = lockAcquire();
|
||||
defer lockRelease(flags);
|
||||
return ring.read(offset, out);
|
||||
}
|
||||
|
||||
/// klog_status: the ring cursors plus the boot wall-clock anchor.
|
||||
pub fn status() abi.KlogStatus {
|
||||
const flags = lockAcquire();
|
||||
defer lockRelease(flags);
|
||||
var s = ring.status();
|
||||
s.boot_unix_seconds = wall_clock.bootSeconds();
|
||||
return s;
|
||||
}
|
||||
|
||||
/// Emit a one-byte checkpoint/POST code (I/O port 0x80) — the always-available
|
||||
/// progress channel for when there is no text output at all. Independent of the
|
||||
/// sink list, so it works even before any sink is registered.
|
||||
|
||||
+373
-149
@@ -34,6 +34,7 @@ const ipc = @import("ipc-synchronous.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const vfs = @import("vfs.zig");
|
||||
const log = @import("log.zig");
|
||||
const wall_clock = @import("wall-clock.zig");
|
||||
|
||||
@@ -58,8 +59,9 @@ pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pa
|
||||
|
||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
|
||||
/// 1 GiB window is far more than any user heap needs today.
|
||||
/// process bump-allocates from `heap_arena_base` upward via a per-address-space cursor
|
||||
/// (`scheduler.addressSpaceMmapNextPtr`, shared by its threads); a 1 GiB window is far more
|
||||
/// than any user heap needs today.
|
||||
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
||||
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
||||
|
||||
@@ -69,8 +71,8 @@ pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
|
||||
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
||||
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
||||
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
|
||||
/// `Task.device_map_next`.
|
||||
/// device pages user-accessible widens no kernel mapping. Per-address-space cursor
|
||||
/// (`scheduler.addressSpaceDeviceMapNextPtr`).
|
||||
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
|
||||
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||
|
||||
@@ -81,17 +83,17 @@ pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
|
||||
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
|
||||
|
||||
/// The shared-memory arena: where `shm_create`/`shm_map` place shared cacheable regions, in
|
||||
/// The shared-memory arena: where `shared_memory_create`/`shared_memory_map` place shared cacheable regions, in
|
||||
/// PML4[230] — a user-exclusive region distinct from the DMA arena. The frames are owned by
|
||||
/// a refcounted shm object and freed when its last capability drops, not on teardown, so the
|
||||
/// mapping carries `device_grant`. Per-process cursor in `Task.shm_map_next` (docs/display-v2.md).
|
||||
pub const shm_arena_base: u64 = 0x0000_7300_0000_0000;
|
||||
pub const shm_arena_end: u64 = shm_arena_base + (256 << 20); // 256 MiB per process
|
||||
/// a refcounted shared-memory object and freed when its last capability drops, not on teardown, so the
|
||||
/// mapping carries `device_grant`. Per-process cursor in `Task.shared_memory_map_next` (docs/display-v2.md).
|
||||
pub const shared_memory_arena_base: u64 = 0x0000_7300_0000_0000;
|
||||
pub const shared_memory_arena_end: u64 = shared_memory_arena_base + (256 << 20); // 256 MiB per process
|
||||
|
||||
/// Largest single `shm_create`, in pages (32 MiB) — enough for a 4K framebuffer surface;
|
||||
/// also an overflow guard on the page count. shm frames are contiguous (like DMA), so this
|
||||
/// Largest single `shared_memory_create`, in pages (32 MiB) — enough for a 4K framebuffer surface;
|
||||
/// also an overflow guard on the page count. shared-memory frames are contiguous (like DMA), so this
|
||||
/// bounds the contiguous allocation asked of the frame allocator.
|
||||
const maximum_shm_pages = 8192;
|
||||
const maximum_shared_memory_pages = 8192;
|
||||
|
||||
/// Largest single `mmap` grant, in pages (32 MiB). Big enough for a display service's
|
||||
/// back buffer at up to 4K (3840x2160x4 ≈ 8100 pages); the user heap otherwise grows in
|
||||
@@ -137,6 +139,17 @@ var ramdisk_image: ?[]const u8 = null;
|
||||
/// handoff) so a user-space supervisor can `system_spawn` binaries out of it.
|
||||
pub fn setInitialRamdisk(image: []const u8) void {
|
||||
ramdisk_image = image;
|
||||
vfs.setInitialRamdisk(image); // the kernel VFS serves the same bytes at /system
|
||||
}
|
||||
|
||||
/// Spawn a bundled binary from the kernel by path. Used exactly once, to start
|
||||
/// /system/services/init (PID 1) — every other spawn goes through the
|
||||
/// `system_spawn` syscall.
|
||||
pub fn spawnBundled(name: []const u8) !void {
|
||||
const image = ramdisk_image orelse return error.NoInitialRamdisk;
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return error.BadInitialRamdisk;
|
||||
const item = rd.find(name) orelse return error.NotBundled;
|
||||
try spawnProcess(item.blob, 4, &.{item.name});
|
||||
}
|
||||
|
||||
/// The system_call surface, dispatched on the saved system_call number (`abi.SystemCall`).
|
||||
@@ -163,7 +176,7 @@ fn fail(state: *architecture.CpuState) void {
|
||||
|
||||
fn system_call(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
const user = t.aspace != 0;
|
||||
const user = t.address_space != 0;
|
||||
if (user) {
|
||||
// A condemned process (process_kill caught it running) dies at its next
|
||||
// kernel entry — before it can spawn, claim, or message anything else.
|
||||
@@ -223,13 +236,20 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.process_signal => systemProcessSignal(state),
|
||||
.timer_bind => systemTimerBind(state),
|
||||
.klog_read => systemKlogRead(state),
|
||||
.klog_status => systemKlogStatus(state),
|
||||
.fs_resolve => systemFsResolve(state),
|
||||
.fs_node => systemFsNode(state),
|
||||
.fs_mount => systemFsMount(state),
|
||||
.fs_unmount => systemFsUnmount(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
.shm_create => systemShmCreate(state),
|
||||
.shm_map => systemShmMap(state),
|
||||
.shm_physical => systemShmPhysical(state),
|
||||
.shared_memory_create => systemSharedMemoryCreate(state),
|
||||
.shared_memory_map => systemSharedMemoryMap(state),
|
||||
.shared_memory_physical => systemSharedMemoryPhysical(state),
|
||||
.thread_spawn => systemThreadSpawn(state),
|
||||
.current_core => systemCurrentCore(state),
|
||||
.thread_self => systemThreadSelf(state),
|
||||
.thread_join => systemThreadJoin(state),
|
||||
.set_thread_pointer => systemSetThreadPointer(state),
|
||||
.futex_wait => systemFutexWait(state),
|
||||
.futex_wake => systemFutexWake(state),
|
||||
.thread_exit => {
|
||||
@@ -253,6 +273,13 @@ fn failErr(state: *architecture.CpuState, errno: i64) void {
|
||||
/// create_ipc_endpoint() -> handle: allocate an endpoint and install it in the
|
||||
/// caller's handle table.
|
||||
fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
|
||||
// Under the big kernel lock: this allocates from the kernel heap and mutates the
|
||||
// caller's handle table. A multi-threaded process (e.g. the display's compositor +
|
||||
// mouse-listener threads) can drive this concurrently from two cores, so the endpoint
|
||||
// allocation and every other lock holder must serialize (heap.zig: "a lock comes with
|
||||
// threads/SMP").
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return failErr(state, ipc.ENOMEM);
|
||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||
if (h < 0) {
|
||||
@@ -265,6 +292,10 @@ fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
|
||||
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
||||
/// well-known id so other processes can find it.
|
||||
fn systemIpcRegister(state: *architecture.CpuState) void {
|
||||
// Under the big kernel lock: mutates the global service registry and endpoint
|
||||
// refcounts, which threads of the same (or another) process can race.
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
|
||||
@@ -273,6 +304,11 @@ fn systemIpcRegister(state: *architecture.CpuState) void {
|
||||
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
||||
/// handle to it in the caller.
|
||||
fn systemIpcLookup(state: *architecture.CpuState) void {
|
||||
// Under the big kernel lock: reads the global registry, takes an endpoint reference,
|
||||
// and installs a handle — all racy against concurrent threads (this is the path the
|
||||
// display's mouse-listener thread takes to reach the compositor endpoint).
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||
@@ -314,7 +350,7 @@ fn systemIpcReplyWait(state: *architecture.CpuState) void {
|
||||
fn systemIpcSend(state: *architecture.CpuState) void {
|
||||
const me = scheduler.current();
|
||||
const endpoint = ipc.resolveHandle(me, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const r = ipc.send(endpoint, me.aspace, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), me.id);
|
||||
const r = ipc.send(endpoint, me.address_space, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), me.id);
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
}
|
||||
|
||||
@@ -324,7 +360,7 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||
const maximum = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
||||
@@ -348,14 +384,14 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
} else fail(state);
|
||||
}
|
||||
|
||||
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
||||
/// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into
|
||||
/// this address space (strong-uncacheable) and return the register base address.
|
||||
/// The claim is the capability — a process can only map hardware it owns.
|
||||
fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const resource_index = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
// Read the broker table under the lock: ring-3 device_register (M19) now
|
||||
// mutates it concurrently on other cores, so a lock-free read here could
|
||||
// see a torn resource (and a torn length used to panic the arithmetic
|
||||
@@ -373,18 +409,23 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
if (r.len == 0) return fail(state);
|
||||
if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state);
|
||||
|
||||
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
||||
const first = r.start & ~@as(u64, page_size - 1);
|
||||
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
|
||||
const pages = (last - first) / page_size + 1;
|
||||
const base_v = t.device_map_next;
|
||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||
|
||||
// A framebuffer resource asks (via its flag) to be mapped write-combining rather
|
||||
// than the strong-uncacheable default that register MMIO needs.
|
||||
const write_combining = (r.flags & device_abi.resource_flag_write_combining) != 0;
|
||||
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len, write_combining);
|
||||
t.device_map_next = base_v + pages * page_size;
|
||||
|
||||
// Per-address-space cursor + shared page tables → serialize under the big lock,
|
||||
// same as mmap (docs/threading-plan.md M7).
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const cursor = scheduler.addressSpaceDeviceMapNextPtr(t.address_space) orelse return fail(state);
|
||||
if (cursor.* == 0) cursor.* = device_arena_base; // seed the arena lazily
|
||||
const base_v = cursor.*;
|
||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||
architecture.mapUserDeviceInto(t.address_space, base_v, r.start, r.len, write_combining);
|
||||
cursor.* = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||
}
|
||||
|
||||
@@ -413,7 +454,7 @@ pub fn resolveIoPort(t: *scheduler.Task, device_id: u64, resource_index: u64, of
|
||||
/// is fine. See docs/drivers.md.
|
||||
fn systemIoRead(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const width = architecture.systemCallArg(state, 3);
|
||||
const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state);
|
||||
architecture.setSystemCallResult(state, architecture.pioRead(@intCast(width), port));
|
||||
@@ -424,14 +465,14 @@ fn systemIoRead(state: *architecture.CpuState) void {
|
||||
/// gate as `io_read`.
|
||||
fn systemIoWrite(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const width = architecture.systemCallArg(state, 3);
|
||||
const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state);
|
||||
architecture.pioWrite(@intCast(width), port, @intCast(architecture.systemCallArg(state, 4)));
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): grant `len` bytes (rounded up to
|
||||
/// dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): grant `len` bytes (rounded up to
|
||||
/// whole pages) of DMA-capable memory — physically contiguous, zeroed, pinned, and
|
||||
/// strong-uncacheable (coherent) — mapping it into the caller's DMA arena and handing
|
||||
/// back both the virtual address to touch and the physical address to program into the
|
||||
@@ -443,7 +484,7 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const flags = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or len == 0) return fail(state);
|
||||
if (t.address_space == 0 or len == 0) return fail(state);
|
||||
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
const max_phys: u64 = if (flags & abi.dma_below_4g != 0) (@as(u64, 4) << 30) else ~@as(u64, 0);
|
||||
@@ -459,14 +500,14 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
// Zero through the physmap (the frames aren't mapped in the caller yet), then map.
|
||||
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
|
||||
@memset(kernel_view[0 .. pages * page_size], 0);
|
||||
architecture.mapUserDmaInto(t.aspace, base_v, phys, pages * page_size);
|
||||
architecture.mapUserDmaInto(t.address_space, base_v, phys, pages * page_size);
|
||||
|
||||
t.dma_map_next = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
||||
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
||||
}
|
||||
|
||||
/// dma_free(vaddr, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||
/// it can never unmap-and-free the caller's stack, heap, or an MMIO grant; only pages
|
||||
/// actually mapped are freed (an unmapped hole is skipped). Teardown also reclaims any
|
||||
/// DMA pages left mapped at exit (they carry no `device_grant`, so `freeSubtree` frees
|
||||
@@ -475,93 +516,93 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
const base_v = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base_v + i * page_size;
|
||||
if (architecture.translate(t.aspace, va)) |phys| {
|
||||
architecture.unmapUserPageInto(t.aspace, va);
|
||||
if (architecture.translate(t.address_space, va)) |phys| {
|
||||
architecture.unmapUserPageInto(t.address_space, va);
|
||||
pmm.free(phys);
|
||||
}
|
||||
}
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
||||
/// shared_memory_create(len) -> virtual_address (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
||||
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
|
||||
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
|
||||
/// caller's shared-memory arena — and hand back the virtual address plus a capability handle. Unlike
|
||||
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
|
||||
/// its frames are owned by a refcounted object: the handle is passed to another process as
|
||||
/// an `ipc_call` send_cap, that process `shm_map`s it, and the frames free only when the
|
||||
/// an `ipc_call` send_cap, that process `shared_memory_map`s it, and the frames free only when the
|
||||
/// last capability drops (docs/display-v2.md — the compositor↔native-driver and
|
||||
/// app↔compositor surface path).
|
||||
fn systemShmCreate(state: *architecture.CpuState) void {
|
||||
fn systemSharedMemoryCreate(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or len == 0) return fail(state);
|
||||
if (t.address_space == 0 or len == 0) return fail(state);
|
||||
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
|
||||
if (pages == 0 or pages > maximum_shared_memory_pages) return fail(state);
|
||||
|
||||
// Reserve arena virtual space up front, so a mapping failure needs no rollback.
|
||||
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||
const base_v = t.shm_map_next;
|
||||
if (base_v + pages * page_size > shm_arena_end) return fail(state); // arena exhausted
|
||||
if (t.shared_memory_map_next == 0) t.shared_memory_map_next = shared_memory_arena_base;
|
||||
const base_v = t.shared_memory_map_next;
|
||||
if (base_v + pages * page_size > shared_memory_arena_end) return fail(state); // arena exhausted
|
||||
|
||||
const phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse return fail(state);
|
||||
// Zero through the physmap (the frames aren't mapped in the caller yet).
|
||||
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
|
||||
@memset(kernel_view[0 .. pages * page_size], 0);
|
||||
|
||||
const shm = ipc.createShm(phys, pages) orelse {
|
||||
const shared_memory = ipc.createSharedMemory(phys, pages) orelse {
|
||||
for (0..pages) |i| pmm.free(phys + i * page_size);
|
||||
return fail(state);
|
||||
};
|
||||
const handle = ipc.installShmHandle(t, shm);
|
||||
const handle = ipc.installSharedMemoryHandle(t, shared_memory);
|
||||
if (handle < 0) {
|
||||
ipc.dropShmRef(shm); // last ref: frees the object and its frames
|
||||
ipc.dropSharedMemoryReference(shared_memory); // last ref: frees the object and its frames
|
||||
return fail(state);
|
||||
}
|
||||
|
||||
architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size);
|
||||
t.shm_map_next = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v); // vaddr for the CPU
|
||||
architecture.mapUserSharedInto(t.address_space, base_v, phys, pages * page_size);
|
||||
t.shared_memory_map_next = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v); // virtual_address for the CPU
|
||||
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
|
||||
}
|
||||
|
||||
/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller
|
||||
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
|
||||
/// shared_memory_map(cap) -> virtual_address: map the shared region named by a capability handle the caller
|
||||
/// received (via an `ipc_call` send_cap) into its shared-memory arena — the same physical frames the
|
||||
/// creator sees — returning the virtual address. The handle already holds a reference (taken
|
||||
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
|
||||
fn systemShmMap(state: *architecture.CpuState) void {
|
||||
fn systemSharedMemoryMap(state: *architecture.CpuState) void {
|
||||
const cap = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
|
||||
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||
const base_v = t.shm_map_next;
|
||||
const size = shm.pages * page_size;
|
||||
if (base_v + size > shm_arena_end) return fail(state);
|
||||
const shared_memory = ipc.resolveSharedMemory(t, cap) orelse return fail(state); // not a shared-memory handle we hold
|
||||
if (t.shared_memory_map_next == 0) t.shared_memory_map_next = shared_memory_arena_base;
|
||||
const base_v = t.shared_memory_map_next;
|
||||
const size = shared_memory.pages * page_size;
|
||||
if (base_v + size > shared_memory_arena_end) return fail(state);
|
||||
|
||||
architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size);
|
||||
t.shm_map_next = base_v + size;
|
||||
architecture.mapUserSharedInto(t.address_space, base_v, shared_memory.phys, size);
|
||||
t.shared_memory_map_next = base_v + size;
|
||||
architecture.setSystemCallResult(state, base_v);
|
||||
}
|
||||
|
||||
/// shm_physical(cap) -> paddr: the guest-physical base of a shared region the caller holds a
|
||||
/// shared_memory_physical(cap) -> physical_address: the guest-physical base of a shared region the caller holds a
|
||||
/// capability for. The frames are contiguous (allocated by `allocContiguous`), so a single
|
||||
/// physical base + length describes the whole region — which is exactly what a driver needs
|
||||
/// to hand a shm surface to a device (virtio-gpu `attach_backing`). Only a holder of the
|
||||
/// to hand a shared-memory surface to a device (virtio-gpu `attach_backing`). Only a holder of the
|
||||
/// capability can ask; there is no ambient way to turn a virtual address into a physical one.
|
||||
fn systemShmPhysical(state: *architecture.CpuState) void {
|
||||
fn systemSharedMemoryPhysical(state: *architecture.CpuState) void {
|
||||
const cap = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||
architecture.setSystemCallResult(state, shm.phys);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const shared_memory = ipc.resolveSharedMemory(t, cap) orelse return fail(state); // not a shared-memory handle we hold
|
||||
architecture.setSystemCallResult(state, shared_memory.phys);
|
||||
}
|
||||
|
||||
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
||||
@@ -580,10 +621,10 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||
const parent_id = architecture.systemCallArg(state, 0);
|
||||
const descriptor_ptr = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
|
||||
var descriptor: device_abi.DeviceDescriptor = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
if (!ipc.copyFromUser(t.address_space, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
|
||||
// Under the big kernel lock: the broker's table is also mutated by the
|
||||
// death sweep (releaseAllOwnedBy) and read by enumerate on other cores —
|
||||
@@ -632,8 +673,11 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
||||
|
||||
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
|
||||
// Exact path first, basename fallback second; either way argv[0] (and hence
|
||||
// the task name, and the log ring's attribution) is the stored full path.
|
||||
const item = rd.find(name) orelse return fail(state); // no bundled binary by that name
|
||||
var argv: [maximum_arguments][]const u8 = undefined;
|
||||
argv[0] = name;
|
||||
argv[0] = item.name;
|
||||
var argc: usize = 1;
|
||||
if (arguments_len != 0) {
|
||||
const blob = @as([*]const u8, @ptrFromInt(arguments_ptr))[0..arguments_len];
|
||||
@@ -645,15 +689,8 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
||||
}
|
||||
}
|
||||
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!std.mem.eql(u8, item.name, name)) continue;
|
||||
const child = spawnProcessSupervised(item.blob, 4, argv[0..argc], t.id, exit_endpoint) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, child);
|
||||
return;
|
||||
}
|
||||
fail(state); // no bundled binary by that name
|
||||
const child = spawnProcessSupervised(item.blob, 4, argv[0..argc], t.id, exit_endpoint) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, child);
|
||||
}
|
||||
|
||||
/// thread_spawn(entry, stack_top, arg) -> tid: start a task that shares the **caller's**
|
||||
@@ -667,7 +704,7 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
|
||||
const arg = architecture.systemCallArg(state, 2);
|
||||
const exit_handle = architecture.systemCallArg(state, 3);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // kernel tasks own no address space to share
|
||||
if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share
|
||||
if (entry == 0 or entry >= user_half_end) return fail(state);
|
||||
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
|
||||
// The endpoint the thread notifies on exit (how join waits), or none.
|
||||
@@ -675,17 +712,17 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
|
||||
null
|
||||
else
|
||||
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||
const tid = spawnThreadSupervised(t.aspace, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state);
|
||||
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state);
|
||||
architecture.setSystemCallResult(state, tid);
|
||||
}
|
||||
|
||||
/// Spawn a thread sharing `aspace`, taking the exit-endpoint reference under the **same**
|
||||
/// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same**
|
||||
/// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before
|
||||
/// its reference exists. Returns the new thread id, or null on resource exhaustion.
|
||||
fn spawnThreadSupervised(aspace: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 {
|
||||
fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const tid = scheduler.spawnUserLocked(aspace, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null;
|
||||
const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null;
|
||||
if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death
|
||||
return tid;
|
||||
}
|
||||
@@ -700,6 +737,34 @@ fn systemThreadSelf(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, scheduler.currentId());
|
||||
}
|
||||
|
||||
/// set_thread_pointer(addr) -> 0: set the caller's user-space TLS thread pointer. The
|
||||
/// arch layer maps it to IA32_FS_BASE on x86_64, `TPIDR_EL0` on aarch64; the kernel
|
||||
/// never reads it, and the scheduler restores it per task across context switches
|
||||
/// (docs/threading-plan.md M10). `addr` must be a user-half address.
|
||||
fn systemSetThreadPointer(state: *architecture.CpuState) void {
|
||||
const addr = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.address_space == 0) return fail(state); // kernel tasks have no user TLS
|
||||
if (addr >= user_half_end) return fail(state);
|
||||
const flags = sync.enter();
|
||||
scheduler.setThreadPointerLocked(addr);
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// thread_join(tid) -> 0: block until the thread with id `tid` has exited (docs/threading-
|
||||
/// plan.md M9). Needs no per-thread IPC endpoint. The compare-and-block is one critical
|
||||
/// section, so an exit cannot slip between "is it alive?" and the block.
|
||||
fn systemThreadJoin(state: *architecture.CpuState) void {
|
||||
const tid: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const t = scheduler.current();
|
||||
if (t.address_space == 0) return fail(state); // kernel tasks don't join
|
||||
const flags = sync.enter();
|
||||
scheduler.joinThreadLocked(tid);
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// futex_wait(addr, expected, timeout_ns) -> status (docs/threading.md): if the 4-byte
|
||||
/// user word at `addr` still equals `expected`, block until a futex_wake on `addr` or
|
||||
/// (if timeout_ns > 0) the deadline. The compare and the block are one critical section,
|
||||
@@ -710,12 +775,12 @@ fn systemFutexWait(state: *architecture.CpuState) void {
|
||||
const expected: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||
const timeout_ns = architecture.systemCallArg(state, 2);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state);
|
||||
|
||||
const flags = sync.enter();
|
||||
var word_bytes: [4]u8 = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, addr, &word_bytes)) {
|
||||
if (!ipc.copyFromUser(t.address_space, addr, &word_bytes)) {
|
||||
sync.leave(flags);
|
||||
return fail(state);
|
||||
}
|
||||
@@ -739,10 +804,10 @@ fn systemFutexWake(state: *architecture.CpuState) void {
|
||||
const addr = architecture.systemCallArg(state, 0);
|
||||
const count: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state);
|
||||
const flags = sync.enter();
|
||||
const woken = scheduler.futexWakeLocked(t.aspace, addr, count);
|
||||
const woken = scheduler.futexWakeLocked(t.address_space, addr, count);
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, woken);
|
||||
}
|
||||
@@ -757,7 +822,7 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||
const maximum = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
const sz = @sizeOf(abi.ProcessDescriptor);
|
||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||
const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr);
|
||||
@@ -770,7 +835,7 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
||||
/// cannot be a weapon (ids are never reused, so a stale one just misses).
|
||||
fn systemProcessKill(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
const r = killProcess(t.id, @intCast(id));
|
||||
@@ -903,7 +968,7 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
||||
if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||
if (target.address_space == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||
if (target.supervisor != caller_id) return -ipc.EPERM;
|
||||
target.exit_reason = .killed;
|
||||
if (target.state == .running) {
|
||||
@@ -973,7 +1038,7 @@ var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exi
|
||||
/// secret between cooperating processes. -ENOSPC when the table is full.
|
||||
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
@@ -993,7 +1058,7 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||
/// delivered immediately on bind, coalesced into one notification.
|
||||
fn systemSignalBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
@@ -1013,7 +1078,7 @@ fn systemSignalBind(state: *architecture.CpuState) void {
|
||||
/// targets accumulate the signal in their pending mask.
|
||||
fn systemProcessSignal(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
const signal = architecture.systemCallArg(state, 1);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
@@ -1021,7 +1086,7 @@ fn systemProcessSignal(state: *architecture.CpuState) void {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH);
|
||||
if (target.aspace == 0) return failErr(state, ipc.ESRCH);
|
||||
if (target.address_space == 0) return failErr(state, ipc.ESRCH);
|
||||
if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM);
|
||||
target.pending_signals |= @as(u32, 1) << @intCast(signal);
|
||||
if (target.signal_endpoint) |raw| {
|
||||
@@ -1058,7 +1123,7 @@ fn timerSweepLocked() void {
|
||||
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
|
||||
fn systemTimerBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const ms = architecture.systemCallArg(state, 1);
|
||||
const flags = sync.enter();
|
||||
@@ -1075,7 +1140,7 @@ fn systemTimerBind(state: *architecture.CpuState) void {
|
||||
|
||||
fn systemProcessExitReason(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
const r = exitReasonOf(t.id, @intCast(id));
|
||||
@@ -1101,7 +1166,7 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
||||
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
||||
fn systemIrqBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
||||
@@ -1121,7 +1186,7 @@ fn systemIrqBind(state: *architecture.CpuState) void {
|
||||
fn systemMsiBind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||
if (owner != t.id) return fail(state); // not claimed by this process
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||
@@ -1140,7 +1205,7 @@ fn systemMsiBind(state: *architecture.CpuState) void {
|
||||
/// more arrives until the driver says it has serviced the hardware.
|
||||
fn systemIrqAck(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
|
||||
@@ -1152,7 +1217,6 @@ fn systemIrqAck(state: *architecture.CpuState) void {
|
||||
/// Whether the debug_write stream sits at the start of a line — the last emitted
|
||||
/// byte was a newline (true at boot: nothing emitted yet). Guarded by the kernel
|
||||
/// lock in `systemDebugWrite`, like the stream it describes.
|
||||
var write_at_line_start: bool = true;
|
||||
|
||||
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
|
||||
/// A bring-up diagnostic — real output goes through the VFS/console later.
|
||||
@@ -1173,31 +1237,42 @@ var write_at_line_start: bool = true;
|
||||
fn systemDebugWrite(state: *architecture.CpuState) void {
|
||||
const ptr = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
const level_raw = architecture.systemCallArg(state, 2);
|
||||
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
||||
const source: [*]const u8 = @ptrFromInt(ptr);
|
||||
// Levels above the enum range clamp to raw — old two-arg callers land
|
||||
// there naturally (garbage in arg 2 stays harmless).
|
||||
const level: abi.KlogLevel = if (level_raw <= @intFromEnum(abi.KlogLevel.raw))
|
||||
@enumFromInt(level_raw)
|
||||
else
|
||||
.raw;
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
@memcpy(write_buffer[0..len], source[0..len]); // keep the latest message
|
||||
write_len = len;
|
||||
write_from_user = architecture.fromUser(state);
|
||||
write_count += 1;
|
||||
log.write(source[0..len]);
|
||||
if (len != 0) write_at_line_start = source[len - 1] == '\n';
|
||||
// The kernel stamps the sender's identity — attribution is structural,
|
||||
// not a prefix convention the payload could forge (and it is stamped
|
||||
// per line inside log.append).
|
||||
log.append(t.id, t.name(), level, source[0..len]);
|
||||
architecture.setSystemCallResult(state, len);
|
||||
} else {
|
||||
fail(state);
|
||||
}
|
||||
}
|
||||
|
||||
/// klog_read(offset, ptr, len) -> bytes copied: copy the kernel's in-memory
|
||||
/// diagnostic log (the RAM sink in log.zig) out to the user buffer at `ptr`,
|
||||
/// starting at `offset`. Returns the count copied — 0 once `offset` reaches the
|
||||
/// end — so a program reads the whole log by looping from 0 until it gets 0.
|
||||
/// klog_read(offset, ptr, len) -> bytes copied: copy tagged log-ring stream
|
||||
/// bytes beginning at stream offset `offset` out to the user buffer at `ptr`.
|
||||
/// Returns the count copied — 0 means caught up — and fails once `offset` has
|
||||
/// fallen behind the ring's tail (the records were overwritten) or lies past
|
||||
/// its head; the reader re-syncs via klog_status. A reader parses
|
||||
/// [KlogRecordHeader][name][message] frames out of the byte stream (abi.zig).
|
||||
///
|
||||
/// The mirror of `debug_write`: the same overflow-safe user-half bounds check,
|
||||
/// but the copy runs kernel -> user. Written under the kernel lock so the source
|
||||
/// snapshot can't grow underneath the copy. A read-only diagnostic — it exposes
|
||||
/// only the log the kernel already broadcasts to serial, nothing else.
|
||||
/// but the copy runs kernel -> user, under the log lock (inside log.readAt) so
|
||||
/// the stream can't move underneath the copy. A read-only diagnostic.
|
||||
fn systemKlogRead(state: *architecture.CpuState) void {
|
||||
const offset = architecture.systemCallArg(state, 0);
|
||||
const ptr = architecture.systemCallArg(state, 1);
|
||||
@@ -1205,21 +1280,155 @@ fn systemKlogRead(state: *architecture.CpuState) void {
|
||||
// Confine the whole destination span to the user (low) half. `len <=
|
||||
// user_half_end - ptr` bounds the length without an overflowing add.
|
||||
if (ptr < user_half_end and len <= user_half_end - ptr) {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const snapshot = log.ramSnapshot();
|
||||
var n: usize = 0;
|
||||
if (offset < snapshot.len) {
|
||||
n = @min(len, snapshot.len - offset);
|
||||
const dest: [*]u8 = @ptrFromInt(ptr);
|
||||
@memcpy(dest[0..n], snapshot[offset..][0..n]);
|
||||
}
|
||||
const dest: [*]u8 = @ptrFromInt(ptr);
|
||||
const n = log.readAt(offset, dest[0..len]) orelse return fail(state);
|
||||
architecture.setSystemCallResult(state, n);
|
||||
} else {
|
||||
fail(state);
|
||||
}
|
||||
}
|
||||
|
||||
/// klog_status(ptr) -> 0: copy a KlogStatus — the ring's live cursors plus the
|
||||
/// boot wall-clock anchor — out to the user buffer at `ptr`. How a log reader
|
||||
/// finds the oldest retained offset, detects lost records (sequence gaps), and
|
||||
/// names a per-boot log directory (boot_unix_seconds).
|
||||
fn systemKlogStatus(state: *architecture.CpuState) void {
|
||||
const ptr = architecture.systemCallArg(state, 0);
|
||||
const size = @sizeOf(abi.KlogStatus);
|
||||
if (ptr < user_half_end and size <= user_half_end - ptr) {
|
||||
var status = log.status();
|
||||
const dest: [*]u8 = @ptrFromInt(ptr);
|
||||
@memcpy(dest[0..size], std.mem.asBytes(&status)[0..size]);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
} else {
|
||||
fail(state);
|
||||
}
|
||||
}
|
||||
|
||||
/// fs_resolve(path_ptr, path_len, flags, out_ptr, out_cap): route a path
|
||||
/// through the kernel mount table (docs/vfs-protocol.md). Kernel-served ->
|
||||
/// rax=fs_route_kernel, rdx=node token. Backend-served -> rax=fs_route_backend,
|
||||
/// rdx=an endpoint handle in the caller's table (deduplicated), and the
|
||||
/// rewritten mount-relative path copied into `out` behind a u16 length prefix. Fails for unknown paths, create-intent on /system, or an
|
||||
/// undersized out buffer.
|
||||
fn systemFsResolve(state: *architecture.CpuState) void {
|
||||
const path_ptr = architecture.systemCallArg(state, 0);
|
||||
const path_len = architecture.systemCallArg(state, 1);
|
||||
const flags = architecture.systemCallArg(state, 2);
|
||||
const out_ptr = architecture.systemCallArg(state, 3);
|
||||
const out_cap = architecture.systemCallArg(state, 4);
|
||||
if (path_len == 0 or path_len > 224 or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
|
||||
if (out_cap != 0 and (out_ptr >= user_half_end or out_ptr + out_cap > user_half_end)) return fail(state);
|
||||
const path = @as([*]const u8, @ptrFromInt(path_ptr))[0..path_len];
|
||||
const t = scheduler.current();
|
||||
|
||||
const flags_lock = sync.enter();
|
||||
defer sync.leave(flags_lock);
|
||||
switch (vfs.resolvePath(path, flags & abi.fs_flag_create != 0)) {
|
||||
.kernel_node => |node_token| {
|
||||
architecture.setSystemCallResult(state, abi.fs_route_kernel);
|
||||
architecture.setSystemCallResult2(state, node_token);
|
||||
},
|
||||
.backend => |*backend| {
|
||||
// The rewritten path goes back in the out buffer behind a u16
|
||||
// length prefix (a third result register would collide with r8's
|
||||
// argument role in the userspace stub).
|
||||
if (backend.path_len + 2 > out_cap) return fail(state);
|
||||
const handle = ipc.installHandleDeduped(t, backend.endpoint);
|
||||
if (handle < 0) return fail(state);
|
||||
const destination: [*]u8 = @ptrFromInt(out_ptr);
|
||||
destination[0] = @intCast(backend.path_len & 0xFF);
|
||||
destination[1] = @intCast(backend.path_len >> 8);
|
||||
@memcpy(destination[2..][0..backend.path_len], backend.path[0..backend.path_len]);
|
||||
architecture.setSystemCallResult(state, abi.fs_route_backend);
|
||||
architecture.setSystemCallResult2(state, @intCast(handle));
|
||||
},
|
||||
.not_found => fail(state),
|
||||
}
|
||||
}
|
||||
|
||||
/// fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: serve a
|
||||
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
|
||||
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
|
||||
/// of the immutable initrd never take the kernel lock.
|
||||
fn systemFsNode(state: *architecture.CpuState) void {
|
||||
const operation = architecture.systemCallArg(state, 0);
|
||||
const node_token = architecture.systemCallArg(state, 1);
|
||||
const offset = architecture.systemCallArg(state, 2);
|
||||
const buf_ptr = architecture.systemCallArg(state, 3);
|
||||
const buf_len = architecture.systemCallArg(state, 4);
|
||||
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
|
||||
const capped = @min(buf_len, 64 * 1024); // bound any single copy
|
||||
const destination: [*]u8 = @ptrFromInt(buf_ptr);
|
||||
switch (operation) {
|
||||
abi.fs_node_read => {
|
||||
const n = vfs.nodeRead(node_token, offset, destination[0..capped]) orelse return fail(state);
|
||||
architecture.setSystemCallResult(state, n);
|
||||
},
|
||||
abi.fs_node_status => {
|
||||
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
|
||||
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
|
||||
@memcpy(destination[0..@sizeOf(abi.FileAttributes)], std.mem.asBytes(&attributes));
|
||||
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
|
||||
},
|
||||
abi.fs_node_readdir => {
|
||||
const header_size = @sizeOf(abi.DirectoryEntryHeader);
|
||||
if (capped < header_size) return fail(state);
|
||||
var name_buffer: [64]u8 = undefined;
|
||||
const result = vfs.nodeReaddir(node_token, offset, &name_buffer) orelse {
|
||||
architecture.setSystemCallResult(state, 0); // past the end
|
||||
return;
|
||||
};
|
||||
var header = result.header;
|
||||
const total = header_size + @min(result.name_len, capped - header_size);
|
||||
@memcpy(destination[0..header_size], std.mem.asBytes(&header));
|
||||
@memcpy(destination[header_size..total], name_buffer[0 .. total - header_size]);
|
||||
architecture.setSystemCallResult(state, total);
|
||||
},
|
||||
else => fail(state),
|
||||
}
|
||||
}
|
||||
|
||||
/// fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len):
|
||||
/// mount a userspace filesystem at an absolute prefix. Possession of the
|
||||
/// backend endpoint handle is the capability — the same trust as the old
|
||||
/// router's cap-passing mount. The mount takes its own endpoint reference.
|
||||
fn systemFsMount(state: *architecture.CpuState) void {
|
||||
const prefix_ptr = architecture.systemCallArg(state, 0);
|
||||
const prefix_len = architecture.systemCallArg(state, 1);
|
||||
const backend_handle = architecture.systemCallArg(state, 2);
|
||||
const rewrite_ptr = architecture.systemCallArg(state, 3);
|
||||
const rewrite_len = architecture.systemCallArg(state, 4);
|
||||
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
||||
if (rewrite_len > 32) return fail(state);
|
||||
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
|
||||
const t = scheduler.current();
|
||||
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
|
||||
const rewrite = if (rewrite_len == 0) "" else @as([*]const u8, @ptrFromInt(rewrite_ptr))[0..rewrite_len];
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF);
|
||||
endpoint.refcount += 1; // the mount table's reference
|
||||
if (!vfs.mountBackend(prefix, endpoint, rewrite)) {
|
||||
ipc.dropRef(endpoint);
|
||||
return fail(state);
|
||||
}
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// fs_unmount(prefix_ptr, prefix_len): remove a backend mount.
|
||||
fn systemFsUnmount(state: *architecture.CpuState) void {
|
||||
const prefix_ptr = architecture.systemCallArg(state, 0);
|
||||
const prefix_len = architecture.systemCallArg(state, 1);
|
||||
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
||||
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (!vfs.unmount(prefix)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
|
||||
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
||||
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
||||
@@ -1228,37 +1437,52 @@ fn systemKlogRead(state: *architecture.CpuState) void {
|
||||
fn systemMmap(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
|
||||
if (t.address_space == 0) return fail(state); // not a user process — nothing to map into
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
|
||||
|
||||
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
|
||||
const base = t.heap_next;
|
||||
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||
// Reserve a disjoint range under a *brief* lock (the cursor is shared by every thread
|
||||
// in this address space). The mapping below then takes the lock **per page**, not for
|
||||
// the whole grant: the big lock is held with interrupts disabled, so pinning it across
|
||||
// a multi-MiB memset+map would freeze every other core on its next tick — which timed
|
||||
// the `affinity` scenario out (docs/threading-plan.md M7).
|
||||
const base = reserve: {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const cursor = scheduler.addressSpaceMmapNextPtr(t.address_space) orelse return fail(state);
|
||||
if (cursor.* == 0) cursor.* = heap_arena_base; // seed the arena lazily
|
||||
const b = cursor.*;
|
||||
if (b + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||
cursor.* = b + pages * page_size; // reserve now, so concurrent grants can't overlap
|
||||
break :reserve b;
|
||||
};
|
||||
|
||||
// Map page by page. On mid-way frame exhaustion, roll back the pages already mapped
|
||||
// (unmap + free) so no partial grant leaks into the address space — the same
|
||||
// all-or-nothing guarantee as before, but without a fixed scratch array, so the
|
||||
// per-call size can be a multi-MiB framebuffer.
|
||||
// Map the reserved range page by page, each page under a short-held lock (the range is
|
||||
// already reserved, so pages can't overlap another thread's; the lock only serializes
|
||||
// the shared page-table walk). On mid-way frame exhaustion, roll back the mapped pages
|
||||
// so no partial grant leaks — the reserved-but-unmapped tail of the arena is left
|
||||
// fallow (a rare, bounded address-space leak, not a memory leak).
|
||||
var mapped: usize = 0;
|
||||
while (mapped < pages) : (mapped += 1) {
|
||||
const flags = sync.enter();
|
||||
const frame = pmm.alloc() orelse {
|
||||
var i: usize = 0;
|
||||
while (i < mapped) : (i += 1) {
|
||||
const va = base + i * page_size;
|
||||
if (architecture.translate(t.aspace, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.aspace, va);
|
||||
if (architecture.translate(t.address_space, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.address_space, va);
|
||||
pmm.free(physical);
|
||||
}
|
||||
}
|
||||
sync.leave(flags);
|
||||
return fail(state);
|
||||
};
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||
architecture.mapUserPageInto(t.aspace, base + mapped * page_size, frame, true, false); // RW + NX
|
||||
architecture.mapUserPageInto(t.address_space, base + mapped * page_size, frame, true, false); // RW + NX
|
||||
sync.leave(flags);
|
||||
}
|
||||
t.heap_next = base + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base);
|
||||
architecture.setSystemCallResult(state, base); // the cursor was already advanced at reserve
|
||||
}
|
||||
|
||||
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
||||
@@ -1270,14 +1494,14 @@ fn systemMunmap(state: *architecture.CpuState) void {
|
||||
const base = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or base % page_size != 0) return fail(state);
|
||||
if (t.address_space == 0 or base % page_size != 0) return fail(state);
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base + i * page_size;
|
||||
if (architecture.translate(t.aspace, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.aspace, va);
|
||||
if (architecture.translate(t.address_space, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.address_space, va);
|
||||
pmm.free(physical);
|
||||
}
|
||||
}
|
||||
@@ -1341,7 +1565,7 @@ const maximum_segments = 16;
|
||||
const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
||||
|
||||
const Segment = struct {
|
||||
vaddr: u64,
|
||||
virtual_address: u64,
|
||||
memsz: u64,
|
||||
filesz: u64,
|
||||
off: u64,
|
||||
@@ -1389,7 +1613,7 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
||||
if (w and x) return error.BadSegment; // W^X, even for init
|
||||
|
||||
const seg = Segment{
|
||||
.vaddr = phdr.p_vaddr,
|
||||
.virtual_address = phdr.p_vaddr,
|
||||
.memsz = phdr.p_memsz,
|
||||
.filesz = phdr.p_filesz,
|
||||
.off = phdr.p_offset,
|
||||
@@ -1398,9 +1622,9 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
||||
};
|
||||
// No overlap with any earlier segment (page-granular, since mapping is).
|
||||
for (segs[0..count]) |other| {
|
||||
const a_end = seg.vaddr + seg.pages() * page_size;
|
||||
const b_end = other.vaddr + other.pages() * page_size;
|
||||
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
|
||||
const a_end = seg.virtual_address + seg.pages() * page_size;
|
||||
const b_end = other.virtual_address + other.pages() * page_size;
|
||||
if (seg.virtual_address < b_end and other.virtual_address < a_end) return error.BadSegment;
|
||||
}
|
||||
total_pages += seg.pages();
|
||||
if (total_pages > maximum_pages) return error.ProgramTooBig;
|
||||
@@ -1411,17 +1635,17 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
||||
|
||||
// The entry point must land inside an executable segment.
|
||||
for (segs[0..count]) |seg| {
|
||||
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
|
||||
if (seg.executable and ehdr.e_entry >= seg.virtual_address and ehdr.e_entry < seg.virtual_address + seg.memsz)
|
||||
return .{ .count = count, .entry = ehdr.e_entry };
|
||||
}
|
||||
return error.BadEntry;
|
||||
}
|
||||
|
||||
/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed
|
||||
/// Load one page of a segment into address space `address_space`: a fresh frame, zeroed
|
||||
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
||||
/// On a later failure the whole address space is torn down, which frees every
|
||||
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||
fn loadPageInto(address_space: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0);
|
||||
@@ -1430,7 +1654,7 @@ fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) I
|
||||
const n = @min(page_size, seg.filesz - page_off);
|
||||
@memcpy(destination[0..n], image[seg.off + page_off ..][0..n]);
|
||||
}
|
||||
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
||||
architecture.mapUserPageInto(address_space, seg.virtual_address + page_off, frame, seg.writable, seg.executable);
|
||||
}
|
||||
|
||||
/// Build the System V AMD64 process-entry block at the top of a process's stack
|
||||
@@ -1517,11 +1741,11 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
||||
errdefer architecture.destroyAddressSpace(aspace);
|
||||
const address_space = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
||||
errdefer architecture.destroyAddressSpace(address_space);
|
||||
|
||||
for (segs[0..parsed.count]) |seg| {
|
||||
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
||||
for (0..seg.pages()) |i| try loadPageInto(address_space, image, seg, i);
|
||||
}
|
||||
|
||||
// The stack: `user_stack_pages` zeroed pages below stack_top_virtual, RW + NX.
|
||||
@@ -1535,10 +1759,10 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
||||
const page_virtual = stack_base_virtual + i * page_size;
|
||||
if (i == parameters.user_stack_pages - 1)
|
||||
user_sp = buildEntryStack(stack_page, page_virtual, argv);
|
||||
architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX
|
||||
architecture.mapUserPageInto(address_space, page_virtual, stack_frame, true, false); // RW + NX
|
||||
}
|
||||
|
||||
const child = scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
||||
const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
||||
return error.OutOfMemory;
|
||||
// The child holds a reference to its exit endpoint from birth to death. Taken
|
||||
// only now, after nothing can fail; the lock is still held, so the child
|
||||
|
||||
+210
-58
@@ -31,7 +31,10 @@ const number_priorities = 8;
|
||||
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
|
||||
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
|
||||
|
||||
const State = enum { free, ready, running, blocked };
|
||||
// `reaping` = the task has exited and is queued on its core's reap list; its slot must not
|
||||
// be reused (freeSlot skips it) until the reaper has freed its kernel stack and set it
|
||||
// `free` (docs/threading-plan.md M8/M9).
|
||||
const State = enum { free, ready, running, blocked, reaping };
|
||||
|
||||
pub const Task = struct {
|
||||
id: u32 = 0,
|
||||
@@ -75,21 +78,25 @@ pub const Task = struct {
|
||||
ipc_wait_endpoint: ?*anyopaque = null,
|
||||
// Physical root of this task's address space, or 0 for a kernel task (which
|
||||
// runs on the shared kernel page tables). A user task carries its own.
|
||||
aspace: u64 = 0,
|
||||
address_space: u64 = 0,
|
||||
user_ip: u64 = 0, // user-mode entry point (user task only)
|
||||
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
||||
user_arg: u64 = 0, // value delivered in the user's rdi at first entry: 0 for a
|
||||
// process (its _start ignores it), the closure pointer for a thread (docs/threading.md)
|
||||
user_arg: u64 = 0, // value delivered in the user's first argument register at first entry
|
||||
// (rdi on x86_64, via architecture.jumpToUserArg): 0 for a process (its _start ignores
|
||||
// it), the closure pointer for a thread (docs/threading.md)
|
||||
// The user address this task is blocked on in futex_wait (0 = not futex-waiting).
|
||||
// Cleared to 0 by futexWakeLocked as the "woken, not timed out" signal (docs/threading.md).
|
||||
futex_addr: u64 = 0,
|
||||
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
|
||||
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
||||
// as the user heap grows; user task only.
|
||||
heap_next: u64 = 0,
|
||||
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
|
||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
||||
device_map_next: u64 = 0,
|
||||
// The task id this task is blocked in `thread_join` on (0 = not joining). Woken by
|
||||
// `wakeJoinersLocked` when that task exits (docs/threading-plan.md M9).
|
||||
join_target: u32 = 0,
|
||||
// This task's user-space TLS thread pointer — 0 until set via `set_thread_pointer`.
|
||||
// Architecture-neutral: the arch layer maps it to the FS base on x86_64, `TPIDR_EL0` on
|
||||
// aarch64. Restored on every context switch to this task (docs/threading-plan.md M10).
|
||||
thread_pointer: u64 = 0,
|
||||
// The mmap / MMIO grant-arena cursors moved from Task to the per-address-space object
|
||||
// (`AddressSpaceRef`, below) so threads sharing one address space hand out disjoint grants
|
||||
// — see addressSpaceMmapNextPtr / addressSpaceDeviceMapNextPtr (docs/threading-plan.md M7).
|
||||
// --- synchronous IPC (ipc_sync.zig) ---
|
||||
// Per-process handle table: a small-int handle names a kernel capability object.
|
||||
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
|
||||
@@ -100,13 +107,13 @@ pub const Task = struct {
|
||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||
ipc_client: ?*Task = null,
|
||||
ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS)
|
||||
ipc_send_ptr: u64 = 0, // client: outgoing message (virtual_address in this task's address space)
|
||||
ipc_send_len: u64 = 0,
|
||||
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
|
||||
ipc_reply_ptr: u64 = 0, // client: reply buffer (virtual_address)
|
||||
ipc_reply_cap: u64 = 0,
|
||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
||||
shm_map_next: u64 = 0, // bump pointer into this task's shared-memory arena (0 = unseeded)
|
||||
shared_memory_map_next: u64 = 0, // bump pointer into this task's shared-memory arena (0 = unseeded)
|
||||
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
||||
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||
@@ -148,17 +155,57 @@ var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||
/// only when the **last** task on an address space exits. All access is under the big
|
||||
/// kernel lock. There can be no more live address spaces than tasks, so the table is
|
||||
/// sized to the task pool and never overflows in practice.
|
||||
const AspaceRef = struct { root: u64 = 0, count: u32 = 0 };
|
||||
var aspace_refs = [_]AspaceRef{.{}} ** maximum_tasks;
|
||||
var aspace_destroy_count: u64 = 0;
|
||||
// The per-address-space kernel object: a reference count plus the grant-arena cursors.
|
||||
// One live entry per address space; threads sharing an address space share this entry,
|
||||
// so their mmap/mmio grants bump one cursor and never overlap (docs/threading-plan.md M7).
|
||||
// `mmap_next`/`device_map_next` are 0 until process.zig seeds them to the arena base.
|
||||
const AddressSpaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0 };
|
||||
var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks;
|
||||
var address_space_destroy_count: u64 = 0;
|
||||
|
||||
/// Total bytes of task **kernel** stacks currently allocated from the kernel heap —
|
||||
/// incremented when a task is created, decremented when the reaper frees a dead task's
|
||||
/// stack. A test-observable proof that the reaper reclaims every stack (docs/threading-
|
||||
/// plan.md M8): with no live tasks beyond the baseline, this returns to its baseline.
|
||||
var live_stack_bytes: usize = 0;
|
||||
|
||||
/// Test-observable: bytes of task kernel stacks currently live (see `live_stack_bytes`).
|
||||
pub fn liveStackBytes() usize {
|
||||
return live_stack_bytes;
|
||||
}
|
||||
|
||||
/// Free a dead task's kernel stack and drop it from `live_stack_bytes`. The task must be
|
||||
/// off that stack already (killed while not running, or reaped after it switched away).
|
||||
/// Caller holds the kernel lock.
|
||||
fn reapStackLocked(t: *Task) void {
|
||||
if (t.stack.len == 0) return; // boot/idle tasks run on a static stack — nothing to free
|
||||
live_stack_bytes -= t.stack.len;
|
||||
heap.allocator().free(t.stack);
|
||||
t.stack = &.{};
|
||||
t.kstack_top = 0;
|
||||
}
|
||||
|
||||
/// Free every `.reaping` task queued on this core's reap list and mark each `.free` (now
|
||||
/// its slot may be reused). The tasks are all off their stacks (they switched away), and
|
||||
/// the caller holds the lock, so freeing is safe (docs/threading-plan.md M8/M9).
|
||||
fn drainReapListLocked(pc: *PerCpu) void {
|
||||
var node = pc.reap_list;
|
||||
pc.reap_list = null;
|
||||
while (node) |t| {
|
||||
node = t.next; // save the link before we clear it
|
||||
t.next = null;
|
||||
reapStackLocked(t);
|
||||
t.state = .free; // reusable only now, after the stack is freed
|
||||
}
|
||||
}
|
||||
|
||||
/// Take a reference to address space `root` (0 = a kernel task, which owns none).
|
||||
/// Returns false only if the ref table is full — bounded by `maximum_tasks`, so in
|
||||
/// practice it never is. Caller holds the kernel lock.
|
||||
fn retainAspace(root: u64) bool {
|
||||
fn retainAddressSpace(root: u64) bool {
|
||||
if (root == 0) return true;
|
||||
var free: ?*AspaceRef = null;
|
||||
for (&aspace_refs) |*entry| {
|
||||
var free: ?*AddressSpaceRef = null;
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) {
|
||||
entry.count += 1;
|
||||
return true;
|
||||
@@ -173,34 +220,54 @@ fn retainAspace(root: u64) bool {
|
||||
/// Drop a reference to `root`; destroy the address space when the **last** one drops.
|
||||
/// A `root` with no entry — never retained, e.g. a hand-built test space — is
|
||||
/// destroyed directly, preserving the pre-refcount behaviour. Caller holds the lock.
|
||||
fn releaseAspace(root: u64) void {
|
||||
fn releaseAddressSpace(root: u64) void {
|
||||
if (root == 0) return;
|
||||
for (&aspace_refs) |*entry| {
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count == 0 or entry.root != root) continue;
|
||||
entry.count -= 1;
|
||||
if (entry.count == 0) {
|
||||
entry.root = 0;
|
||||
architecture.destroyAddressSpace(root);
|
||||
aspace_destroy_count += 1;
|
||||
address_space_destroy_count += 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
architecture.destroyAddressSpace(root);
|
||||
aspace_destroy_count += 1;
|
||||
address_space_destroy_count += 1;
|
||||
}
|
||||
|
||||
/// Test-observable: how many address spaces are live (entries with a nonzero count).
|
||||
pub fn liveAspaceCount() u32 {
|
||||
pub fn liveAddressSpaceCount() u32 {
|
||||
var live: u32 = 0;
|
||||
for (&aspace_refs) |*entry| {
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0) live += 1;
|
||||
}
|
||||
return live;
|
||||
}
|
||||
|
||||
/// Test-observable: total address-space destructions since boot.
|
||||
pub fn aspaceDestroyCount() u64 {
|
||||
return aspace_destroy_count;
|
||||
pub fn addressSpaceDestroyCount() u64 {
|
||||
return address_space_destroy_count;
|
||||
}
|
||||
|
||||
/// Pointer to the mmap grant-arena cursor for address space `root`, so the mmap syscall
|
||||
/// can read-and-bump it. Per-address-space (not per-task), so sibling threads get
|
||||
/// disjoint grants. **Caller holds the kernel lock** (the entry is stable while held).
|
||||
/// Null only if `root` was never retained — which can't happen for a live user task.
|
||||
pub fn addressSpaceMmapNextPtr(root: u64) ?*u64 {
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) return &entry.mmap_next;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Pointer to the MMIO grant-arena cursor for address space `root` (see
|
||||
/// `addressSpaceMmapNextPtr`). Caller holds the kernel lock.
|
||||
pub fn addressSpaceDeviceMapNextPtr(root: u64) ?*u64 {
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) return &entry.device_map_next;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
var next_id: u32 = 1;
|
||||
|
||||
@@ -221,11 +288,18 @@ pub const PerCpu = struct {
|
||||
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
|
||||
index: u32 = 0, // dense 0-based core index
|
||||
online: bool = false, // has this core finished bring-up?
|
||||
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
||||
loaded_address_space: u64 = 0, // the address-space root currently loaded on this core
|
||||
loaded_thread_pointer: u64 = 0, // the TLS thread pointer currently loaded on this core (docs/threading-plan.md M10)
|
||||
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||
pinned_bitmap: u8 = 0,
|
||||
// Tasks that ended while running on THIS core: they could not free the kernel stack
|
||||
// they were standing on, so each pushed itself onto this list (`.reaping` state, linked
|
||||
// via `Task.next`) and switched away. The next task to run on this core — or the timer
|
||||
// tick — frees their stacks from its own stack, safely (docs/threading-plan.md M8). A
|
||||
// *list* (not one slot) so a second death before the first is drained can't lose it.
|
||||
reap_list: ?*Task = null,
|
||||
};
|
||||
|
||||
const maximum_cpus = parameters.maximum_cpus;
|
||||
@@ -257,7 +331,7 @@ var preemption_enabled = true;
|
||||
/// boot, before interrupts are enabled — so no lock is needed here.
|
||||
pub fn init(boot_priority: Priority) void {
|
||||
const pc = &cpus[0];
|
||||
pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() };
|
||||
pc.* = .{ .index = 0, .online = true, .loaded_address_space = architecture.kernelPageTable() };
|
||||
architecture.setCpuLocal(0, @intFromPtr(pc));
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
pc.current = &tasks[0];
|
||||
@@ -296,7 +370,7 @@ pub fn secondaryMain() callconv(.c) noreturn {
|
||||
pc.current = t;
|
||||
pc.idle = t;
|
||||
pc.online = true;
|
||||
pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||
pc.loaded_address_space = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||
sync.leave(flags);
|
||||
|
||||
architecture.enableInterrupts(); // the timer now preempts this idle context into work
|
||||
@@ -382,7 +456,7 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
return ok;
|
||||
}
|
||||
|
||||
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
||||
/// Spawn a **user** task: a task with its own address space (`address_space`) that starts
|
||||
/// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]).
|
||||
/// `supervisor` is the id of the spawning process (0 = the kernel) — the kill
|
||||
/// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller
|
||||
@@ -391,23 +465,24 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
/// lands in `user_task_trampoline`.
|
||||
/// Returns the new process id, or null (creating nothing) if the table is full or
|
||||
/// out of memory.
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `address_space` holds it
|
||||
/// across the whole spawn, so the address space and the task appear atomically).
|
||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
||||
pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
||||
const t = freeSlot() orelse return null;
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch return null;
|
||||
// Take this task's reference to the address space before we commit the slot, so a
|
||||
// failure here leaves nothing to unwind (the caller still owns the raw `aspace`).
|
||||
if (!retainAspace(aspace)) {
|
||||
// failure here leaves nothing to unwind (the caller still owns the raw `address_space`).
|
||||
if (!retainAddressSpace(address_space)) {
|
||||
heap.allocator().free(stack);
|
||||
return null;
|
||||
}
|
||||
live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8)
|
||||
t.* = .{
|
||||
.id = next_id,
|
||||
.state = .ready,
|
||||
.priority = priority,
|
||||
.stack = stack,
|
||||
.aspace = aspace,
|
||||
.address_space = address_space,
|
||||
.user_ip = entry,
|
||||
.user_sp = user_sp,
|
||||
.user_arg = user_arg,
|
||||
@@ -445,6 +520,7 @@ fn startUserTask() void {
|
||||
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
||||
const t = freeSlot() orelse @panic("sched: task table full");
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||
live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8)
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
@@ -485,7 +561,7 @@ fn schedule() void {
|
||||
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
||||
/// user-mode interrupt lands on a good stack) and its address space (only when
|
||||
/// it differs from what's loaded — every page-table switch is a full TLB flush),
|
||||
/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used
|
||||
/// then switch registers/stacks. Kernel tasks (address_space == 0, no kstack_top used
|
||||
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
|
||||
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
|
||||
/// workload. The big kernel lock is held and interrupts are off throughout, so no
|
||||
@@ -493,12 +569,25 @@ fn schedule() void {
|
||||
/// `save_sp` receives the outgoing task's stack pointer.
|
||||
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
||||
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
|
||||
const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable();
|
||||
if (want != pc.loaded_aspace) {
|
||||
const want = if (next.address_space != 0) next.address_space else architecture.kernelPageTable();
|
||||
if (want != pc.loaded_address_space) {
|
||||
architecture.loadPageTable(want);
|
||||
pc.loaded_aspace = want;
|
||||
pc.loaded_address_space = want;
|
||||
}
|
||||
// Restore the next task's user TLS thread pointer — only on change, the same
|
||||
// conditional-load discipline as CR3 above (docs/threading-plan.md M10).
|
||||
if (next.thread_pointer != pc.loaded_thread_pointer) {
|
||||
architecture.setThreadPointer(next.thread_pointer);
|
||||
pc.loaded_thread_pointer = next.thread_pointer;
|
||||
}
|
||||
architecture.switchContext(save_sp, next.sp);
|
||||
// Resumed now (switchContext returned into our own switchTo frame). Re-fetch the core
|
||||
// via thisCpu(): the `pc` parameter is from *our* earlier switchTo call, so it names
|
||||
// the core we last ran on — stale if we migrated. switchContext only swaps stacks on
|
||||
// the current core, so thisCpu() is the core the just-dead task died on. If a task
|
||||
// died switching to us, free its kernel stack: we're on ours so it's safe, and the big
|
||||
// lock is still held so its slot can't have been reused (docs/threading-plan.md M8).
|
||||
drainReapListLocked(thisCpu());
|
||||
}
|
||||
|
||||
/// Voluntarily give up the CPU to the next ready task.
|
||||
@@ -546,12 +635,12 @@ pub fn futexWaitLocked(addr: u64, timeout_ms: u64) FutexResult {
|
||||
}
|
||||
|
||||
/// Wake up to `count` tasks blocked in `futex_wait` on `addr` in address space
|
||||
/// `aspace`. Precondition: the big kernel lock is held. Returns how many woke.
|
||||
pub fn futexWakeLocked(aspace: u64, addr: u64, count: u32) u32 {
|
||||
/// `address_space`. Precondition: the big kernel lock is held. Returns how many woke.
|
||||
pub fn futexWakeLocked(address_space: u64, addr: u64, count: u32) u32 {
|
||||
var woken: u32 = 0;
|
||||
for (&tasks) |*t| {
|
||||
if (woken >= count) break;
|
||||
if (t.state == .blocked and t.aspace == aspace and t.futex_addr == addr) {
|
||||
if (t.state == .blocked and t.address_space == address_space and t.futex_addr == addr) {
|
||||
t.futex_addr = 0; // the "woken, not timed out" signal to futexWaitLocked
|
||||
t.wake_at = 0;
|
||||
t.state = .ready;
|
||||
@@ -562,6 +651,55 @@ pub fn futexWakeLocked(aspace: u64, addr: u64, count: u32) u32 {
|
||||
return woken;
|
||||
}
|
||||
|
||||
// --- thread join (docs/threading-plan.md M9) --------------------------------
|
||||
//
|
||||
// join needs no per-thread IPC endpoint: `thread_join(tid)` blocks the caller until the
|
||||
// task with id `tid` has exited, and the exit paths wake any joiner. The caller only ever
|
||||
// reclaims the joined thread's *user* stack (which the thread vacated the moment it entered
|
||||
// the kernel to exit), so waking at exit time — not reap time — is safe.
|
||||
|
||||
/// True if a task with id `tid` is still live (has not exited). Caller holds the lock.
|
||||
fn aliveTid(tid: u32) bool {
|
||||
for (&tasks) |*t| {
|
||||
if (t.id == tid and t.state != .free and t.state != .reaping) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Block the current task until the task with id `tid` exits (or return at once if it
|
||||
/// already has / never existed). **Precondition:** the big kernel lock is held; returns
|
||||
/// with it still held. Woken by `wakeJoinersLocked`.
|
||||
pub fn joinThreadLocked(tid: u32) void {
|
||||
while (aliveTid(tid)) {
|
||||
const t = current();
|
||||
t.join_target = tid;
|
||||
t.state = .blocked;
|
||||
schedule(); // woken when the joined task exits; lock handed off across the switch
|
||||
t.join_target = 0;
|
||||
}
|
||||
}
|
||||
|
||||
/// Set the calling task's user TLS thread pointer and load it now. Persisted on the Task so
|
||||
/// context switches restore it (docs/threading-plan.md M10). Caller holds the kernel lock.
|
||||
pub fn setThreadPointerLocked(addr: u64) void {
|
||||
const pc = thisCpu();
|
||||
pc.current.thread_pointer = addr;
|
||||
architecture.setThreadPointer(addr);
|
||||
pc.loaded_thread_pointer = addr;
|
||||
}
|
||||
|
||||
/// Wake every task blocked in `thread_join` on `tid` — called from the exit paths once the
|
||||
/// exiting task's state is `.free`. Caller holds the lock.
|
||||
fn wakeJoinersLocked(tid: u32) void {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .blocked and t.join_target == tid) {
|
||||
t.join_target = 0;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- event-based blocking -------------------------------------------------
|
||||
//
|
||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
||||
@@ -663,7 +801,7 @@ fn removeFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task
|
||||
/// Precondition: the big kernel lock is held.
|
||||
pub fn taskByIdLocked(id: u32) ?*Task {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state != .free and t.id == id) return t;
|
||||
if (t.state != .free and t.state != .reaping and t.id == id) return t;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -673,7 +811,7 @@ pub fn taskByIdLocked(id: u32) ?*Task {
|
||||
/// Precondition: the big kernel lock is held.
|
||||
pub fn forgetIpcClientLocked(t: *Task) void {
|
||||
for (&tasks) |*other| {
|
||||
if (other.state != .free and other.ipc_client == t) other.ipc_client = null;
|
||||
if (other.state != .free and other.state != .reaping and other.ipc_client == t) other.ipc_client = null;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -765,7 +903,11 @@ pub var reap_task_hook: ?*const fn (*Task) void = null;
|
||||
fn reapKillPendingLocked() void {
|
||||
const pc = thisCpu();
|
||||
const cur = pc.current;
|
||||
if (cur.kill_pending and cur.aspace != 0 and !cur.in_system_call) {
|
||||
// Safety net: if a dying task switched to a *fresh* task (which enters via
|
||||
// task_trampoline, not switchTo's tail), its stack is still queued here. The dying
|
||||
// task switched away before this tick, so it is off its stack — drain now (M8).
|
||||
drainReapListLocked(pc);
|
||||
if (cur.kill_pending and cur.address_space != 0 and !cur.in_system_call) {
|
||||
if (terminate_current_hook) |hook| hook(); // noreturn
|
||||
}
|
||||
if (reap_task_hook) |hook| {
|
||||
@@ -806,7 +948,11 @@ pub fn setPreemption(enabled: bool) void {
|
||||
pub fn exit() noreturn {
|
||||
_ = sync.enter();
|
||||
const pc = thisCpu();
|
||||
pc.current.state = .free;
|
||||
// Queue this task for reaping: `.reaping` keeps its slot out of freeSlot until its
|
||||
// stack is freed; `next` links it on the core's reap list (M8/M9).
|
||||
pc.current.state = .reaping;
|
||||
pc.current.next = pc.reap_list;
|
||||
pc.reap_list = pc.current;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
@@ -831,17 +977,21 @@ pub fn exitUser() noreturn {
|
||||
pub fn exitUserLocked() noreturn {
|
||||
const pc = thisCpu();
|
||||
const dying = pc.current;
|
||||
const as = dying.aspace;
|
||||
const as = dying.address_space;
|
||||
if (as != 0) {
|
||||
const kroot = architecture.kernelPageTable();
|
||||
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
||||
pc.loaded_aspace = kroot;
|
||||
releaseAspace(as); // destroys only when this was the last task on the space
|
||||
pc.loaded_address_space = kroot;
|
||||
releaseAddressSpace(as); // destroys only when this was the last task on the space
|
||||
}
|
||||
dying.state = .free;
|
||||
dying.aspace = 0;
|
||||
dying.state = .reaping; // dead but its slot stays reserved until the stack is freed
|
||||
wakeJoinersLocked(dying.id); // let any thread_join(dying.id) return (M9)
|
||||
dying.address_space = 0;
|
||||
dying.kill_pending = false;
|
||||
dying.in_system_call = false;
|
||||
// Queue for reaping: the task we switch to (or the next tick) frees this stack (M8/M9).
|
||||
dying.next = pc.reap_list;
|
||||
pc.reap_list = dying;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
@@ -857,12 +1007,14 @@ pub fn exitUserLocked() noreturn {
|
||||
/// task isn't running). The kernel stack is leaked, as in `exitUser` (no reaper
|
||||
/// yet). Precondition: the big kernel lock is held.
|
||||
pub fn destroyTaskLocked(t: *Task) void {
|
||||
if (t.aspace != 0) releaseAspace(t.aspace); // destroys only on the last reference
|
||||
t.aspace = 0;
|
||||
if (t.address_space != 0) releaseAddressSpace(t.address_space); // destroys only on the last reference
|
||||
reapStackLocked(t); // safe to free now: `t` is not running on any core (M8)
|
||||
t.address_space = 0;
|
||||
t.kill_pending = false;
|
||||
t.in_system_call = false;
|
||||
t.wake_at = 0;
|
||||
t.state = .free;
|
||||
wakeJoinersLocked(t.id); // a killed thread's joiners must return too (M9)
|
||||
}
|
||||
|
||||
/// Snapshot the task table into `out` (up to its length), returning the total
|
||||
@@ -876,7 +1028,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||
defer sync.leave(flags);
|
||||
var total: u64 = 0;
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .free) continue;
|
||||
if (t.state == .free or t.state == .reaping) continue; // reaping = already exited
|
||||
if (total < out.len) {
|
||||
const d = &out[total];
|
||||
d.* = .{
|
||||
@@ -886,7 +1038,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||
.ready => .ready,
|
||||
.running => .running,
|
||||
.blocked => .blocked,
|
||||
.free => unreachable,
|
||||
.free, .reaping => unreachable,
|
||||
})),
|
||||
.priority = t.priority,
|
||||
.name_length = t.name_length,
|
||||
@@ -900,7 +1052,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||
|
||||
/// Whether the running task is a user process (has its own address space).
|
||||
pub fn currentIsUserProcess() bool {
|
||||
return current().aspace != 0;
|
||||
return current().address_space != 0;
|
||||
}
|
||||
|
||||
pub fn currentId() u32 {
|
||||
|
||||
+476
-134
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,368 @@
|
||||
//! The kernel-resident VFS root: the mount table and the kernel-backed nodes.
|
||||
//!
|
||||
//! The kernel's job here is NAMING, never data plumbing to userspace backends —
|
||||
//! the mechanism is **resolve + redirect**:
|
||||
//!
|
||||
//! - `fs_resolve(path)` walks the mount table. A path under a KERNEL-backed
|
||||
//! mount (the initrd at /system, the scratch ram nodes) resolves to a
|
||||
//! stateless node TOKEN served directly by `fs_node` (read/status/readdir
|
||||
//! with copy-out). A path under a USERSPACE mount (the fat server at
|
||||
//! /mnt/usb and /var) resolves to the backend's ENDPOINT: the kernel
|
||||
//! installs a (deduplicated) handle in the caller's table, rewrites the
|
||||
//! path mount-relative, and the caller speaks the unchanged vfs-protocol
|
||||
//! to the backend over the ordinary ipc_call rendezvous. The kernel never
|
||||
//! blocks on a userspace server.
|
||||
//!
|
||||
//! - Kernel node tokens are PERMANENT for a boot: the initrd is immutable and
|
||||
//! ram nodes are never reclaimed — no open-handle state, no close, no sweep
|
||||
//! on client death. Backend file state lives in the backend, which sweeps
|
||||
//! dead clients itself via the published exit events.
|
||||
//!
|
||||
//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the
|
||||
//! backend endpoint handle is the capability, exactly the trust of the old
|
||||
//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount
|
||||
//! into the backend's namespace ("/var" -> fat's "/var" subtree while the same
|
||||
//! backend also serves "/mnt/usb" from its root), so FHS paths stay decoupled
|
||||
//! from which volume happens to carry them.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const ipc = @import("ipc-synchronous.zig");
|
||||
|
||||
// --- node tokens -------------------------------------------------------------
|
||||
|
||||
/// Kind lives in the top byte of a token; the index below. Tokens are permanent
|
||||
/// for a boot, so userspace may cache them freely.
|
||||
pub const token_kind_shift = 56;
|
||||
pub const token_kind_initrd_file: u64 = 1;
|
||||
pub const token_kind_initrd_directory: u64 = 2;
|
||||
pub const token_kind_ram: u64 = 3;
|
||||
|
||||
fn token(kind: u64, index: u64) u64 {
|
||||
return (kind << token_kind_shift) | index;
|
||||
}
|
||||
|
||||
fn tokenKind(t: u64) u64 {
|
||||
return t >> token_kind_shift;
|
||||
}
|
||||
|
||||
fn tokenIndex(t: u64) u64 {
|
||||
return t & ((@as(u64, 1) << token_kind_shift) - 1);
|
||||
}
|
||||
|
||||
// --- the mount table ---------------------------------------------------------
|
||||
|
||||
pub const maximum_mounts = 8;
|
||||
const maximum_prefix = 64;
|
||||
const maximum_rewrite = 32;
|
||||
|
||||
const MountKind = enum(u8) { kernel_initrd, backend };
|
||||
|
||||
const Mount = struct {
|
||||
used: bool = false,
|
||||
prefix: [maximum_prefix]u8 = undefined,
|
||||
prefix_len: usize = 0,
|
||||
kind: MountKind = .backend,
|
||||
backend: ?*ipc.Endpoint = null, // referenced while mounted
|
||||
rewrite: [maximum_rewrite]u8 = undefined,
|
||||
rewrite_len: usize = 0,
|
||||
|
||||
fn prefixSlice(self: *const Mount) []const u8 {
|
||||
return self.prefix[0..self.prefix_len];
|
||||
}
|
||||
fn rewriteSlice(self: *const Mount) []const u8 {
|
||||
return self.rewrite[0..self.rewrite_len];
|
||||
}
|
||||
};
|
||||
|
||||
var mounts: [maximum_mounts]Mount = @splat(.{});
|
||||
|
||||
/// The initrd image (set once at boot) and its derived directory table.
|
||||
var ramdisk_image: ?[]const u8 = null;
|
||||
|
||||
const maximum_directories = 8;
|
||||
const Directory = struct {
|
||||
path: [maximum_prefix]u8 = undefined,
|
||||
path_len: usize = 0,
|
||||
parent: usize = 0, // index into `directories`; 0 is /system itself
|
||||
|
||||
fn slice(self: *const Directory) []const u8 {
|
||||
return self.path[0..self.path_len];
|
||||
}
|
||||
};
|
||||
var directories: [maximum_directories]Directory = @splat(.{});
|
||||
var directory_count: usize = 0;
|
||||
|
||||
// --- pure path helpers (ported from the userspace router, with its tests) ----
|
||||
|
||||
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by
|
||||
/// a path separator — return the path relative to the mount ("/" for an exact
|
||||
/// match, otherwise the tail beginning with '/'). Null when not under the
|
||||
/// mount, so "/mnt/usb" never captures "/mnt/usbextra".
|
||||
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
||||
if (path.len < mount_prefix.len) return null;
|
||||
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
||||
if (path.len == mount_prefix.len) return "/";
|
||||
if (path[mount_prefix.len] != '/') return null;
|
||||
return path[mount_prefix.len..];
|
||||
}
|
||||
|
||||
pub fn isAbsolute(path: []const u8) bool {
|
||||
return path.len > 0 and path[0] == '/';
|
||||
}
|
||||
|
||||
/// The parent directory portion of an initrd path ("/system/services/fat" ->
|
||||
/// "/system/services").
|
||||
fn parentOf(path: []const u8) []const u8 {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/') orelse return path[0..0];
|
||||
if (slash == 0) return path[0..1];
|
||||
return path[0..slash];
|
||||
}
|
||||
|
||||
// --- boot wiring -------------------------------------------------------------
|
||||
|
||||
/// Publish the initrd as the kernel-backed /system mount and derive its bounded
|
||||
/// directory table (the unique parents of the entry paths). Called once at boot.
|
||||
pub fn setInitialRamdisk(image: []const u8) void {
|
||||
ramdisk_image = image;
|
||||
installMount("/system", .kernel_initrd, null, "");
|
||||
|
||||
// Directory 0 is /system itself.
|
||||
directories[0] = .{ .parent = 0 };
|
||||
@memcpy(directories[0].path[0..7], "/system");
|
||||
directories[0].path_len = 7;
|
||||
directory_count = 1;
|
||||
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
// Register every ancestor directory strictly below /system.
|
||||
var parent = parentOf(item.name);
|
||||
while (parent.len > 7) : (parent = parentOf(parent)) {
|
||||
if (directoryIndex(parent) == null and directory_count < maximum_directories) {
|
||||
var d = &directories[directory_count];
|
||||
@memcpy(d.path[0..parent.len], parent);
|
||||
d.path_len = parent.len;
|
||||
d.parent = 0; // fixed up below once all exist
|
||||
directory_count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Parent links (a second pass so out-of-order registration doesn't matter).
|
||||
for (directories[1..directory_count]) |*d| {
|
||||
d.parent = directoryIndex(parentOf(d.slice())) orelse 0;
|
||||
}
|
||||
}
|
||||
|
||||
fn directoryIndex(path: []const u8) ?usize {
|
||||
for (directories[0..directory_count], 0..) |*d, i| {
|
||||
if (std.mem.eql(u8, d.slice(), path)) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void {
|
||||
// Remount replaces: a restarted backend re-mounts its prefix.
|
||||
var slot: ?*Mount = null;
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||
if (m.backend) |old| ipc.dropRef(old);
|
||||
slot = m;
|
||||
break;
|
||||
}
|
||||
if (slot == null and !m.used) slot = m;
|
||||
}
|
||||
const m = slot orelse return;
|
||||
m.* = .{ .used = true, .kind = kind, .backend = backend };
|
||||
@memcpy(m.prefix[0..prefix.len], prefix);
|
||||
m.prefix_len = prefix.len;
|
||||
@memcpy(m.rewrite[0..rewrite.len], rewrite);
|
||||
m.rewrite_len = rewrite.len;
|
||||
}
|
||||
|
||||
// --- resolve -----------------------------------------------------------------
|
||||
|
||||
pub const Resolved = union(enum) {
|
||||
/// Kernel-served: a permanent node token.
|
||||
kernel_node: u64,
|
||||
/// Backend-served: the endpoint plus the rewritten mount-relative path.
|
||||
backend: struct { endpoint: *ipc.Endpoint, path: [maximum_rewrite + maximum_prefix + 160]u8, path_len: usize },
|
||||
not_found: void,
|
||||
};
|
||||
|
||||
/// Longest-prefix match over the mount table, then per-kind resolution.
|
||||
/// `create`-intent on the immutable /system fails here (EROFS-style).
|
||||
pub fn resolvePath(path: []const u8, wants_create: bool) Resolved {
|
||||
if (!isAbsolute(path)) {
|
||||
return .{ .not_found = {} }; // bare names have no kernel namespace (ramfs retired)
|
||||
}
|
||||
var best: ?*Mount = null;
|
||||
var best_relative: []const u8 = undefined;
|
||||
for (&mounts) |*m| {
|
||||
if (!m.used) continue;
|
||||
const relative = underMount(path, m.prefixSlice()) orelse continue;
|
||||
if (best == null or m.prefix_len > best.?.prefix_len) {
|
||||
best = m;
|
||||
best_relative = relative;
|
||||
}
|
||||
}
|
||||
const m = best orelse return .{ .not_found = {} };
|
||||
switch (m.kind) {
|
||||
.kernel_initrd => {
|
||||
if (wants_create) return .{ .not_found = {} }; // read-only
|
||||
return resolveInitrd(path);
|
||||
},
|
||||
.backend => {
|
||||
const endpoint = m.backend orelse return .{ .not_found = {} };
|
||||
if (endpoint.dead) {
|
||||
// The backend died: treat the mount as gone (it re-mounts on
|
||||
// restart) and release our reference lazily.
|
||||
ipc.dropRef(endpoint);
|
||||
m.backend = null;
|
||||
m.used = false;
|
||||
return .{ .not_found = {} };
|
||||
}
|
||||
var out: Resolved = .{ .backend = .{ .endpoint = endpoint, .path = undefined, .path_len = 0 } };
|
||||
const rewrite = m.rewriteSlice();
|
||||
const tail = if (std.mem.eql(u8, best_relative, "/") and rewrite.len != 0) "" else best_relative;
|
||||
const total = rewrite.len + tail.len;
|
||||
if (total > out.backend.path.len or total == 0) {
|
||||
if (rewrite.len == 0 and tail.len == 0) return .{ .not_found = {} };
|
||||
if (total > out.backend.path.len) return .{ .not_found = {} };
|
||||
}
|
||||
@memcpy(out.backend.path[0..rewrite.len], rewrite);
|
||||
@memcpy(out.backend.path[rewrite.len..][0..tail.len], tail);
|
||||
out.backend.path_len = total;
|
||||
return out;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn resolveInitrd(path: []const u8) Resolved {
|
||||
if (directoryIndex(path)) |index| return .{ .kernel_node = token(token_kind_initrd_directory, index) };
|
||||
const image = ramdisk_image orelse return .{ .not_found = {} };
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return .{ .not_found = {} };
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (std.mem.eql(u8, item.name, path)) return .{ .kernel_node = token(token_kind_initrd_file, i) };
|
||||
}
|
||||
return .{ .not_found = {} };
|
||||
}
|
||||
|
||||
// --- fs_node: serving kernel-backed nodes ------------------------------------
|
||||
|
||||
/// Read `out.len` bytes of an initrd file at `offset`. Returns bytes copied
|
||||
/// (0 at EOF) or null for a bad token. Lock-free: the initrd is immutable.
|
||||
pub fn nodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
||||
if (tokenKind(node_token) != token_kind_initrd_file) return null;
|
||||
const image = ramdisk_image orelse return null;
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
||||
const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null;
|
||||
if (offset >= item.blob.len) return 0;
|
||||
const n = @min(out.len, item.blob.len - @as(usize, @intCast(offset)));
|
||||
@memcpy(out[0..n], item.blob[@intCast(offset)..][0..n]);
|
||||
return n;
|
||||
}
|
||||
|
||||
/// A node's metadata in vfs-protocol FileStatus shape (size, kind, mtime).
|
||||
pub fn nodeStatus(node_token: u64) ?abi.FileAttributes {
|
||||
switch (tokenKind(node_token)) {
|
||||
token_kind_initrd_file => {
|
||||
const image = ramdisk_image orelse return null;
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
||||
const item = rd.entry(@intCast(tokenIndex(node_token))) orelse return null;
|
||||
return .{ .size = item.blob.len, .kind = abi.file_kind_regular };
|
||||
},
|
||||
token_kind_initrd_directory => {
|
||||
if (tokenIndex(node_token) >= directory_count) return null;
|
||||
return .{ .size = 0, .kind = abi.file_kind_directory };
|
||||
},
|
||||
else => return null,
|
||||
}
|
||||
}
|
||||
|
||||
/// The `cursor`th child of an initrd directory: fills `name_out`, returns the
|
||||
/// entry header, or null past the end / bad token. Cursor enumerates
|
||||
/// subdirectories first, then files whose parent is this directory — stable,
|
||||
/// because the initrd is immutable.
|
||||
pub fn nodeReaddir(node_token: u64, cursor: u64, name_out: []u8) ?struct { header: abi.DirectoryEntryHeader, name_len: usize } {
|
||||
if (tokenKind(node_token) != token_kind_initrd_directory) return null;
|
||||
const directory_index = tokenIndex(node_token);
|
||||
if (directory_index >= directory_count) return null;
|
||||
const self_path = directories[@intCast(directory_index)].slice();
|
||||
|
||||
var index: u64 = 0;
|
||||
// Subdirectories whose parent is this directory.
|
||||
for (directories[0..directory_count], 0..) |*d, i| {
|
||||
if (i == directory_index) continue;
|
||||
if (d.parent != directory_index) continue;
|
||||
if (i == 0) continue;
|
||||
if (index == cursor) {
|
||||
const name = d.slice()[self_path.len + 1 ..];
|
||||
const n = @min(name.len, name_out.len);
|
||||
@memcpy(name_out[0..n], name[0..n]);
|
||||
return .{ .header = .{ .kind = abi.file_kind_directory, .name_len = @intCast(n), .size = 0 }, .name_len = n };
|
||||
}
|
||||
index += 1;
|
||||
}
|
||||
// Files directly inside this directory.
|
||||
const image = ramdisk_image orelse return null;
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return null;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!std.mem.eql(u8, parentOf(item.name), self_path)) continue;
|
||||
if (index == cursor) {
|
||||
const name = item.name[self_path.len + 1 ..];
|
||||
const n = @min(name.len, name_out.len);
|
||||
@memcpy(name_out[0..n], name[0..n]);
|
||||
return .{ .header = .{ .kind = abi.file_kind_regular, .name_len = @intCast(n), .size = item.blob.len }, .name_len = n };
|
||||
}
|
||||
index += 1;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- mount/unmount (syscall bodies; caller resolved the handle) --------------
|
||||
|
||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
||||
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
||||
/// shadowing or replacing /system.
|
||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
||||
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
||||
if (rewrite.len > maximum_rewrite) return false;
|
||||
if (underMount(prefix, "/system") != null) return false; // the initrd is not shadowable
|
||||
installMount(prefix, .backend, backend, rewrite);
|
||||
return true;
|
||||
}
|
||||
|
||||
pub fn unmount(prefix: []const u8) bool {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||
if (m.backend) |endpoint| ipc.dropRef(endpoint);
|
||||
m.* = .{};
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// --- tests (host) ------------------------------------------------------------
|
||||
|
||||
test "underMount matches only at path boundaries" {
|
||||
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
|
||||
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
|
||||
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null);
|
||||
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null);
|
||||
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
|
||||
try std.testing.expect(underMount("greeting", "/mnt/usb") == null);
|
||||
}
|
||||
|
||||
test "parentOf walks toward the root" {
|
||||
try std.testing.expectEqualStrings("/system/services", parentOf("/system/services/fat"));
|
||||
try std.testing.expectEqualStrings("/system", parentOf("/system/services"));
|
||||
try std.testing.expectEqualStrings("/", parentOf("/system"));
|
||||
}
|
||||
@@ -24,3 +24,9 @@ pub fn init() void {
|
||||
pub fn nowSeconds() u64 {
|
||||
return boot_unix_seconds + (architecture.nanos() -% boot_nanos) / 1_000_000_000;
|
||||
}
|
||||
|
||||
/// The wall-clock time of boot itself (the RTC anchor) — what klog_status hands
|
||||
/// the logger service to name a per-boot log directory. Zero until `init` runs.
|
||||
pub fn bootSeconds() u64 {
|
||||
return boot_unix_seconds;
|
||||
}
|
||||
|
||||
@@ -21,11 +21,6 @@ const power = runtime.power_protocol;
|
||||
/// integer decode names the opcodes instead of bare 0x0A/0x0B/… (docs/coding-standards.md).
|
||||
const opcodes = aml.opcodes;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// The claimed acpi-tables node and the resource index of its broad io_port
|
||||
// window — the Hal routes every port access through this one claim.
|
||||
var node_id: u64 = 0;
|
||||
@@ -163,12 +158,12 @@ pub fn main(init: runtime.process.Init) void {
|
||||
};
|
||||
var namespace = result.namespace;
|
||||
const devices = aml.deviceCount(&namespace);
|
||||
writeLine("/system/services/acpi: parsed {d} AML blob(s), {d} namespace devices\n", .{ block_count, devices });
|
||||
std.log.info("parsed {d} AML blob(s), {d} namespace devices", .{ block_count, devices });
|
||||
if (floor) |minimum| {
|
||||
if (devices >= minimum) {
|
||||
_ = runtime.system.write("acpi-parse: ok\n");
|
||||
} else {
|
||||
writeLine("acpi-parse: too few (ring-3 {d} < floor {d})\n", .{ devices, minimum });
|
||||
std.log.info("acpi-parse: too few (ring-3 {d} < floor {d})", .{ devices, minimum });
|
||||
}
|
||||
// Self-verify mode is standalone (no manager); stop before reporting.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
@@ -216,9 +211,9 @@ fn onInit(endpoint: runtime.ipc.Handle) bool {
|
||||
const hid = entry.hid[0..entry.hid_len];
|
||||
const desc = acpi_ids.description(hid);
|
||||
if (desc.len != 0)
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources) — {s}\n", .{ hid, entry.device_id, entry.resource_count, desc })
|
||||
std.log.info("reported {s} (device {d}, {d} resources) — {s}", .{ hid, entry.device_id, entry.resource_count, desc })
|
||||
else
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources)\n", .{ hid, entry.device_id, entry.resource_count });
|
||||
std.log.info("reported {s} (device {d}, {d} resources)", .{ hid, entry.device_id, entry.resource_count });
|
||||
if (manager) |h| {
|
||||
var report = protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
||||
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
||||
@@ -226,7 +221,7 @@ fn onInit(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
||||
}
|
||||
}
|
||||
writeLine("/system/services/acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
||||
std.log.info("reported {d} device(s) to the manager", .{registered_count});
|
||||
|
||||
armPowerButton(endpoint);
|
||||
return true;
|
||||
@@ -373,7 +368,7 @@ fn publishNotify(node: *aml.Node, code: u64) void {
|
||||
const which: power.Event = if (std.mem.eql(u8, hid[0..7], "PNP0C0A")) .battery else if (std.mem.eql(u8, hid[0..7], "ACPI0003")) .ac else if (std.mem.eql(u8, hid[0..7], "PNP0C0D")) .lid else .notify;
|
||||
var event = power.EventMessage{ .event = @intFromEnum(which), .code = @truncate(code) };
|
||||
event.hid = hid;
|
||||
writeLine("power: notify {s} code {d}\n", .{ hid[0..7], code });
|
||||
std.log.info("power: notify {s} code {d}", .{ hid[0..7], code });
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
|
||||
@@ -500,7 +495,7 @@ fn registerDevice(node: *aml.Node, hid: [8]u8, interpreter: *aml.Interpreter) vo
|
||||
applyCrs(&descriptor, node, interpreter);
|
||||
|
||||
const id = device.register(node_id, &descriptor) orelse {
|
||||
writeLine("/system/services/acpi: register refused for {s}\n", .{hid[0..@intCast(hid_len)]});
|
||||
std.log.info("register refused for {s}", .{hid[0..@intCast(hid_len)]});
|
||||
return;
|
||||
};
|
||||
registered[registered_count] = .{ .hid = hid, .hid_len = @intCast(hid_len), .device_id = id, .resource_count = descriptor.resource_count };
|
||||
|
||||
@@ -24,14 +24,6 @@ const protocol = runtime.device_manager_protocol;
|
||||
const device = runtime.device;
|
||||
const system = runtime.system;
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so output
|
||||
/// from the drivers this manager starts (which run concurrently) can never land
|
||||
/// in the middle of it.
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
/// The PCI class/subclass/prog-IF triple of an xHCI (USB 3) host controller —
|
||||
/// Serial Bus Controller / USB Controller / XHCI — named from pci-class.zig rather
|
||||
/// than written as the bare 0x0C0330 (docs/coding-standards.md, "Named values").
|
||||
@@ -56,8 +48,8 @@ const virtio_gpu_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
/// its registered id as argv[1].
|
||||
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
return switch (identity) {
|
||||
xhci_pci_class => "usb-xhci-bus",
|
||||
virtio_gpu_pci_class => "virtio-gpu",
|
||||
xhci_pci_class => "/system/drivers/usb-xhci-bus",
|
||||
virtio_gpu_pci_class => "/system/drivers/virtio-gpu",
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
@@ -67,8 +59,8 @@ fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
/// nodes the kernel used to build). ps2-bus is a singleton that finds both its
|
||||
/// devices by hid once spawned, so keyboard and mouse map to the same name.
|
||||
fn hidDriverFor(hid: []const u8) ?[]const u8 {
|
||||
if (std.mem.eql(u8, hid, "PNP0303")) return "ps2-bus"; // PS/2 keyboard
|
||||
if (std.mem.eql(u8, hid, "PNP0F13")) return "ps2-bus"; // PS/2 mouse
|
||||
if (std.mem.eql(u8, hid, "PNP0303")) return "/system/drivers/ps2-bus"; // PS/2 keyboard
|
||||
if (std.mem.eql(u8, hid, "PNP0F13")) return "/system/drivers/ps2-bus"; // PS/2 mouse
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -95,9 +87,9 @@ fn usbDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
@intFromEnum(usb_ids.mass_storage.Protocol.bulk_only),
|
||||
);
|
||||
return switch (identity) {
|
||||
keyboard => "usb-hid-keyboard",
|
||||
mouse => "usb-hid-mouse",
|
||||
storage => "usb-storage",
|
||||
keyboard => "/system/drivers/usb-hid-keyboard",
|
||||
mouse => "/system/drivers/usb-hid-mouse",
|
||||
storage => "/system/drivers/usb-storage",
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
@@ -132,7 +124,7 @@ const DriverState = enum {
|
||||
|
||||
const Driver = struct {
|
||||
used: bool = false,
|
||||
name_buffer: [24]u8 = undefined,
|
||||
name_buffer: [64]u8 = undefined, // fits a full binary path (abi.maximum_process_name)
|
||||
name_len: usize = 0,
|
||||
// The assigned device id (becomes argv[1]), or protocol.no_device.
|
||||
device_id: u64 = protocol.no_device,
|
||||
@@ -223,7 +215,7 @@ fn addChild(parent: u64, bus_address: u64, identity: u64, device_id: u64, report
|
||||
fn pruneChildrenOf(reporter: u32) void {
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.reporter == reporter) {
|
||||
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
|
||||
child.used = false;
|
||||
const event = protocol.ChildRemoved{ .parent = child.parent, .bus_address = child.bus_address };
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
@@ -270,7 +262,7 @@ fn addDriver(name: []const u8, device_id: u64, speaks_protocol: bool) void {
|
||||
spawnDriver(driver);
|
||||
return;
|
||||
}
|
||||
writeLine("/system/services/device-manager: driver table full; cannot supervise {s}\n", .{name});
|
||||
std.log.info("driver table full; cannot supervise {s}", .{name});
|
||||
}
|
||||
|
||||
/// (Re)spawn a driver instance: supervised on the manager's own endpoint, the
|
||||
@@ -285,7 +277,7 @@ fn spawnDriver(driver: *Driver) void {
|
||||
argument_count = 1;
|
||||
}
|
||||
const child = system.spawnSupervised(driver.name(), arguments[0..argument_count], manager_endpoint) orelse {
|
||||
writeLine("/system/services/device-manager: failed to spawn {s}\n", .{driver.name()});
|
||||
std.log.info("failed to spawn {s}", .{driver.name()});
|
||||
driver.state = .failed;
|
||||
return;
|
||||
};
|
||||
@@ -299,9 +291,9 @@ fn spawnDriver(driver: *Driver) void {
|
||||
driver.state = .running;
|
||||
}
|
||||
if (driver.device_id != protocol.no_device) {
|
||||
writeLine("/system/services/device-manager: spawned {s} for device {d}\n", .{ driver.name(), driver.device_id });
|
||||
std.log.info("spawned {s} for device {d}", .{ driver.name(), driver.device_id });
|
||||
} else {
|
||||
writeLine("/system/services/device-manager: spawned {s}\n", .{driver.name()});
|
||||
std.log.info("spawned {s}", .{driver.name()});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -313,7 +305,7 @@ fn onDriverExit(driver: *Driver) void {
|
||||
const reason = runtime.process.exitReason(driver.process_id) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
driver.state = .stopped;
|
||||
writeLine("/system/services/device-manager: {s} exited cleanly; not restarting\n", .{driver.name()});
|
||||
std.log.info("{s} exited cleanly; not restarting", .{driver.name()});
|
||||
return;
|
||||
}
|
||||
const now = system.clock();
|
||||
@@ -321,13 +313,13 @@ fn onDriverExit(driver: *Driver) void {
|
||||
driver.restarts = if (alive_ns < fast_death_ns) driver.restarts + 1 else 1;
|
||||
if (driver.restarts >= crash_loop_cap) {
|
||||
driver.state = .failed;
|
||||
writeLine("/system/services/device-manager: {s} is failing repeatedly (crash loop); giving up\n", .{driver.name()});
|
||||
std.log.info("{s} is failing repeatedly (crash loop); giving up", .{driver.name()});
|
||||
return;
|
||||
}
|
||||
const delay_ms = backoff_base_ms << @intCast(driver.restarts - 1);
|
||||
driver.state = .restarting;
|
||||
driver.restart_due_ns = now + delay_ms * 1_000_000;
|
||||
writeLine("/system/services/device-manager: restarting {s} in {d} ms (died: {s})\n", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||
std.log.info("restarting {s} in {d} ms (died: {s})", .{ driver.name(), delay_ms, @tagName(reason) });
|
||||
_ = system.timerOnce(manager_endpoint, delay_ms + 50);
|
||||
}
|
||||
|
||||
@@ -338,7 +330,7 @@ fn onDriverExit(driver: *Driver) void {
|
||||
fn sweepDeadlines() void {
|
||||
const now = system.clock();
|
||||
if (test_kill_pid != 0 and now >= test_kill_due_ns) {
|
||||
writeLine("/system/services/device-manager: test mode: killing the reporter\n", .{});
|
||||
std.log.info("test mode: killing the reporter", .{});
|
||||
_ = system.kill(test_kill_pid);
|
||||
test_kill_pid = 0;
|
||||
}
|
||||
@@ -346,7 +338,7 @@ fn sweepDeadlines() void {
|
||||
if (!driver.used) continue;
|
||||
switch (driver.state) {
|
||||
.awaiting_hello => if (now >= driver.hello_deadline_ns) {
|
||||
writeLine("/system/services/device-manager: {s} missed its hello deadline\n", .{driver.name()});
|
||||
std.log.info("{s} missed its hello deadline", .{driver.name()});
|
||||
_ = system.kill(driver.process_id);
|
||||
// The exit notification finishes the job via onDriverExit.
|
||||
},
|
||||
@@ -423,13 +415,13 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
var status: i32 = 0;
|
||||
if (hello.version != protocol.version) {
|
||||
status = -1;
|
||||
writeLine("/system/services/device-manager: refused hello (version {d}) from process {d}\n", .{ hello.version, sender });
|
||||
std.log.info("refused hello (version {d}) from process {d}", .{ hello.version, sender });
|
||||
} else if (driverByProcess(sender)) |driver| {
|
||||
driver.state = .running;
|
||||
writeLine("/system/services/device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||
std.log.info("hello from {s} (device {d})", .{ driver.name(), hello.device_id });
|
||||
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
|
||||
// the normal restart policy respawns it — the compositor must survive and re-attach.
|
||||
if (test_scanout_restart_mode and !test_scanout_killed and std.mem.eql(u8, driver.name(), "virtio-gpu")) {
|
||||
if (test_scanout_restart_mode and !test_scanout_killed and std.mem.eql(u8, driver.name(), "/system/drivers/virtio-gpu")) {
|
||||
test_scanout_killed = true;
|
||||
test_kill_pid = sender;
|
||||
test_kill_due_ns = system.clock() + 1_500_000_000;
|
||||
@@ -437,7 +429,7 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
}
|
||||
} else {
|
||||
status = -1;
|
||||
writeLine("/system/services/device-manager: hello from unknown process {d}\n", .{sender});
|
||||
std.log.info("hello from unknown process {d}", .{sender});
|
||||
}
|
||||
const hello_reply = protocol.HelloReply{ .status = status };
|
||||
@memcpy(reply[0..protocol.reply_size], std.mem.asBytes(&hello_reply));
|
||||
@@ -453,7 +445,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
var status: i32 = 0;
|
||||
if (driverByProcess(sender)) |driver| {
|
||||
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||
writeLine("/system/services/device-manager: child added (device {d} port {d}, identity {d}) by {s}\n", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||
if (status == 0) publishEvent(message[0..protocol.child_added_size]);
|
||||
// Matching from reports (M19.3): a registered child whose identity
|
||||
// names a driver gets one, once — re-reports after a bus restart
|
||||
@@ -498,7 +490,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
// Only the xHCI reporter is the drill's victim — pci-bus also reports
|
||||
// now, and whichever finishes second must not trigger the kill.
|
||||
if (driverByProcess(sender)) |driver| {
|
||||
if (std.mem.eql(u8, driver.name(), "usb-xhci-bus")) {
|
||||
if (std.mem.eql(u8, driver.name(), "/system/drivers/usb-xhci-bus")) {
|
||||
// Delayed, not immediate: the device-list scenario's subscriber
|
||||
// needs a window to enumerate and subscribe before the events.
|
||||
test_usb_killed = true;
|
||||
@@ -519,7 +511,7 @@ fn onChildRemoved(message: []const u8, reply: []u8, sender: u32) usize {
|
||||
var status: i32 = -1;
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == sender) {
|
||||
writeLine("/system/services/device-manager: child removed (device {d} port {d})\n", .{ child.parent, child.bus_address });
|
||||
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
|
||||
child.used = false;
|
||||
status = 0;
|
||||
}
|
||||
|
||||
@@ -1,15 +1,19 @@
|
||||
//! system/services/display-demo — a hardware-free client of the display service, the
|
||||
//! `input-source` analog for the compositor. It creates a wallpaper, a rectangle it moves
|
||||
//! each frame, and a small cursor, then drives the compositor in a present loop — proof
|
||||
//! that a *separate process* can compose a moving scene through the display service over
|
||||
//! IPC, exercising the layer client API and damage-driven present end to end
|
||||
//! `input-source` analog for the compositor. It creates a wallpaper and a rectangle it
|
||||
//! slides each frame, then drives the compositor in a present loop — proof that a
|
||||
//! *separate process* can compose a moving scene through the display service over IPC,
|
||||
//! exercising the layer client API and damage-driven present end to end
|
||||
//! (docs/display.md). It logs `display-demo: ok` once it has driven a run of frames.
|
||||
//!
|
||||
//! It draws no cursor and reads no input: the on-screen cursor is the display service's
|
||||
//! own, tracked by the service's mouse-listener thread (docs/display.md). The demo's job
|
||||
//! is only to prove client-driven animation, so its loop runs on its own frame timer and
|
||||
//! is deliberately independent of the mouse.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const display = runtime.display;
|
||||
const system = runtime.system;
|
||||
const time = runtime.time;
|
||||
const input = runtime.input;
|
||||
|
||||
pub fn main() void {
|
||||
const mode = display.info() orelse {
|
||||
@@ -28,15 +32,6 @@ pub fn main() void {
|
||||
const box = display.createLayer(0, box_y, box_w, box_h, 1) orelse return createFailed();
|
||||
_ = box.fill(0, 0, box_w, box_h, display.color(0xE0, 0x60, 0x40));
|
||||
|
||||
// A little cursor on top. Its position is signed (the layer API is i32) and clamped to
|
||||
// the screen; mouse motion arrives as relative deltas we accumulate below.
|
||||
var cursor_x: i32 = @intCast(mode.width / 2);
|
||||
var cursor_y: i32 = @intCast(mode.height / 2);
|
||||
const cursor_max_x: i32 = @as(i32, @intCast(mode.width)) - 12;
|
||||
const cursor_max_y: i32 = @as(i32, @intCast(mode.height)) - 12;
|
||||
const cursor = display.createLayer(cursor_x, cursor_y, 12, 12, 2) orelse return createFailed();
|
||||
_ = cursor.fill(0, 0, 12, 12, display.color(0xF0, 0xF0, 0xF0));
|
||||
|
||||
_ = display.present();
|
||||
_ = system.write("display-demo: scene up; animating\n");
|
||||
|
||||
@@ -45,18 +40,7 @@ pub fn main() void {
|
||||
var dx: i32 = 8;
|
||||
var frame: u32 = 0;
|
||||
|
||||
var mouse = input.subscribeMouse(); // type: ?input.MouseSubscriber
|
||||
if (mouse == null) _ = system.write("display-demo: no mouse; animating without it\n");
|
||||
|
||||
while (true) : (frame += 1) {
|
||||
if (mouse) |*ms| {
|
||||
if (ms.next()) |event| {
|
||||
cursor_x = clamp(cursor_x + event.dx, 0, cursor_max_x);
|
||||
cursor_y = clamp(cursor_y + event.dy, 0, cursor_max_y);
|
||||
_ = cursor.configure(cursor_x, cursor_y, 2, true);
|
||||
}
|
||||
}
|
||||
|
||||
x += dx;
|
||||
if (x <= 0) {
|
||||
x = 0;
|
||||
@@ -74,13 +58,6 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Clamp `v` to the inclusive range [lo, hi].
|
||||
fn clamp(v: i32, lo: i32, hi: i32) i32 {
|
||||
if (v < lo) return lo;
|
||||
if (v > hi) return hi;
|
||||
return v;
|
||||
}
|
||||
|
||||
fn createFailed() void {
|
||||
_ = system.write("display-demo: create failed\n");
|
||||
}
|
||||
|
||||
@@ -16,8 +16,10 @@ const scanout_protocol = runtime.scanout_protocol;
|
||||
const Rect = compositor.Rect;
|
||||
const Surface = compositor.Surface;
|
||||
|
||||
/// The current display mode, as a backend reports it.
|
||||
pub const Info = struct { width: u32, height: u32, pitch: u32, format: u32 };
|
||||
/// The current display mode, as a backend reports it. `refresh_hz` is the panel's
|
||||
/// refresh rate from EDID (0 = unknown) — the frame clock's pacing seed; without vblank
|
||||
/// it fixes the rate, never the phase (docs/display-v2.md, "Fenced is not vsync").
|
||||
pub const Info = struct { width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32 };
|
||||
|
||||
/// Enumeration scratch — a `DeviceDescriptor` is large, and only one scan is ever needed.
|
||||
var device_table: [64]device.DeviceDescriptor = undefined;
|
||||
@@ -26,7 +28,7 @@ var device_table: [64]device.DeviceDescriptor = undefined;
|
||||
/// framebuffer write-combining as the front buffer, and keeps a cacheable back buffer of
|
||||
/// the same geometry as the compose target. `present` streams the damaged rectangle from
|
||||
/// the back buffer to the LFB (sequential WC writes; the LFB is never read). No mode-set,
|
||||
/// no vsync — the portable floor (docs/display-v2.md).
|
||||
/// no present fence — the portable floor (docs/display-v2.md).
|
||||
pub const Gop = struct {
|
||||
device_id: u64,
|
||||
front: [*]volatile u8, // the LFB (write-combining)
|
||||
@@ -35,13 +37,14 @@ pub const Gop = struct {
|
||||
height: u32,
|
||||
pitch: u32,
|
||||
format: u32,
|
||||
refresh_hz: u32, // from the boot EDID via the display0 node (0 = unknown)
|
||||
|
||||
/// The framebuffer's id and geometry, captured together. `findDisplay` reads these out of
|
||||
/// the enumeration table and returns them by value, so the caller never re-reads the table
|
||||
/// across later syscalls (`device_enumerate` writes the whole table straight into this
|
||||
/// process's memory; reading a descriptor's tail again after other syscalls have run is a
|
||||
/// window we simply avoid by copying the few fields we need up front).
|
||||
const Found = struct { id: u64, width: u32, height: u32, pitch: u32, format: u32 };
|
||||
const Found = struct { id: u64, width: u32, height: u32, pitch: u32, format: u32, refresh_hz: u32 };
|
||||
|
||||
/// The first `display`-class device with a *valid* (non-zero) geometry, or null. A zero
|
||||
/// geometry is treated as "not ready yet" so the caller retries — a real framebuffer always
|
||||
@@ -52,7 +55,7 @@ pub const Gop = struct {
|
||||
for (device_table[0..n]) |*d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.display)) continue;
|
||||
if (d.display.width == 0 or d.display.height == 0 or d.display.pitch == 0) continue;
|
||||
return .{ .id = d.id, .width = d.display.width, .height = d.display.height, .pitch = d.display.pitch, .format = d.display.format };
|
||||
return .{ .id = d.id, .width = d.display.width, .height = d.display.height, .pitch = d.display.pitch, .format = d.display.format, .refresh_hz = d.display.refresh_hz };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -93,11 +96,12 @@ pub const Gop = struct {
|
||||
.height = found.height,
|
||||
.pitch = found.pitch,
|
||||
.format = found.format,
|
||||
.refresh_hz = found.refresh_hz,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn info(self: *const Gop) Info {
|
||||
return .{ .width = self.width, .height = self.height, .pitch = self.pitch, .format = self.format };
|
||||
return .{ .width = self.width, .height = self.height, .pitch = self.pitch, .format = self.format, .refresh_hz = self.refresh_hz };
|
||||
}
|
||||
|
||||
/// The cacheable compose target (the back buffer).
|
||||
@@ -110,27 +114,53 @@ pub const Gop = struct {
|
||||
};
|
||||
}
|
||||
|
||||
/// Stream the damaged rectangle from the back buffer to the write-combining LFB, row by
|
||||
/// row (sequential writes — what WC memory wants; the LFB is never read).
|
||||
pub fn present(self: *const Gop, damage: Rect) void {
|
||||
const c = damage.intersect(.{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) });
|
||||
if (c.isEmpty()) return;
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
const off = @as(usize, @intCast(y)) * self.pitch;
|
||||
const src: [*]const u32 = @ptrCast(@alignCast(self.back + off));
|
||||
const dst: [*]volatile u32 = @ptrCast(@alignCast(self.front + off));
|
||||
var x: i32 = c.x;
|
||||
while (x < c.right()) : (x += 1) dst[@intCast(x)] = src[@intCast(x)];
|
||||
/// Stream each damaged rectangle from the back buffer to the write-combining LFB, row
|
||||
/// by row (sequential writes — what WC memory wants; the LFB is never read). The rows
|
||||
/// are copied by `presentSpan` below, which widens the stores by hand: `volatile`
|
||||
/// keeps the compiler from eliding or reordering framebuffer writes, but it also
|
||||
/// forbids it from merging them, so a naive per-pixel loop is stuck at one 4-byte
|
||||
/// store per iteration. Keeping each copy small (the damage list) and each store wide
|
||||
/// shrinks the window in which scanout can sample a half-written frame.
|
||||
pub fn present(self: *const Gop, damage: []const Rect) void {
|
||||
const bounds = Rect{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) };
|
||||
for (damage) |rect| {
|
||||
const c = rect.intersect(bounds);
|
||||
if (c.isEmpty()) continue;
|
||||
const span: usize = @intCast(c.w);
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
const offset = @as(usize, @intCast(y)) * self.pitch + @as(usize, @intCast(c.x)) * 4;
|
||||
const source: [*]const u32 = @ptrCast(@alignCast(self.back + offset));
|
||||
const front_row: [*]volatile u32 = @ptrCast(@alignCast(self.front + offset));
|
||||
presentSpan(front_row, source, span);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// Copy `count` pixels into the write-combining front buffer with 8-byte volatile stores
|
||||
/// (plus a 4-byte head/tail where the span isn't 8-aligned — pixel spans are always
|
||||
/// 4-aligned). The loads come from the cacheable back buffer and are assembled into a
|
||||
/// `u64` in registers, so nothing here reads the front buffer.
|
||||
fn presentSpan(destination: [*]volatile u32, source: [*]const u32, count: usize) void {
|
||||
var i: usize = 0;
|
||||
if (i < count and (@intFromPtr(destination) & 7) != 0) {
|
||||
destination[0] = source[0];
|
||||
i = 1;
|
||||
}
|
||||
while (i + 2 <= count) : (i += 2) {
|
||||
const pair = @as(u64, source[i]) | (@as(u64, source[i + 1]) << 32);
|
||||
const wide: *volatile u64 = @ptrCast(@alignCast(destination + i));
|
||||
wide.* = pair;
|
||||
}
|
||||
if (i < count) destination[i] = source[i];
|
||||
}
|
||||
|
||||
/// A display mode the native backend can switch to.
|
||||
pub const Mode = scanout_protocol.Mode;
|
||||
|
||||
/// The native virtio-gpu backend: the compositor composes into a **shared** scanout surface
|
||||
/// (an `shm` region the driver created and handed over) and `present` asks the driver to put
|
||||
/// (a shared-memory region the driver created and handed over) and `present` asks the driver to put
|
||||
/// a frame on the panel over its `.scanout` endpoint. Unlike GOP there is no local copy — the
|
||||
/// surface *is* the device's resource backing, so compositing writes land straight where the
|
||||
/// driver transfers-and-flushes from (x86 DMA is cache-coherent, so the cacheable shared pages
|
||||
@@ -143,17 +173,19 @@ pub const VirtioGpu = struct {
|
||||
width: u32, // the active mode
|
||||
height: u32,
|
||||
format: u32,
|
||||
refresh_hz: u32, // from the driver's EDID read, carried in the announce (0 = unknown)
|
||||
scanout: ipc.Handle, // the driver's present + mode channel (looked up on `.scanout`)
|
||||
|
||||
pub fn info(self: *const VirtioGpu) Info {
|
||||
return .{ .width = self.width, .height = self.height, .pitch = self.stride * 4, .format = self.format };
|
||||
return .{ .width = self.width, .height = self.height, .pitch = self.stride * 4, .format = self.format, .refresh_hz = self.refresh_hz };
|
||||
}
|
||||
pub fn surface(self: *const VirtioGpu) Surface {
|
||||
return .{ .pixels = self.pixels, .stride = self.stride, .width = self.width, .height = self.height };
|
||||
}
|
||||
/// Ask the driver to present. The composited pixels are already in the shared surface, so
|
||||
/// this is a single request over `.scanout`; the driver transfers + fenced-flushes.
|
||||
pub fn present(self: *const VirtioGpu, damage: Rect) void {
|
||||
/// this is a single request over `.scanout` regardless of how many damage rectangles
|
||||
/// accumulated; the driver transfers + fenced-flushes the whole frame.
|
||||
pub fn present(self: *const VirtioGpu, damage: []const Rect) void {
|
||||
_ = damage;
|
||||
var request = scanout_protocol.Request{
|
||||
.operation = @intFromEnum(scanout_protocol.Operation.present),
|
||||
@@ -210,7 +242,7 @@ pub const Backend = union(enum) {
|
||||
inline else => |*b| b.surface(),
|
||||
};
|
||||
}
|
||||
pub fn present(self: *const Backend, damage: Rect) void {
|
||||
pub fn present(self: *const Backend, damage: []const Rect) void {
|
||||
switch (self.*) {
|
||||
inline else => |*b| b.present(damage),
|
||||
}
|
||||
@@ -236,9 +268,13 @@ pub const Backend = union(enum) {
|
||||
.virtio => true,
|
||||
};
|
||||
}
|
||||
/// Whether this backend has a vblank/fence for tear-free present (virtio-gpu: yes, V5 — every
|
||||
/// flush is fenced, so the device signals completion when the frame is actually on screen).
|
||||
pub fn hasVsync(self: *const Backend) bool {
|
||||
/// Whether this backend's present is **fenced** — it completes only once the device has
|
||||
/// consumed the frame (virtio-gpu: every flush carries a fence the used-ring ack waits on).
|
||||
/// A fence gives completion feedback and tear-free snapshot presents; it is *not* vblank —
|
||||
/// nothing paces presents to the display's refresh (base virtio-gpu 2D has no vblank event
|
||||
/// at all). True vsync needs a native driver's vblank interrupt. See docs/display-v2.md,
|
||||
/// "Fenced is not vsync".
|
||||
pub fn hasFencedPresent(self: *const Backend) bool {
|
||||
return switch (self.*) {
|
||||
.gop => false,
|
||||
.virtio => true,
|
||||
|
||||
@@ -57,6 +57,177 @@ pub const Rect = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// The dirty screen regions accumulated between presents. Kept as a *list* of rectangles,
|
||||
/// not one bounding box: when two small things move far apart — the cursor on one side of
|
||||
/// the screen, an animating layer on the other — a single bounding box unites them into a
|
||||
/// huge region, and presenting it streams megabytes to the framebuffer for a few thousand
|
||||
/// changed pixels. The long copy widens the window in which scanout (or QEMU's display
|
||||
/// refresh) samples a half-written frame — visible as tearing and cursor trails. Small
|
||||
/// separate rectangles keep each copy, and that window, tight.
|
||||
///
|
||||
/// A new rectangle that overlaps an existing entry is united into it (repainting a modest
|
||||
/// superset is harmless — compositing is idempotent); the grown entry is *not* re-merged
|
||||
/// against the rest, so entries may overlap, which costs only a duplicate repaint. When
|
||||
/// the table is full the newcomer folds into the last entry — degrading toward the old
|
||||
/// bounding-box behaviour instead of dropping damage.
|
||||
pub const DamageList = struct {
|
||||
pub const capacity = 16;
|
||||
|
||||
rects: [capacity]Rect = [_]Rect{Rect.empty} ** capacity,
|
||||
count: usize = 0,
|
||||
|
||||
pub fn add(self: *DamageList, r: Rect) void {
|
||||
if (r.isEmpty()) return;
|
||||
for (self.rects[0..self.count]) |*existing| {
|
||||
if (!existing.intersect(r).isEmpty()) {
|
||||
existing.* = existing.unite(r);
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (self.count < capacity) {
|
||||
self.rects[self.count] = r;
|
||||
self.count += 1;
|
||||
return;
|
||||
}
|
||||
self.rects[capacity - 1] = self.rects[capacity - 1].unite(r);
|
||||
}
|
||||
|
||||
pub fn isEmpty(self: *const DamageList) bool {
|
||||
return self.count == 0;
|
||||
}
|
||||
|
||||
pub fn slice(self: *const DamageList) []const Rect {
|
||||
return self.rects[0..self.count];
|
||||
}
|
||||
|
||||
pub fn clear(self: *DamageList) void {
|
||||
self.count = 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// The alternative damage tracker: a **fixed tile grid**, the scheme browser compositors
|
||||
/// and tile-based GPUs use. The screen is divided into `tile_size`-pixel tiles up front;
|
||||
/// `add` marks the tiles a rectangle touches (a bit per tile — merging is free and exact,
|
||||
/// no heuristics), and `collect` walks the grid turning runs of adjacent dirty tiles into
|
||||
/// repaint rectangles (horizontal runs, then equal-span rows merged vertically, so
|
||||
/// full-screen damage collapses back to a single rectangle).
|
||||
///
|
||||
/// Trade-off against `DamageList`: tracking is O(1) with a strictly bounded worst case
|
||||
/// (never more than the dirty tiles), but repaints are quantized — a 1-pixel change
|
||||
/// repaints a whole tile. Which wins depends on the workload; the display service has a
|
||||
/// compile-time switch (`damage_mode`) to compare them.
|
||||
pub const TileGrid = struct {
|
||||
pub const tile_size = 64;
|
||||
pub const maximum_columns = 128; // supports screens up to 8192 px wide…
|
||||
pub const maximum_rows = 128; // …and 8192 px tall (beyond that, edge tiles stretch)
|
||||
pub const maximum_tiles = maximum_columns * maximum_rows;
|
||||
/// The most rectangles `collect` produces; extras fold into the last (never dropped).
|
||||
pub const maximum_rects = 64;
|
||||
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
columns: u32 = 0,
|
||||
rows: u32 = 0,
|
||||
dirty_count: u32 = 0,
|
||||
dirty: [maximum_tiles]bool = [_]bool{false} ** maximum_tiles,
|
||||
|
||||
/// Size the grid for a screen. Also clears it — callers reset on a geometry change,
|
||||
/// where the mode-set paths damage the whole new screen anyway.
|
||||
pub fn reset(self: *TileGrid, width: u32, height: u32) void {
|
||||
self.width = width;
|
||||
self.height = height;
|
||||
self.columns = @min((width + tile_size - 1) / tile_size, maximum_columns);
|
||||
self.rows = @min((height + tile_size - 1) / tile_size, maximum_rows);
|
||||
self.clear();
|
||||
}
|
||||
|
||||
pub fn matches(self: *const TileGrid, width: u32, height: u32) bool {
|
||||
return self.width == width and self.height == height;
|
||||
}
|
||||
|
||||
pub fn isEmpty(self: *const TileGrid) bool {
|
||||
return self.dirty_count == 0;
|
||||
}
|
||||
|
||||
pub fn clear(self: *TileGrid) void {
|
||||
@memset(&self.dirty, false);
|
||||
self.dirty_count = 0;
|
||||
}
|
||||
|
||||
/// Mark every tile `r` touches. Clips to the screen first, so out-of-range
|
||||
/// rectangles are harmless.
|
||||
pub fn add(self: *TileGrid, r: Rect) void {
|
||||
const screen = Rect{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) };
|
||||
const c = r.intersect(screen);
|
||||
if (c.isEmpty()) return;
|
||||
const column_first: u32 = @intCast(@divTrunc(c.x, tile_size));
|
||||
const row_first: u32 = @intCast(@divTrunc(c.y, tile_size));
|
||||
const column_last: u32 = @min(@as(u32, @intCast(@divTrunc(c.right() - 1, tile_size))), self.columns - 1);
|
||||
const row_last: u32 = @min(@as(u32, @intCast(@divTrunc(c.bottom() - 1, tile_size))), self.rows - 1);
|
||||
var row = row_first;
|
||||
while (row <= row_last) : (row += 1) {
|
||||
var column = column_first;
|
||||
while (column <= column_last) : (column += 1) {
|
||||
const index = row * self.columns + column;
|
||||
if (!self.dirty[index]) {
|
||||
self.dirty[index] = true;
|
||||
self.dirty_count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The screen rectangle covered by tiles [column_first, column_end) of `row`. Edge
|
||||
/// tiles clamp to the true screen size (the last column/row may be partial — or, on a
|
||||
/// screen wider than the grid supports, stretched to cover the remainder).
|
||||
fn tileSpanRect(self: *const TileGrid, column_first: u32, column_end: u32, row: u32) Rect {
|
||||
const x: i32 = @intCast(column_first * tile_size);
|
||||
const y: i32 = @intCast(row * tile_size);
|
||||
const right: i32 = if (column_end >= self.columns) @intCast(self.width) else @intCast(column_end * tile_size);
|
||||
const bottom: i32 = if (row + 1 >= self.rows) @intCast(self.height) else @intCast((row + 1) * tile_size);
|
||||
return .{ .x = x, .y = y, .w = right - x, .h = bottom - y };
|
||||
}
|
||||
|
||||
/// Turn the dirty tiles into repaint rectangles in `out`: coalesce each row's runs of
|
||||
/// adjacent dirty tiles, then merge a run into the rectangle directly above it when
|
||||
/// the spans match — so a dirty block of tiles becomes one rectangle. Returns the
|
||||
/// filled prefix of `out`.
|
||||
pub fn collect(self: *const TileGrid, out: []Rect) []Rect {
|
||||
var count: usize = 0;
|
||||
var row: u32 = 0;
|
||||
while (row < self.rows) : (row += 1) {
|
||||
var column: u32 = 0;
|
||||
while (column < self.columns) {
|
||||
if (!self.dirty[row * self.columns + column]) {
|
||||
column += 1;
|
||||
continue;
|
||||
}
|
||||
var run_end = column + 1;
|
||||
while (run_end < self.columns and self.dirty[row * self.columns + run_end]) run_end += 1;
|
||||
const rect = self.tileSpanRect(column, run_end, row);
|
||||
column = run_end;
|
||||
|
||||
var merged = false;
|
||||
for (out[0..count]) |*existing| {
|
||||
if (existing.x == rect.x and existing.w == rect.w and existing.bottom() == rect.y) {
|
||||
existing.h += rect.h;
|
||||
merged = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (merged) continue;
|
||||
if (count < out.len) {
|
||||
out[count] = rect;
|
||||
count += 1;
|
||||
} else {
|
||||
out[count - 1] = out[count - 1].unite(rect);
|
||||
}
|
||||
}
|
||||
}
|
||||
return out[0..count];
|
||||
}
|
||||
};
|
||||
|
||||
/// A block of 32-bit pixels: `pixels` addressed row-major with `stride` pixels between
|
||||
/// row starts (≥ width — the framebuffer's stride is pitch/4, a layer's is its width).
|
||||
pub const Surface = struct {
|
||||
@@ -74,15 +245,17 @@ pub const Surface = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// Fill `rect` of `s` with the native pixel `colour`, clipped to `s`'s bounds.
|
||||
/// Fill `rect` of `s` with the native pixel `colour`, clipped to `s`'s bounds. Each row is
|
||||
/// one `@memset` over the clipped span, so the compiler vectorizes it and the bounds check
|
||||
/// runs once per row, not once per pixel.
|
||||
pub fn fillRect(s: Surface, rect: Rect, colour: u32) void {
|
||||
const c = rect.intersect(s.bounds());
|
||||
if (c.isEmpty()) return;
|
||||
const x0: usize = @intCast(c.x);
|
||||
const span: usize = @intCast(c.w);
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
const r = s.row(@intCast(y));
|
||||
var x: i32 = c.x;
|
||||
while (x < c.right()) : (x += 1) r[@intCast(x)] = colour;
|
||||
@memset((s.row(@intCast(y)) + x0)[0..span], colour);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -94,35 +267,36 @@ pub fn composite(dst: Surface, dx: i32, dy: i32, layer: Surface, clip: Rect) voi
|
||||
const on_screen = Rect{ .x = dx, .y = dy, .w = @intCast(layer.width), .h = @intCast(layer.height) };
|
||||
const region = on_screen.intersect(clip).intersect(dst.bounds());
|
||||
if (region.isEmpty()) return;
|
||||
const span: usize = @intCast(region.w);
|
||||
const dst_x: usize = @intCast(region.x);
|
||||
const src_x: usize = @intCast(region.x - dx);
|
||||
var y: i32 = region.y;
|
||||
while (y < region.bottom()) : (y += 1) {
|
||||
const src = layer.row(@intCast(y - dy));
|
||||
const d = dst.row(@intCast(y));
|
||||
var x: i32 = region.x;
|
||||
while (x < region.right()) : (x += 1) {
|
||||
d[@intCast(x)] = src[@intCast(x - dx)];
|
||||
}
|
||||
const source_row = layer.row(@intCast(y - dy)) + src_x;
|
||||
const destination_row = dst.row(@intCast(y)) + dst_x;
|
||||
@memcpy(destination_row[0..span], source_row[0..span]);
|
||||
}
|
||||
}
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels from `src` (raw little-endian bytes, row-major,
|
||||
/// tightly packed) into `dst` at (`dx`, `dy`), clipped to `dst`'s bounds. `src` is read
|
||||
/// with `readInt` because it comes straight out of an IPC message buffer and carries no
|
||||
/// alignment guarantee. Returns without touching anything if `src` is short.
|
||||
/// tightly packed) into `dst` at (`dx`, `dy`), clipped to `dst`'s bounds. `src` comes
|
||||
/// straight out of an IPC message buffer and carries no alignment guarantee, so each
|
||||
/// clipped row is a byte-wise `@memcpy` — which equals the old per-pixel little-endian
|
||||
/// `readInt` on every danos target (all little-endian) without the alignment concern.
|
||||
/// Returns without touching anything if `src` is short.
|
||||
pub fn blitTile(dst: Surface, dx: i32, dy: i32, src: []const u8, w: u32, h: u32) void {
|
||||
if (src.len < @as(usize, w) * h * 4) return;
|
||||
var ty: u32 = 0;
|
||||
while (ty < h) : (ty += 1) {
|
||||
const yy = dy + @as(i32, @intCast(ty));
|
||||
if (yy < 0 or yy >= dst.height) continue;
|
||||
const drow = dst.row(@intCast(yy));
|
||||
var tx: u32 = 0;
|
||||
while (tx < w) : (tx += 1) {
|
||||
const xx = dx + @as(i32, @intCast(tx));
|
||||
if (xx < 0 or xx >= dst.width) continue;
|
||||
const off = (@as(usize, ty) * w + tx) * 4;
|
||||
drow[@intCast(xx)] = std.mem.readInt(u32, src[off..][0..4], .little);
|
||||
}
|
||||
const region = Rect.init(dx, dy, @intCast(w), @intCast(h)).intersect(dst.bounds());
|
||||
if (region.isEmpty()) return;
|
||||
const span: usize = @intCast(region.w);
|
||||
const tile_x: usize = @intCast(region.x - dx);
|
||||
const dst_x: usize = @intCast(region.x);
|
||||
var y: i32 = region.y;
|
||||
while (y < region.bottom()) : (y += 1) {
|
||||
const tile_y: usize = @intCast(y - dy);
|
||||
const offset = (tile_y * w + tile_x) * 4;
|
||||
const destination_row = dst.row(@intCast(y)) + dst_x;
|
||||
@memcpy(std.mem.sliceAsBytes(destination_row[0..span]), src[offset..][0 .. span * 4]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -179,6 +353,92 @@ test "composite honours the damage rectangle" {
|
||||
try std.testing.expectEqual(@as(u32, 0), back[4 * 8 + 4]); // outside damage
|
||||
}
|
||||
|
||||
test "damage list keeps disjoint rectangles separate and merges overlap" {
|
||||
var list = DamageList{};
|
||||
list.add(Rect.init(0, 0, 10, 10));
|
||||
list.add(Rect.init(100, 100, 10, 10)); // far away: its own entry
|
||||
try std.testing.expectEqual(@as(usize, 2), list.slice().len);
|
||||
list.add(Rect.init(5, 5, 10, 10)); // overlaps the first: united into it
|
||||
try std.testing.expectEqual(@as(usize, 2), list.slice().len);
|
||||
try std.testing.expectEqual(Rect.init(0, 0, 15, 15), list.slice()[0]);
|
||||
try std.testing.expect(!list.isEmpty());
|
||||
list.clear();
|
||||
try std.testing.expect(list.isEmpty());
|
||||
}
|
||||
|
||||
test "damage list folds overflow into the last entry instead of dropping it" {
|
||||
var list = DamageList{};
|
||||
var i: i32 = 0;
|
||||
while (i < DamageList.capacity) : (i += 1) {
|
||||
list.add(Rect.init(i * 100, 0, 10, 10)); // disjoint: fills every slot
|
||||
}
|
||||
try std.testing.expectEqual(@as(usize, DamageList.capacity), list.slice().len);
|
||||
const overflow = Rect.init(0, 5000, 10, 10);
|
||||
list.add(overflow);
|
||||
try std.testing.expectEqual(@as(usize, DamageList.capacity), list.slice().len);
|
||||
const last = list.slice()[DamageList.capacity - 1];
|
||||
try std.testing.expect(!last.intersect(overflow).isEmpty()); // still covered
|
||||
}
|
||||
|
||||
test "damage list ignores empty rectangles" {
|
||||
var list = DamageList{};
|
||||
list.add(Rect.empty);
|
||||
try std.testing.expect(list.isEmpty());
|
||||
}
|
||||
|
||||
test "tile grid coalesces a run of adjacent tiles into one rectangle" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(256, 128); // 4×2 tiles of 64 px
|
||||
grid.add(Rect.init(10, 10, 100, 10)); // spans tiles (0,0) and (1,0)
|
||||
var scratch: [TileGrid.maximum_rects]Rect = undefined;
|
||||
const rects = grid.collect(&scratch);
|
||||
try std.testing.expectEqual(@as(usize, 1), rects.len);
|
||||
try std.testing.expectEqual(Rect.init(0, 0, 128, 64), rects[0]);
|
||||
}
|
||||
|
||||
test "tile grid: full-screen damage collapses back to a single rectangle" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(1280, 720); // 20×12 tiles; the bottom row is partial (720 = 11*64 + 16)
|
||||
grid.add(Rect.init(0, 0, 1280, 720));
|
||||
var scratch: [TileGrid.maximum_rects]Rect = undefined;
|
||||
const rects = grid.collect(&scratch);
|
||||
try std.testing.expectEqual(@as(usize, 1), rects.len);
|
||||
try std.testing.expectEqual(Rect.init(0, 0, 1280, 720), rects[0]);
|
||||
}
|
||||
|
||||
test "tile grid keeps far-apart damage as separate rectangles" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(1280, 720);
|
||||
grid.add(Rect.init(0, 0, 10, 10)); // top-left tile
|
||||
grid.add(Rect.init(1000, 600, 10, 10)); // a far-away tile
|
||||
var scratch: [TileGrid.maximum_rects]Rect = undefined;
|
||||
const rects = grid.collect(&scratch);
|
||||
try std.testing.expectEqual(@as(usize, 2), rects.len);
|
||||
}
|
||||
|
||||
test "tile grid clamps edge tiles to the true screen size" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(100, 100); // 2×2 tiles, both partial in each axis
|
||||
grid.add(Rect.init(0, 0, 100, 100));
|
||||
var scratch: [TileGrid.maximum_rects]Rect = undefined;
|
||||
const rects = grid.collect(&scratch);
|
||||
try std.testing.expectEqual(@as(usize, 1), rects.len);
|
||||
try std.testing.expectEqual(Rect.init(0, 0, 100, 100), rects[0]);
|
||||
}
|
||||
|
||||
test "tile grid clear empties it and reset resizes it" {
|
||||
var grid = TileGrid{};
|
||||
grid.reset(256, 256);
|
||||
grid.add(Rect.init(0, 0, 256, 256));
|
||||
try std.testing.expect(!grid.isEmpty());
|
||||
grid.clear();
|
||||
try std.testing.expect(grid.isEmpty());
|
||||
try std.testing.expect(grid.matches(256, 256));
|
||||
grid.reset(512, 512);
|
||||
try std.testing.expect(!grid.matches(256, 256));
|
||||
try std.testing.expect(grid.isEmpty());
|
||||
}
|
||||
|
||||
test "blitTile copies a packed tile, clipping and reading unaligned bytes" {
|
||||
var back = [_]u32{0} ** (4 * 4);
|
||||
const dst = Surface{ .pixels = &back, .stride = 4, .width = 4, .height = 4 };
|
||||
|
||||
@@ -10,7 +10,9 @@
|
||||
//! z-order, and visibility. Clients create layers, draw into them by command (`fill_rect`,
|
||||
//! `blit_tile`), mark `damage`, and ask for a `present`; the compositor repaints only the
|
||||
//! damaged region — clear it, paint the visible layers bottom-to-top into the backend's
|
||||
//! surface, then `backend.present(damage)`. Shared-memory client surfaces are later
|
||||
//! surface, then `backend.present(damage)`. Presents are paced by a ~60 Hz **frame clock**
|
||||
//! (see `schedulePresent`), so any number of client presents and cursor moves inside one
|
||||
//! interval coalesce into a single frame. Shared-memory client surfaces are later
|
||||
//! (docs/display-v2.md).
|
||||
|
||||
const std = @import("std");
|
||||
@@ -21,6 +23,8 @@ const backend_mod = @import("backend.zig");
|
||||
const protocol = runtime.display_protocol;
|
||||
const ipc = runtime.ipc;
|
||||
const system = runtime.system;
|
||||
const input = runtime.input;
|
||||
const Thread = runtime.Thread;
|
||||
const Rect = compositor.Rect;
|
||||
const Surface = compositor.Surface;
|
||||
|
||||
@@ -48,8 +52,10 @@ var pending_modeset_check: bool = false;
|
||||
var background: u32 = 0;
|
||||
|
||||
/// The layer stack. A fixed table (a compositor has few top-level surfaces during
|
||||
/// bring-up); each used slot owns an mmap'd surface. `damage` accumulates the dirty
|
||||
/// screen region since the last `present`, so a present touches only what changed.
|
||||
/// bring-up); each used slot owns an mmap'd surface. `damage_list` accumulates the dirty
|
||||
/// screen rectangles since the last `present`, so a present touches only what changed —
|
||||
/// and keeps far-apart changes (the cursor here, an animating layer there) as *separate*
|
||||
/// small copies rather than one huge bounding box (see compositor.DamageList).
|
||||
const maximum_layers = 16;
|
||||
|
||||
const Layer = struct {
|
||||
@@ -63,7 +69,67 @@ const Layer = struct {
|
||||
};
|
||||
|
||||
var layers: [maximum_layers]Layer = [_]Layer{.{}} ** maximum_layers;
|
||||
var damage: Rect = Rect.empty;
|
||||
|
||||
/// Which damage tracker drives `present` — a compile-time A/B switch (both are in
|
||||
/// compositor.zig with the trade-off discussion):
|
||||
/// .list — free-form dirty rectangles (tight bounds, heuristic merging)
|
||||
/// .grid — a fixed 64-px tile grid (exact O(1) merging, tile-quantized repaints)
|
||||
const DamageMode = enum { list, grid };
|
||||
const damage_mode: DamageMode = .grid;
|
||||
|
||||
var damage_list: compositor.DamageList = .{};
|
||||
var damage_grid: compositor.TileGrid = .{};
|
||||
|
||||
/// The **frame clock**: client `present` requests and cursor motion don't repaint
|
||||
/// immediately — they accumulate damage and arm a one-shot timer, and the tick composites
|
||||
/// everything pending as one frame. That paces presents to ~60 Hz no matter how fast
|
||||
/// clients draw or the mouse moves (previously every mouse event became a full present).
|
||||
/// No backend has a real vblank to pace by (docs/display-v2.md, "Fenced is not vsync");
|
||||
/// this is the software stand-in, the same strategy Linux uses atop virtio-gpu. Bring-up
|
||||
/// paths that need pixels on screen *now* (initialise, the self-checks) still call
|
||||
/// `present()` directly.
|
||||
///
|
||||
/// The interval comes from the *active backend's* panel refresh rate (EDID: the loader
|
||||
/// captures it for the GOP floor while firmware still runs; the native driver reads its
|
||||
/// own and carries it in the announce). `updateFrameClock` re-derives it whenever the
|
||||
/// backend changes — the boot framebuffer's clock dies with the GOP floor at upgrade.
|
||||
/// Without a rate the clock defaults to 60 Hz, and it is clamped to [30, 120] Hz so a
|
||||
/// mis-parsed EDID can neither starve nor flood the compositor.
|
||||
var frame_interval_milliseconds: u64 = 16;
|
||||
var frame_timer_armed = false;
|
||||
|
||||
/// Derive the frame-clock interval from the active backend's refresh rate and log what
|
||||
/// the clock is now pacing to. Called at bring-up and again on every backend change.
|
||||
fn updateFrameClock() void {
|
||||
const reported = backend.info().refresh_hz;
|
||||
const rate: u64 = if (reported == 0) 60 else @min(@max(reported, 30), 120);
|
||||
frame_interval_milliseconds = @max(1000 / rate, 1);
|
||||
var line: [96]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "display: frame clock {d} Hz ({s})\n", .{
|
||||
1000 / frame_interval_milliseconds,
|
||||
if (reported == 0) "default" else "panel EDID",
|
||||
}) catch return);
|
||||
}
|
||||
|
||||
/// Arm the frame clock unless a tick is already pending: any number of requests inside
|
||||
/// one interval coalesce into that single tick's present.
|
||||
fn schedulePresent() void {
|
||||
if (frame_timer_armed) return;
|
||||
frame_timer_armed = true;
|
||||
_ = system.timerOnce(service_endpoint, frame_interval_milliseconds);
|
||||
}
|
||||
|
||||
/// A timer landing — the frame clock, or the deferred first native present armed by
|
||||
/// `attach_scanout`: present the accumulated damage, then run the one-shot mode-set
|
||||
/// self-check if the native upgrade queued it.
|
||||
fn frameTick() void {
|
||||
frame_timer_armed = false;
|
||||
present();
|
||||
if (pending_modeset_check) {
|
||||
pending_modeset_check = false;
|
||||
modesetSelfCheck();
|
||||
}
|
||||
}
|
||||
|
||||
// --- geometry helpers -------------------------------------------------------
|
||||
|
||||
@@ -76,9 +142,20 @@ fn layerScreenRect(l: *const Layer) Rect {
|
||||
return .{ .x = l.x, .y = l.y, .w = @intCast(l.surface.width), .h = @intCast(l.surface.height) };
|
||||
}
|
||||
|
||||
/// Add `r` (screen coordinates) to the pending damage, clipped to the screen.
|
||||
/// Add `r` (screen coordinates) to the pending damage, clipped to the screen. In grid
|
||||
/// mode the grid re-sizes itself lazily when the screen geometry changes — every
|
||||
/// geometry-changing path (`attach_scanout`, `set_mode`) damages the whole new screen
|
||||
/// right after, so damage pending from the old geometry is safely superseded.
|
||||
fn addDamage(r: Rect) void {
|
||||
damage = damage.unite(r.intersect(screenRect()));
|
||||
const clipped = r.intersect(screenRect());
|
||||
switch (damage_mode) {
|
||||
.list => damage_list.add(clipped),
|
||||
.grid => {
|
||||
const mode = backend.info();
|
||||
if (!damage_grid.matches(mode.width, mode.height)) damage_grid.reset(mode.width, mode.height);
|
||||
damage_grid.add(clipped);
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// --- layer operations (called from onMessage and the self-check) ------------
|
||||
@@ -181,21 +258,29 @@ fn compositeInto(clip: Rect) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Composite the accumulated damage into the backend's surface, hand it to the backend to
|
||||
/// put on screen, then clear the damage. A no-op when nothing is dirty. The frame counter
|
||||
/// advances regardless, so callers can name frames.
|
||||
/// Composite each accumulated damage rectangle into the backend's surface, hand the list
|
||||
/// to the backend to put on screen, then clear the damage. A no-op when nothing is dirty.
|
||||
/// The frame counter advances regardless, so callers can name frames.
|
||||
fn present() void {
|
||||
const dirty = damage.intersect(screenRect());
|
||||
if (!dirty.isEmpty()) {
|
||||
compositeInto(dirty);
|
||||
var scratch: [compositor.TileGrid.maximum_rects]Rect = undefined;
|
||||
const dirty: []const Rect = switch (damage_mode) {
|
||||
.list => damage_list.slice(),
|
||||
.grid => damage_grid.collect(&scratch),
|
||||
};
|
||||
const had_damage = dirty.len != 0;
|
||||
if (had_damage) {
|
||||
for (dirty) |region| compositeInto(region);
|
||||
backend.present(dirty);
|
||||
}
|
||||
damage = Rect.empty;
|
||||
switch (damage_mode) {
|
||||
.list => damage_list.clear(),
|
||||
.grid => damage_grid.clear(),
|
||||
}
|
||||
frames += 1;
|
||||
|
||||
// The first present after a native upgrade confirms the composited frame actually reached
|
||||
// the shared scanout surface (the automated stand-in for "it's on screen").
|
||||
if (pending_native_verify and !dirty.isEmpty()) {
|
||||
if (pending_native_verify and had_damage) {
|
||||
pending_native_verify = false;
|
||||
verifyNativePresent();
|
||||
}
|
||||
@@ -219,13 +304,13 @@ fn verifyNativePresent() void {
|
||||
/// present channel, switch the backend to virtio-gpu, and queue a full-screen repaint. The
|
||||
/// present is deferred to a timer (see `service_endpoint`) so it happens after this reply
|
||||
/// unblocks the driver and it starts serving `.scanout`.
|
||||
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, capability: ?ipc.Handle, reply: []u8) usize {
|
||||
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, refresh_hz: u32, capability: ?ipc.Handle, reply: []u8) usize {
|
||||
const cap = capability orelse return fail(reply);
|
||||
if (width == 0 or height == 0 or stride < width) return fail(reply);
|
||||
const mapped = runtime.shm.map(cap) orelse return fail(reply);
|
||||
const mapped = runtime.shared_memory.map(cap) orelse return fail(reply);
|
||||
const scanout = ipc.lookup(.scanout) orelse return fail(reply);
|
||||
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
|
||||
// scanout. (The previous shared mapping leaks — there is no shm_unmap syscall yet — but the
|
||||
// scanout. (The previous shared mapping leaks — there is no shared_memory_unmap syscall yet — but the
|
||||
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
|
||||
const reattach = switch (backend) {
|
||||
.virtio => true,
|
||||
@@ -238,9 +323,11 @@ fn attachScanout(stride: u32, width: u32, height: u32, format: u32, capability:
|
||||
.width = width,
|
||||
.height = height,
|
||||
.format = format,
|
||||
.refresh_hz = refresh_hz,
|
||||
.scanout = scanout,
|
||||
} };
|
||||
background = protocol.pack(format, 0x20, 0x30, 0x48); // re-pack the wallpaper for the mode
|
||||
updateFrameClock(); // the GOP floor's clock dies here — pace by the GPU's EDID now
|
||||
addDamage(screenRect()); // the whole new surface must be painted
|
||||
pending_native_verify = true;
|
||||
if (!reattach) pending_modeset_check = true; // the mode-set self-check runs once, on first upgrade
|
||||
@@ -255,7 +342,8 @@ fn attachScanout(stride: u32, width: u32, height: u32, format: u32, capability:
|
||||
/// After the native upgrade is verified, prove the runtime-resolution-change and fenced-present
|
||||
/// paths: query the driver's modes, switch to one that differs from the current, re-composite
|
||||
/// the whole screen at the new size, and confirm the backend now reports that geometry. The
|
||||
/// present goes through the driver's fenced flush, so a clean present is a vsync present.
|
||||
/// present goes through the driver's fenced flush, so a clean present is a *fenced* present —
|
||||
/// completion-acknowledged and tear-free, not vblank-paced (docs/display-v2.md).
|
||||
fn modesetSelfCheck() void {
|
||||
if (!backend.canModeSet()) return;
|
||||
var mode_list: [4]backend_mod.Mode = undefined;
|
||||
@@ -287,7 +375,7 @@ fn modesetSelfCheck() void {
|
||||
if (now.width == wanted.width and now.height == wanted.height) {
|
||||
var line: [80]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "display: mode set to {d}x{d}, verified\n", .{ now.width, now.height }) catch "display: mode set, verified\n");
|
||||
if (backend.hasVsync()) _ = system.write("display: vsync present ok\n");
|
||||
if (backend.hasFencedPresent()) _ = system.write("display: fenced present ok\n");
|
||||
} else {
|
||||
_ = system.write("display: mode set FAILED (geometry unchanged)\n");
|
||||
}
|
||||
@@ -328,6 +416,157 @@ fn fail_check(_: []const u8) void {
|
||||
_ = system.write("display: compositor self-check FAILED (setup)\n");
|
||||
}
|
||||
|
||||
// --- cursor + mouse-input thread --------------------------------------------
|
||||
//
|
||||
// The compositor is the single owner of the framebuffer: only the main service
|
||||
// loop touches `backend` and the layer stack. A dedicated listener thread (spawned
|
||||
// in `initialise`) blocks on the input service's mouse stream, accumulates relative
|
||||
// motion into an absolute cursor position, and hands that position to the main loop
|
||||
// through `cursor_channel` — a single-slot latest-value cell (the renderer wants
|
||||
// where the cursor *is*, not a replay of every delta). The listener never touches
|
||||
// the compositor; it only writes the channel and pokes the main loop awake with a
|
||||
// self-directed `ipc.send`, which arrives as a message-notification in the service
|
||||
// loop (docs/threading.md, docs/display.md). Shared fate: a fault in the listener
|
||||
// takes the whole display down and the supervisor restarts it (docs/resilience.md).
|
||||
|
||||
const cursor_size = 10; // a small square sprite — enough to prove tracking
|
||||
const cursor_z = 0xFFFF_FFFF; // always above client layers
|
||||
const cursor_report_threshold = 5; // px of travel before the tracking marker latches
|
||||
|
||||
var cursor_layer: ?u32 = null;
|
||||
var cursor_origin_x: i32 = 0;
|
||||
var cursor_origin_y: i32 = 0;
|
||||
/// Latched once the cursor has demonstrably tracked a run of motion end to end
|
||||
/// (source -> input service -> listener -> channel -> render): the `display-cursor`
|
||||
/// test's success marker.
|
||||
var cursor_tracking_reported: bool = false;
|
||||
|
||||
const poke_byte = [_]u8{0}; // the poke carries no payload; the value lives in the channel
|
||||
|
||||
/// Shared between the listener thread (producer) and the main loop (consumer).
|
||||
/// Latest-value semantics with a coalesced wake: at most one poke is queued while
|
||||
/// the main loop has not drained the last one, so a fast mouse cannot flood the
|
||||
/// service endpoint.
|
||||
const CursorChannel = struct {
|
||||
lock: Thread.Mutex = .{},
|
||||
poke_endpoint: ipc.Handle = 0,
|
||||
x: i32 = 0,
|
||||
y: i32 = 0,
|
||||
buttons: u32 = 0,
|
||||
dirty: bool = false,
|
||||
poke_pending: bool = false,
|
||||
|
||||
const Snapshot = struct { x: i32, y: i32, buttons: u32 };
|
||||
|
||||
/// Producer (listener thread): record the newest position and, unless a wake is
|
||||
/// already queued, poke the main loop awake.
|
||||
fn publish(self: *CursorChannel, x: i32, y: i32, buttons: u32) void {
|
||||
self.lock.lock();
|
||||
self.x = x;
|
||||
self.y = y;
|
||||
self.buttons = buttons;
|
||||
self.dirty = true;
|
||||
const need_poke = !self.poke_pending;
|
||||
if (need_poke) self.poke_pending = true;
|
||||
self.lock.unlock();
|
||||
if (need_poke) _ = ipc.send(self.poke_endpoint, &poke_byte);
|
||||
}
|
||||
|
||||
/// Consumer (main loop): take the latest position, or null if nothing changed
|
||||
/// since the last take. Clears the wake latch so the next publish pokes again.
|
||||
fn take(self: *CursorChannel) ?Snapshot {
|
||||
self.lock.lock();
|
||||
defer self.lock.unlock();
|
||||
self.poke_pending = false;
|
||||
if (!self.dirty) return null;
|
||||
self.dirty = false;
|
||||
return .{ .x = self.x, .y = self.y, .buttons = self.buttons };
|
||||
}
|
||||
};
|
||||
|
||||
var cursor_channel: CursorChannel = .{};
|
||||
|
||||
fn clampAxis(value: i32, max: i32) i32 {
|
||||
if (value < 0) return 0;
|
||||
if (value > max) return max;
|
||||
return value;
|
||||
}
|
||||
|
||||
/// The mouse-listener thread. Blocks on the input service's mouse stream, accumulates
|
||||
/// relative motion into an absolute position clamped to the screen, and publishes each
|
||||
/// update. Runs for the life of the process; a parked `next()` leaves the core free to
|
||||
/// halt (docs/halting.md). It reads only its own state and the channel — never the
|
||||
/// compositor — so no lock guards the framebuffer.
|
||||
fn mouseListener(width: u32, height: u32) void {
|
||||
var mouse = input.subscribeMouse() orelse {
|
||||
_ = system.write("display: mouse subscribe failed\n");
|
||||
return;
|
||||
};
|
||||
// Our own handle to the compositor's endpoint. IPC handles are per-thread, so we
|
||||
// cannot reuse the main thread's service handle — we look the service up to install a
|
||||
// handle in this thread's table. A poke posted here wakes the compositor loop parked
|
||||
// in replyWait (docs/threading.md: handles do not cross threads).
|
||||
cursor_channel.poke_endpoint = ipc.lookup(.display) orelse {
|
||||
_ = system.write("display: mouse listener could not reach the compositor endpoint\n");
|
||||
return;
|
||||
};
|
||||
const max_x: i32 = @as(i32, @intCast(width)) - 1;
|
||||
const max_y: i32 = @as(i32, @intCast(height)) - 1;
|
||||
var x: i32 = @divTrunc(max_x, 2);
|
||||
var y: i32 = @divTrunc(max_y, 2);
|
||||
var buttons: u32 = 0;
|
||||
while (true) {
|
||||
const event = mouse.next() orelse continue;
|
||||
// Switch on the raw kind (not @enumFromInt, which would panic on a scroll or
|
||||
// future kind): motion moves the cursor, anything else just updates buttons.
|
||||
if (event.kind == @intFromEnum(input.MouseEventKind.motion)) {
|
||||
x = clampAxis(x + event.dx, max_x);
|
||||
y = clampAxis(y + event.dy, max_y);
|
||||
} else {
|
||||
buttons = event.buttons;
|
||||
}
|
||||
cursor_channel.publish(x, y, buttons);
|
||||
}
|
||||
}
|
||||
|
||||
/// Consume the latest cursor position from the channel and move the cursor layer to it.
|
||||
/// Runs on the main loop (the compositor owner) in response to a listener poke.
|
||||
/// `configureLayer` damages both the old and new footprints; the frame clock presents
|
||||
/// them at the next tick, so a fast mouse coalesces to at most ~60 repaints a second.
|
||||
fn renderCursor() void {
|
||||
const snapshot = cursor_channel.take() orelse return;
|
||||
const id = cursor_layer orelse return;
|
||||
_ = configureLayer(id, snapshot.x, snapshot.y, cursor_z, true);
|
||||
schedulePresent();
|
||||
if (!cursor_tracking_reported and
|
||||
@abs(snapshot.x - cursor_origin_x) >= cursor_report_threshold and
|
||||
@abs(snapshot.y - cursor_origin_y) >= cursor_report_threshold)
|
||||
{
|
||||
cursor_tracking_reported = true;
|
||||
_ = system.write("display: cursor tracking mouse ok\n");
|
||||
}
|
||||
}
|
||||
|
||||
/// Create the cursor sprite (a top-z square) at screen centre and spawn the listener
|
||||
/// thread. Called from `initialise` once the backend is up. If either step fails the
|
||||
/// display still serves drawing clients — it just has no cursor.
|
||||
fn startCursorTracking() void {
|
||||
const mode = backend.info();
|
||||
cursor_origin_x = @divTrunc(@as(i32, @intCast(mode.width)), 2);
|
||||
cursor_origin_y = @divTrunc(@as(i32, @intCast(mode.height)), 2);
|
||||
const id = createLayer(cursor_origin_x, cursor_origin_y, cursor_size, cursor_size, cursor_z, true) orelse {
|
||||
_ = system.write("display: could not create cursor layer\n");
|
||||
return;
|
||||
};
|
||||
cursor_layer = id;
|
||||
_ = fillLayer(id, Rect.init(0, 0, cursor_size, cursor_size), protocol.pack(mode.format, 0xF0, 0xF0, 0xF0));
|
||||
present(); // show the cursor at its start position
|
||||
|
||||
_ = Thread.spawn(.{}, mouseListener, .{ mode.width, mode.height }) catch {
|
||||
_ = system.write("display: could not spawn mouse listener\n");
|
||||
};
|
||||
}
|
||||
|
||||
// --- service ----------------------------------------------------------------
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
@@ -347,9 +586,13 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = system.write(std.fmt.bufPrint(&line, "display: online {d}x{d} pitch {d} format {d}\n", .{
|
||||
mode.width, mode.height, mode.pitch, mode.format,
|
||||
}) catch "display: online\n");
|
||||
updateFrameClock();
|
||||
_ = system.write("display: presented frame 0\n");
|
||||
|
||||
selfCheck();
|
||||
|
||||
// Bring up the cursor and the mouse-listener thread now that the backend is live.
|
||||
startCursorTracking();
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -405,11 +648,13 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
||||
return ok(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.present) => {
|
||||
present();
|
||||
// Scheduled, not immediate: the frame clock composites the accumulated damage
|
||||
// at the next tick, so back-to-back client presents coalesce into one frame.
|
||||
schedulePresent();
|
||||
return ok(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.attach_scanout) => {
|
||||
return attachScanout(request.x, request.width, request.height, request.colour, capability, reply);
|
||||
return attachScanout(request.x, request.width, request.height, request.colour, request.y, capability, reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.set_mode) => {
|
||||
if (!backend.setMode(request.width, request.height)) return fail(reply);
|
||||
@@ -435,16 +680,15 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
||||
}
|
||||
}
|
||||
|
||||
/// The only notification the compositor arms is the post-attach present timer: repaint the
|
||||
/// screen into the freshly attached native surface, verify the frame landed, then run the
|
||||
/// one-shot mode-set self-check (V5).
|
||||
/// Two notification sources reach the compositor, and one coalesced badge can carry
|
||||
/// both, so each bit is handled independently. A **message-notification** is a poke from
|
||||
/// the mouse-listener thread (a buffered self-`ipc.send`, `notify_message_bit`): fold the
|
||||
/// newest cursor position into the scene. A **timer** (`notify_timer_bit`) is the frame
|
||||
/// clock — or the deferred first native present after `attach_scanout` — either way,
|
||||
/// present the accumulated damage.
|
||||
fn onNotification(badge: u64) void {
|
||||
_ = badge;
|
||||
present(); // native present + verify (first timer fire after the upgrade)
|
||||
if (pending_modeset_check) {
|
||||
pending_modeset_check = false;
|
||||
modesetSelfCheck();
|
||||
}
|
||||
if (badge & ipc.notify_message_bit != 0) renderCursor();
|
||||
if (badge & ipc.notify_timer_bit != 0) frameTick();
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
|
||||
@@ -24,11 +24,13 @@ pub const Operation = enum(u32) {
|
||||
damage = 6,
|
||||
/// present(): composite the dirty layers and flush to the screen.
|
||||
present = 7,
|
||||
/// attach_scanout(x=stride, width, height, colour=format) + <surface capability>: a native
|
||||
/// scanout driver announces itself, handing over the shared scanout surface as an `ipc_call`
|
||||
/// send_cap. The compositor maps it, looks up the driver's `.scanout` present channel, and
|
||||
/// upgrades off the GOP floor (docs/display-v2.md V4). `x` is the surface's row stride in
|
||||
/// pixels, `colour` the DisplayFormat.
|
||||
/// attach_scanout(x=stride, y=refresh_hz, width, height, colour=format) + <surface
|
||||
/// capability>: a native scanout driver announces itself, handing over the shared scanout
|
||||
/// surface as an `ipc_call` send_cap. The compositor maps it, looks up the driver's
|
||||
/// `.scanout` present channel, and upgrades off the GOP floor (docs/display-v2.md V4).
|
||||
/// `x` is the surface's row stride in pixels, `y` the panel refresh rate from the
|
||||
/// driver's EDID read (0 = unknown; paces the compositor's frame clock), `colour` the
|
||||
/// DisplayFormat.
|
||||
attach_scanout = 8,
|
||||
/// set_mode(width, height): change the display resolution — only a native backend that
|
||||
/// reports `canModeSet` honours it; on the GOP floor it fails (docs/display-v2.md V5).
|
||||
|
||||
+337
-16
@@ -70,6 +70,15 @@ pub const FileSystem = struct {
|
||||
// server before a mutating op. 0 leaves the on-disk timestamps untouched (host
|
||||
// tests that don't care about time, and reads).
|
||||
current_time_epoch: u64 = 0,
|
||||
// Where the next allocateCluster scan starts — clusters below this were seen
|
||||
// in use, so a fresh scan needn't re-read them (frees rewind it). Without
|
||||
// this the scan re-read the FAT from cluster 2 per allocation: measured at
|
||||
// ~1 s/cluster on a part-full volume (a 37 s shutdown log flush).
|
||||
next_free_hint: u32 = 2,
|
||||
// Which absolute LBA `fat_sector` currently holds (0 = none). Lets a FAT
|
||||
// scan serve consecutive entries from one device read; every write through
|
||||
// the sector keeps the cache coherent (writeFatBytes updates it in place).
|
||||
fat_sector_lba: u64 = 0,
|
||||
|
||||
// Every filesystem-relative sector access adds the partition base.
|
||||
fn blockRead(self: *FileSystem, lba: u64, buffer: []u8) bool {
|
||||
@@ -139,7 +148,10 @@ pub const FileSystem = struct {
|
||||
while (done < out.len) {
|
||||
const lba = position / sector_size;
|
||||
const within: usize = @intCast(position % sector_size);
|
||||
if (!self.blockRead(lba, &self.fat_sector)) return false;
|
||||
if (lba != self.fat_sector_lba) {
|
||||
if (!self.blockRead(lba, &self.fat_sector)) return false;
|
||||
self.fat_sector_lba = lba;
|
||||
}
|
||||
const n = @min(out.len - done, sector_size - within);
|
||||
@memcpy(out[done .. done + n], self.fat_sector[within .. within + n]);
|
||||
done += n;
|
||||
@@ -159,7 +171,10 @@ pub const FileSystem = struct {
|
||||
while (done < in.len) {
|
||||
const lba = position / sector_size;
|
||||
const within: usize = @intCast(position % sector_size);
|
||||
if (!self.blockRead(lba, &self.fat_sector)) return false;
|
||||
if (lba != self.fat_sector_lba) {
|
||||
if (!self.blockRead(lba, &self.fat_sector)) return false;
|
||||
self.fat_sector_lba = lba;
|
||||
}
|
||||
const n = @min(in.len - done, sector_size - within);
|
||||
@memcpy(self.fat_sector[within .. within + n], in[done .. done + n]);
|
||||
if (!self.blockWrite(lba, &self.fat_sector)) return false;
|
||||
@@ -237,11 +252,19 @@ pub const FileSystem = struct {
|
||||
|
||||
// Find and claim a free cluster, marking it end-of-chain. Returns its number.
|
||||
fn allocateCluster(self: *FileSystem) ?u32 {
|
||||
var cluster: u32 = 2;
|
||||
while (cluster < self.geometry.cluster_count + 2) : (cluster += 1) {
|
||||
if (self.readFatEntry(cluster) == on_disk.free_cluster) {
|
||||
if (!self.writeFatEntry(cluster, self.endOfChainValue())) return null;
|
||||
return cluster;
|
||||
const limit = self.geometry.cluster_count + 2;
|
||||
// Two passes: hint..end, then 2..hint (the hint only skips known-used
|
||||
// ground, it never hides a freed cluster — freeChain rewinds it).
|
||||
var pass: u2 = 0;
|
||||
while (pass < 2) : (pass += 1) {
|
||||
var cluster: u32 = if (pass == 0) self.next_free_hint else 2;
|
||||
const end: u32 = if (pass == 0) limit else self.next_free_hint;
|
||||
while (cluster < end) : (cluster += 1) {
|
||||
if (self.readFatEntry(cluster) == on_disk.free_cluster) {
|
||||
if (!self.writeFatEntry(cluster, self.endOfChainValue())) return null;
|
||||
self.next_free_hint = cluster + 1;
|
||||
return cluster;
|
||||
}
|
||||
}
|
||||
}
|
||||
return null;
|
||||
@@ -257,6 +280,7 @@ pub const FileSystem = struct {
|
||||
while (cluster >= 2 and cluster < limit and guard < limit) : (guard += 1) {
|
||||
const next = self.readFatEntry(cluster);
|
||||
_ = self.writeFatEntry(cluster, on_disk.free_cluster);
|
||||
if (cluster < self.next_free_hint) self.next_free_hint = cluster;
|
||||
if (self.isEndOfChain(next) or next < 2) break;
|
||||
cluster = next;
|
||||
}
|
||||
@@ -603,6 +627,192 @@ pub const FileSystem = struct {
|
||||
_ = self.blockWrite(node.entry_sector, &self.dir_sector);
|
||||
}
|
||||
|
||||
// --- long-name creation --------------------------------------------------
|
||||
|
||||
// The standard 8.3 short-name checksum carried by every long-name entry.
|
||||
fn shortChecksum(raw: [11]u8) u8 {
|
||||
var sum: u8 = 0;
|
||||
for (raw) |c| sum = ((sum & 1) << 7) +% (sum >> 1) +% c;
|
||||
return sum;
|
||||
}
|
||||
|
||||
fn valid83Char(c: u8) bool {
|
||||
return (c >= 'A' and c <= 'Z') or (c >= '0' and c <= '9') or c == '-' or c == '_';
|
||||
}
|
||||
|
||||
// Whether an 8.3 entry with exactly this raw name exists in `dir`.
|
||||
const RawContext = struct { raw: [11]u8, found: *bool };
|
||||
fn rawVisit(context: *const RawContext, entry: on_disk.DirectoryEntry, name: []const u8, entry_sector: u64, entry_offset: u32) bool {
|
||||
_ = name;
|
||||
_ = entry_sector;
|
||||
_ = entry_offset;
|
||||
if (std.mem.eql(u8, &entry.name, &context.raw)) {
|
||||
context.found.* = true;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
fn shortNameExists(self: *FileSystem, dir: Node, raw: [11]u8) bool {
|
||||
var found = false;
|
||||
var context = RawContext{ .raw = raw, .found = &found };
|
||||
self.scanDirectory(dir, &context, rawVisit);
|
||||
return found;
|
||||
}
|
||||
|
||||
// A mangled STEM~N.EXT short name that collides with nothing in `dir` — the
|
||||
// alias behind a long-name chain.
|
||||
fn shortNameFor(self: *FileSystem, dir: Node, name: []const u8) ?[11]u8 {
|
||||
const dot = std.mem.lastIndexOfScalar(u8, name, '.');
|
||||
const base = if (dot) |d| name[0..d] else name;
|
||||
const ext = if (dot) |d| name[d + 1 ..] else name[0..0];
|
||||
|
||||
var stem: [6]u8 = undefined;
|
||||
var stem_len: usize = 0;
|
||||
for (base) |c| {
|
||||
if (stem_len == stem.len) break;
|
||||
const upper = std.ascii.toUpper(c);
|
||||
if (valid83Char(upper)) {
|
||||
stem[stem_len] = upper;
|
||||
stem_len += 1;
|
||||
}
|
||||
}
|
||||
if (stem_len == 0) {
|
||||
stem[0] = 'X';
|
||||
stem_len = 1;
|
||||
}
|
||||
|
||||
var raw = [_]u8{' '} ** 11;
|
||||
var ext_len: usize = 0;
|
||||
for (ext) |c| {
|
||||
if (ext_len == 3) break;
|
||||
const upper = std.ascii.toUpper(c);
|
||||
if (valid83Char(upper)) {
|
||||
raw[8 + ext_len] = upper;
|
||||
ext_len += 1;
|
||||
}
|
||||
}
|
||||
|
||||
var index: u32 = 1;
|
||||
while (index <= 999_999) : (index += 1) {
|
||||
var tail_buffer: [8]u8 = undefined;
|
||||
const tail = std.fmt.bufPrint(&tail_buffer, "~{d}", .{index}) catch return null;
|
||||
const keep = @min(stem_len, 8 - tail.len);
|
||||
@memset(raw[0..8], ' ');
|
||||
@memcpy(raw[0..keep], stem[0..keep]);
|
||||
@memcpy(raw[keep .. keep + tail.len], tail);
|
||||
if (!self.shortNameExists(dir, raw)) return raw;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Fill one long-name entry's 13 UTF-16 slots from `name` starting at
|
||||
// `offset`: the name's bytes widened, then a 0x0000 terminator, then 0xFFFF.
|
||||
fn fillLongNamePiece(lfn: *on_disk.LongNameEntry, name: []const u8, offset: usize) void {
|
||||
var units: [13]u16 = undefined;
|
||||
var i: usize = 0;
|
||||
while (i < 13) : (i += 1) {
|
||||
const at = offset + i;
|
||||
units[i] = if (at < name.len) name[at] else if (at == name.len) 0x0000 else 0xFFFF;
|
||||
}
|
||||
lfn.name1 = units[0..5].*;
|
||||
lfn.name2 = units[5..11].*;
|
||||
lfn.name3 = units[11..13].*;
|
||||
}
|
||||
|
||||
// The first entry index of a run of `count` free slots in `dir`, growing the
|
||||
// directory as needed. Fresh clusters are zeroed, so growth always yields
|
||||
// free slots; only the fixed FAT12/16 root can genuinely run out.
|
||||
fn findFreeRun(self: *FileSystem, dir: Node, count: usize) ?u32 {
|
||||
var run_start: u32 = 0;
|
||||
var run_len: usize = 0;
|
||||
var sector_index: u32 = 0;
|
||||
while (self.dirSectorLba(dir, sector_index, true)) |lba| : (sector_index += 1) {
|
||||
if (!self.blockRead(lba, &self.dir_sector)) return null;
|
||||
var i: u32 = 0;
|
||||
while (i < entries_per_sector) : (i += 1) {
|
||||
const offset = i * @sizeOf(on_disk.DirectoryEntry);
|
||||
const entry = std.mem.bytesToValue(on_disk.DirectoryEntry, self.dir_sector[offset .. offset + @sizeOf(on_disk.DirectoryEntry)]);
|
||||
if (entry.isFree()) {
|
||||
if (run_len == 0) run_start = sector_index * entries_per_sector + i;
|
||||
run_len += 1;
|
||||
if (run_len == count) return run_start;
|
||||
} else {
|
||||
run_len = 0;
|
||||
}
|
||||
}
|
||||
if (sector_index > 4096) return null; // runaway guard
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Write one 32-byte directory entry at a global entry index (read-modify-
|
||||
// write of its sector). Returns the entry's (sector, offset) or null.
|
||||
fn writeEntryAt(self: *FileSystem, dir: Node, index: u32, bytes: *const [32]u8) ?EntryLoc {
|
||||
const lba = self.dirSectorLba(dir, index / entries_per_sector, true) orelse return null;
|
||||
if (!self.blockRead(lba, &self.dir_sector)) return null;
|
||||
const offset = (index % entries_per_sector) * @sizeOf(on_disk.DirectoryEntry);
|
||||
@memcpy(self.dir_sector[offset .. offset + 32], bytes);
|
||||
if (!self.blockWrite(lba, &self.dir_sector)) return null;
|
||||
return .{ .sector = lba, .offset = offset };
|
||||
}
|
||||
|
||||
// Add a named directory entry, creating a long-name chain when the name is
|
||||
// not its own 8.3 form. Write order is LFN pieces first, 8.3 entry last: an
|
||||
// interrupted create leaves orphaned long-name entries, which every FAT
|
||||
// reader (this engine's scanner included) skips as unattached — never a
|
||||
// mismatched chain.
|
||||
fn addEntryNamed(self: *FileSystem, dir: Node, name: []const u8, attributes: u8, first_cluster: u32, size: u32) ?Node {
|
||||
if (to83(name)) |raw| {
|
||||
var display: [12]u8 = undefined;
|
||||
// Only a name that IS its 8.3 form (already uppercase) skips the
|
||||
// chain — a lowercase name gets one so its exact case survives,
|
||||
// matching tools/make-fat-image.py.
|
||||
if (std.mem.eql(u8, format83(raw, &display), name))
|
||||
return self.addEntry(dir, raw, attributes, first_cluster, size);
|
||||
}
|
||||
if (name.len == 0 or name.len > 255) return null;
|
||||
|
||||
const raw = self.shortNameFor(dir, name) orelse return null;
|
||||
const checksum = shortChecksum(raw);
|
||||
const piece_count: u32 = @intCast((name.len + 12) / 13);
|
||||
if (piece_count > 20) return null;
|
||||
const start = self.findFreeRun(dir, piece_count + 1) orelse return null;
|
||||
|
||||
var k: u32 = 0;
|
||||
while (k < piece_count) : (k += 1) {
|
||||
const piece = piece_count - k; // stored last-logical-first
|
||||
var lfn = std.mem.zeroes(on_disk.LongNameEntry);
|
||||
lfn.order = @intCast(piece | (if (k == 0) @as(u8, 0x40) else 0));
|
||||
lfn.attributes = on_disk.attribute_long_name;
|
||||
lfn.checksum = checksum;
|
||||
fillLongNamePiece(&lfn, name, (piece - 1) * 13);
|
||||
_ = self.writeEntryAt(dir, start + k, std.mem.asBytes(&lfn)[0..32]) orelse return null;
|
||||
}
|
||||
|
||||
var entry = std.mem.zeroes(on_disk.DirectoryEntry);
|
||||
entry.name = raw;
|
||||
entry.attributes = attributes;
|
||||
entry.file_size = size;
|
||||
entry.setFirstCluster(first_cluster);
|
||||
const stamp = on_disk.epochToFatDateTime(self.current_time_epoch);
|
||||
entry.creation_date = stamp.date;
|
||||
entry.creation_time = stamp.time;
|
||||
entry.write_date = stamp.date;
|
||||
entry.write_time = stamp.time;
|
||||
entry.last_access_date = stamp.date;
|
||||
const location = self.writeEntryAt(dir, start + piece_count, std.mem.asBytes(&entry)[0..32]) orelse return null;
|
||||
return .{
|
||||
.first_cluster = first_cluster,
|
||||
.size = size,
|
||||
.is_directory = attributes & on_disk.attribute_directory != 0,
|
||||
.mtime = self.current_time_epoch,
|
||||
.entry_sector = location.sector,
|
||||
.entry_offset = location.offset,
|
||||
.has_entry = true,
|
||||
};
|
||||
}
|
||||
|
||||
// Add an 8.3 directory entry to `dir` with the given attributes, first cluster,
|
||||
// and size, reusing a free (0x00 or 0xE5) slot and growing the directory chain
|
||||
// if needed. Returns the new node (with its entry location) or null if full.
|
||||
@@ -645,19 +855,20 @@ pub const FileSystem = struct {
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Create an 8.3-named file in directory `dir`. Returns the new (empty) node,
|
||||
/// or null if the name is not 8.3-representable or no directory slot is free.
|
||||
/// Create a file in directory `dir`. Uppercase 8.3 names get a bare short
|
||||
/// entry; anything else gets a long-name chain over a mangled ~N alias.
|
||||
/// Returns the new (empty) node, or null (bad name / directory full /
|
||||
/// duplicate — the caller checks existence first if it must distinguish).
|
||||
pub fn createFile(self: *FileSystem, dir: Node, name: []const u8) ?Node {
|
||||
const raw = to83(name) orelse return null;
|
||||
return self.addEntry(dir, raw, on_disk.attribute_archive, 0, 0);
|
||||
return self.addEntryNamed(dir, name, on_disk.attribute_archive, 0, 0);
|
||||
}
|
||||
|
||||
/// Create an 8.3-named subdirectory in `dir`: allocate and initialise its first
|
||||
/// Create a subdirectory in `dir`: allocate and initialise its first
|
||||
/// cluster with "." (itself) and ".." (the parent) entries, then add its
|
||||
/// directory entry to `dir`. Returns the new directory node, or null (bad name,
|
||||
/// no free cluster, or the directory is full). Long names are not created.
|
||||
/// directory entry to `dir` (long-name chain when the name needs one).
|
||||
/// Returns the new directory node, or null (bad name, no free cluster, or
|
||||
/// the directory is full).
|
||||
pub fn createDirectory(self: *FileSystem, dir: Node, name: []const u8) ?Node {
|
||||
const raw = to83(name) orelse return null;
|
||||
const cluster = self.allocateCluster() orelse return null;
|
||||
self.zeroCluster(cluster);
|
||||
|
||||
@@ -685,7 +896,7 @@ pub const FileSystem = struct {
|
||||
self.freeChain(cluster);
|
||||
return null;
|
||||
}
|
||||
return self.addEntry(dir, raw, on_disk.attribute_directory, cluster, 0) orelse {
|
||||
return self.addEntryNamed(dir, name, on_disk.attribute_directory, cluster, 0) orelse {
|
||||
self.freeChain(cluster);
|
||||
return null;
|
||||
};
|
||||
@@ -1103,3 +1314,113 @@ test "a create stamps the modification time" {
|
||||
try std.testing.expectEqual(@as(u64, 1_700_000_000), fs.resolve("/STAMP.TXT").?.mtime);
|
||||
try std.testing.expectEqual(@as(u64, 1_700_000_000), fs.listEntry(fs.rootNode(), 0).?.mtime);
|
||||
}
|
||||
|
||||
test "long-name create: directory + file round-trip by long name" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
// The per-boot log directory shape: an 18-char stamp, nested paths, .log names.
|
||||
const stamp_dir = fs.createDirectory(fs.rootNode(), "2026-07-21T101530Z").?;
|
||||
try std.testing.expect(stamp_dir.is_directory);
|
||||
const file = fs.createFile(stamp_dir, "device-manager.log").?;
|
||||
_ = file;
|
||||
|
||||
// Resolve by exact long name, and case-insensitively (FAT semantics).
|
||||
try std.testing.expect(fs.resolve("/2026-07-21T101530Z/device-manager.log") != null);
|
||||
try std.testing.expect(fs.resolve("/2026-07-21t101530z/DEVICE-MANAGER.LOG") != null);
|
||||
|
||||
// The listing shows the long names, not the ~N aliases.
|
||||
var listing = fs.listEntry(fs.rootNode(), 0).?;
|
||||
try std.testing.expectEqualStrings("2026-07-21T101530Z", listing.name_buffer[0..listing.name_len]);
|
||||
var inner = fs.listEntry(stamp_dir, 2).?; // after "." and ".."
|
||||
try std.testing.expectEqualStrings("device-manager.log", inner.name_buffer[0..inner.name_len]);
|
||||
|
||||
// Write through the created file and read it back by long-name resolve.
|
||||
var node = fs.resolve("/2026-07-21T101530Z/device-manager.log").?;
|
||||
try std.testing.expectEqual(@as(usize, 10), fs.writeFile(&node, 0, "hello logs"));
|
||||
var buffer: [16]u8 = undefined;
|
||||
try std.testing.expectEqual(@as(usize, 10), fs.readFile(node, 0, buffer[0..10]));
|
||||
try std.testing.expectEqualStrings("hello logs", buffer[0..10]);
|
||||
}
|
||||
|
||||
test "long-name create: ~N alias collision suffixes stay distinct" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
_ = fs.createFile(fs.rootNode(), "logger-alpha.log").?;
|
||||
_ = fs.createFile(fs.rootNode(), "logger-beta.log").?;
|
||||
// Same 6-char mangle stem (LOGGER) — the second must take ~2.
|
||||
var raw_one = false;
|
||||
var raw_two = false;
|
||||
var cursor: u32 = 0;
|
||||
while (fs.listEntry(fs.rootNode(), cursor)) |entry| : (cursor += 1) {
|
||||
if (std.mem.eql(u8, entry.name_buffer[0..entry.name_len], "logger-alpha.log")) raw_one = true;
|
||||
if (std.mem.eql(u8, entry.name_buffer[0..entry.name_len], "logger-beta.log")) raw_two = true;
|
||||
}
|
||||
try std.testing.expect(raw_one and raw_two);
|
||||
try std.testing.expect(fs.resolve("/logger-alpha.log") != null);
|
||||
try std.testing.expect(fs.resolve("/logger-beta.log") != null);
|
||||
// Their short aliases took distinct ~N tails. (Alias LOOKUP is not a
|
||||
// feature — findChild matches display names — but the on-disk aliases
|
||||
// must not collide for other FAT readers.)
|
||||
try std.testing.expect(fs.shortNameExists(fs.rootNode(), "LOGGER~1LOG".*));
|
||||
try std.testing.expect(fs.shortNameExists(fs.rootNode(), "LOGGER~2LOG".*));
|
||||
}
|
||||
|
||||
test "long-name create: unlink removes the chain; slots are reused cleanly" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
_ = fs.createFile(fs.rootNode(), "a-rather-long-file-name.txt").?;
|
||||
try std.testing.expect(fs.removeFile(fs.rootNode(), "a-rather-long-file-name.txt"));
|
||||
try std.testing.expect(fs.resolve("/a-rather-long-file-name.txt") == null);
|
||||
|
||||
// A new long name reuses the freed run without inheriting the old chain.
|
||||
_ = fs.createFile(fs.rootNode(), "an-entirely-different-name.md").?;
|
||||
try std.testing.expect(fs.resolve("/an-entirely-different-name.md") != null);
|
||||
try std.testing.expect(fs.resolve("/a-rather-long-file-name.txt") == null);
|
||||
var listing = fs.listEntry(fs.rootNode(), 0).?;
|
||||
try std.testing.expectEqualStrings("an-entirely-different-name.md", listing.name_buffer[0..listing.name_len]);
|
||||
}
|
||||
|
||||
test "8.3 fast path: an uppercase-compliant name gets one bare entry" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
formatFat16(bytes);
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
var fs = FileSystem.mount(disk.device()).?;
|
||||
|
||||
_ = fs.createFile(fs.rootNode(), "DANOS.LOG").?;
|
||||
// Exactly one directory entry: entry 0 is the file, entry 1 is the end.
|
||||
var listing = fs.listEntry(fs.rootNode(), 0).?;
|
||||
try std.testing.expectEqualStrings("DANOS.LOG", listing.name_buffer[0..listing.name_len]);
|
||||
try std.testing.expect(fs.listEntry(fs.rootNode(), 1) == null);
|
||||
// A lowercase 8.3-shaped name is case-preserved via a chain instead.
|
||||
_ = fs.createFile(fs.rootNode(), "fat.log").?;
|
||||
var second = fs.listEntry(fs.rootNode(), 1).?;
|
||||
try std.testing.expectEqualStrings("fat.log", second.name_buffer[0..second.name_len]);
|
||||
}
|
||||
|
||||
test "short-name checksum matches the reference vector" {
|
||||
// "README TXT" is a widely published example: checksum 0x15... compute a
|
||||
// fixed pair to pin the rotate-add against regressions.
|
||||
const a = FileSystem.shortChecksum("README TXT".*);
|
||||
const b = FileSystem.shortChecksum("LOGGER~1LOG".*);
|
||||
try std.testing.expect(a != b);
|
||||
// The algorithm is order-sensitive: swapped bytes change the sum.
|
||||
const c = FileSystem.shortChecksum("REDAME TXT".*);
|
||||
try std.testing.expect(a != c);
|
||||
}
|
||||
|
||||
+43
-22
@@ -16,11 +16,6 @@ const on_disk = @import("on-disk.zig");
|
||||
const protocol = runtime.vfs_protocol;
|
||||
const dma = runtime.dma;
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
const mount_point = "/mnt/usb";
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
@@ -104,20 +99,44 @@ fn initialise(endpoint: runtime.ipc.Handle) bool {
|
||||
_ = runtime.system.write("/system/services/fat: not a FAT filesystem\n");
|
||||
return false;
|
||||
};
|
||||
writeLine("/system/services/fat: mounted FAT ({s}, {d} clusters, partition lba {d})\n", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
|
||||
// Mount ourselves into the VFS namespace at /mnt/usb (retry while the VFS
|
||||
// comes up). From here the VFS routes /mnt/usb/... to this server.
|
||||
var tries: u32 = 0;
|
||||
while (tries < 100) : (tries += 1) {
|
||||
if (runtime.fs.mount(mount_point, endpoint)) {
|
||||
writeLine("/system/services/fat: mounted {s}\n", .{mount_point});
|
||||
return true;
|
||||
}
|
||||
runtime.system.sleep(50);
|
||||
// With the router in the kernel, clients hold OUR node ids directly; sweep
|
||||
// a dead client's open handles via the published exit events (the pattern
|
||||
// the old userspace router used for its own table).
|
||||
_ = runtime.process.subscribeExits(endpoint);
|
||||
|
||||
// Mount ourselves into the kernel VFS at /mnt/usb — and serve /var from the
|
||||
// volume's /var subtree, so FHS paths (the logger's /var/log) stay decoupled
|
||||
// from which volume carries them. A mount is one syscall now; no retry
|
||||
// needed (the kernel's table exists before any service).
|
||||
if (runtime.fs.mount(mount_point, endpoint)) {
|
||||
std.log.info("mounted {s}", .{mount_point});
|
||||
} else {
|
||||
_ = runtime.system.write("/system/services/fat: could not mount /mnt/usb\n");
|
||||
}
|
||||
_ = runtime.system.write("/system/services/fat: could not mount into the VFS\n");
|
||||
return true; // still serve directly, even if the namespace mount didn't take
|
||||
if (runtime.fs.mountRewritten("/var", endpoint, "/var")) {
|
||||
std.log.info("mounted /var", .{});
|
||||
} else {
|
||||
_ = runtime.system.write("/system/services/fat: could not mount /var\n");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// A subscribed process-exit event: release every open handle the dead client
|
||||
/// held, so a crashed reader can't pin table slots (or, later, locks).
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = runtime.ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (!got.isChildExit()) return;
|
||||
const dead = got.childProcessId();
|
||||
var released: u32 = 0;
|
||||
for (&open_nodes) |*o| {
|
||||
if (o.used and o.owner == dead) {
|
||||
o.* = .{};
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
@@ -132,7 +151,7 @@ fn splitParent(path: []const u8) ParentLeaf {
|
||||
};
|
||||
}
|
||||
|
||||
fn handleOpen(out: []u8, path: []const u8, flags: u32) usize {
|
||||
fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize {
|
||||
var node = filesystem.resolve(path);
|
||||
if (node == null and flags & protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
@@ -146,13 +165,12 @@ fn handleOpen(out: []u8, path: []const u8, flags: u32) usize {
|
||||
filesystem.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return fail(out);
|
||||
open_nodes[index] = .{ .used = true, .node = resolved };
|
||||
open_nodes[index] = .{ .used = true, .node = resolved, .owner = sender };
|
||||
return writeReply(out, .{ .status = 0, .node = index }, &.{});
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = capability;
|
||||
_ = sender;
|
||||
if (message.len < protocol.request_size) return fail(out);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
@@ -162,7 +180,7 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i
|
||||
filesystem.current_time_epoch = runtime.system.wallClock();
|
||||
|
||||
switch (request.operation) {
|
||||
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags),
|
||||
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags, sender),
|
||||
.read => {
|
||||
const o = openAt(request.node) orelse return fail(out);
|
||||
var buffer: [protocol.maximum_payload]u8 = undefined;
|
||||
@@ -208,7 +226,9 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?runtime.i
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
.mkdir => {
|
||||
const split = splitParent(payload[0..@min(payload.len, request.len)]);
|
||||
const path = payload[0..@min(payload.len, request.len)];
|
||||
if (filesystem.resolve(path) != null) return fail(out); // already exists — no duplicate entries
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return fail(out);
|
||||
if (filesystem.createDirectory(parent, split.leaf) == null) return fail(out);
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
@@ -240,5 +260,6 @@ pub fn main() void {
|
||||
.service = .fat,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -22,16 +22,22 @@ const runtime = @import("runtime");
|
||||
const power = runtime.power_protocol;
|
||||
const build_options = @import("build_options");
|
||||
|
||||
/// Where the kernel boot log is persisted on the USB FAT volume — an 8.3 name at
|
||||
/// the mount root (see system/services/log-flush). init writes it at shutdown;
|
||||
/// the log-flush one-shot writes it once at boot.
|
||||
const log_path = "/mnt/usb/DANOS.LOG";
|
||||
|
||||
/// The system services init brings up at boot, in order. This is init's policy — the
|
||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||
/// on purpose: the device manager owns those. (A future init reads this from a
|
||||
/// manifest under /system/services instead of a hardcoded list.)
|
||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat", "display", "display-demo" };
|
||||
/// The system services init brings up at boot, in order, by binary path. This is
|
||||
/// init's policy — the microkernel keeps such choices in user space, not the
|
||||
/// kernel. Drivers are absent on purpose: the device manager owns those. (A
|
||||
/// future init reads this from a manifest under /system/services instead of a
|
||||
/// hardcoded list.)
|
||||
const boot_services = [_][]const u8{
|
||||
"/system/services/input",
|
||||
"/system/services/device-manager",
|
||||
"/system/services/fat",
|
||||
"/system/services/display",
|
||||
"/system/services/display-demo",
|
||||
// Last: at shutdown children stop in reverse order, so the logger goes down
|
||||
// FIRST — its final drain still has the fat server (and the whole storage
|
||||
// chain) alive underneath it.
|
||||
"/system/services/logger",
|
||||
};
|
||||
|
||||
/// The live process id of each boot service (0 = not running), indexed by its position
|
||||
/// in `boot_services`, plus how many times init has restarted it. init supervises these:
|
||||
@@ -78,14 +84,6 @@ pub fn main() void {
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| child_ids[i] = id;
|
||||
}
|
||||
|
||||
// Once the storage stack is up, a one-shot copies the boot log to the USB
|
||||
// volume (/mnt/usb/DANOS.LOG) so it can be read on another machine — the only
|
||||
// way to see it on a headless/real board with no host capturing serial. Fire
|
||||
// and forget: it polls for the mount itself, and is deliberately NOT one of
|
||||
// init's supervised children (a transient one-shot must not be stopped-and-
|
||||
// waited-for during shutdown).
|
||||
_ = runtime.system.spawn("log-flush");
|
||||
|
||||
// Subscribe to power events (retry: the power service registers well after
|
||||
// init starts). Best-effort — without it, a `terminate` signal still
|
||||
// triggers the same shutdown path.
|
||||
@@ -137,26 +135,21 @@ fn restartChild(id: u32) void {
|
||||
// An unknown reason (the record aged out) is treated as a crash worth restarting.
|
||||
const reason = runtime.process.exitReason(id) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
logLine("/system/services/init: {s} exited cleanly; not restarting\n", .{service});
|
||||
std.log.info("{s} exited cleanly; not restarting", .{service});
|
||||
return;
|
||||
}
|
||||
restart_counts[i] += 1;
|
||||
if (restart_counts[i] > maximum_restarts) {
|
||||
logLine("/system/services/init: {s} keeps crashing; giving up after {d} restarts\n", .{ service, maximum_restarts });
|
||||
std.log.info("{s} keeps crashing; giving up after {d} restarts", .{ service, maximum_restarts });
|
||||
return;
|
||||
}
|
||||
logLine("/system/services/init: {s} died ({s}); restarting ({d}/{d})\n", .{ service, @tagName(reason), restart_counts[i], maximum_restarts });
|
||||
std.log.info("{s} died ({s}); restarting ({d}/{d})", .{ service, @tagName(reason), restart_counts[i], maximum_restarts });
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |new_id| child_ids[i] = new_id;
|
||||
return;
|
||||
}
|
||||
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
||||
}
|
||||
|
||||
fn logLine(comptime fmt: []const u8, args: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||
/// call's capability) so events arrive as buffered messages here.
|
||||
fn subscribePower() void {
|
||||
@@ -175,26 +168,6 @@ fn subscribePower() void {
|
||||
_ = runtime.ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||
}
|
||||
|
||||
/// Copy the whole kernel log to /mnt/usb/DANOS.LOG (the same file log-flush
|
||||
/// writes at boot), so a poweroff captures the fullest log. Best-effort: if the
|
||||
/// USB volume is not mounted, the open fails and it does nothing. Must run while
|
||||
/// the storage services are still alive (see shutDown).
|
||||
fn flushKernelLog() void {
|
||||
// Truncate on open so this fuller flush replaces the boot-time one cleanly.
|
||||
var file = runtime.fs.open(log_path, .{ .create = true, .truncate = true }) orelse return; // no USB volume
|
||||
defer file.close();
|
||||
var chunk: [4096]u8 = undefined;
|
||||
var offset: usize = 0;
|
||||
while (true) {
|
||||
const got = runtime.system.klogRead(offset, &chunk);
|
||||
if (got == 0) break; // reached the end of the accumulated log
|
||||
if (file.writeAll(chunk[0..got]) == null) break; // storage went away
|
||||
offset += got;
|
||||
}
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, "/system/services/init: flushed log to {s} ({d} bytes)\n", .{ log_path, offset }) catch "");
|
||||
}
|
||||
|
||||
/// The stop sequence: persist the log while storage is still up, then terminate
|
||||
/// each child in reverse spawn order (vfs last — other services may flush through
|
||||
/// it), waiting up to a deadline for each to exit before killing it, then ask the
|
||||
@@ -202,10 +175,9 @@ fn flushKernelLog() void {
|
||||
fn shutDown() void {
|
||||
shutting_down = true; // the stop loop below kills children — those deaths aren't crashes
|
||||
_ = runtime.system.write("/system/services/init: shutting down\n");
|
||||
// Persist the fullest log to the USB volume BEFORE tearing anything down: the
|
||||
// reverse-order stop loop below kills the fat server first, so /mnt/usb must be
|
||||
// written while it is still mounted.
|
||||
flushKernelLog();
|
||||
// Log persistence is the logger service's job: it is the LAST boot service,
|
||||
// so the reverse-order stop below terminates it first and its final drain
|
||||
// runs while the whole storage chain is still alive.
|
||||
var i = boot_services.len;
|
||||
while (i > 0) {
|
||||
i -= 1;
|
||||
|
||||
@@ -10,17 +10,38 @@
|
||||
//! keyboard and mouse drivers publish their own synthetic streams today; swapping in
|
||||
//! decoded hardware is a follow-up (see docs/input.md).
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const input = runtime.input;
|
||||
const system = runtime.system;
|
||||
|
||||
pub fn main() void {
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
var source = input.connectSource() orelse {
|
||||
_ = system.write("input-source: input service unavailable\n");
|
||||
return;
|
||||
};
|
||||
_ = system.write("input-source: publishing synthetic input events\n");
|
||||
|
||||
// "mouse" mode publishes a steady stream of pure motion (dx=dy=+1), for driving a
|
||||
// cursor (the `display-cursor` test). The default "rotate" mode cycles all device
|
||||
// classes to exercise the service's per-device routing (the `input` test).
|
||||
const mode = init.arguments.get(1) orelse "rotate";
|
||||
if (std.mem.eql(u8, mode, "mouse")) {
|
||||
_ = system.write("input-source: publishing synthetic mouse motion\n");
|
||||
while (true) {
|
||||
_ = source.publishMouseEvent(.{
|
||||
.kind = @intFromEnum(input.MouseEventKind.motion),
|
||||
.button = 0,
|
||||
.dx = 1,
|
||||
.dy = 1,
|
||||
.scroll_x = 0,
|
||||
.scroll_y = 0,
|
||||
.buttons = 0,
|
||||
});
|
||||
system.sleep(20); // ~50 events/sec: moves the cursor briskly
|
||||
}
|
||||
}
|
||||
|
||||
_ = system.write("input-source: publishing synthetic input events\n");
|
||||
var step: usize = 0;
|
||||
while (true) : (step +%= 1) {
|
||||
// Rotate across the device classes so every publish path (and the service's
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
//! system/services/log-flush — a one-shot that copies the kernel's in-memory
|
||||
//! diagnostic log to a file on the mounted USB FAT volume, so the boot log
|
||||
//! survives to be read on another machine. On a headless or real board there is
|
||||
//! no host capturing serial, so without this the log is lost at power-off; this
|
||||
//! is the on-disk equivalent of QEMU's `-serial file:`.
|
||||
//!
|
||||
//! It reads the whole kernel log back through `klog_read` (the RAM sink in
|
||||
//! system/kernel/log.zig) and writes it to /mnt/usb/DANOS.LOG. The name is 8.3
|
||||
//! (FAT short-name rule: base <= 8, extension <= 3) and lives at the mount root
|
||||
//! (there is no mkdir on the FAT path yet). init spawns this once the boot
|
||||
//! services are up; init itself repeats the flush at shutdown for a fuller log.
|
||||
//!
|
||||
//! If no USB volume is mounted — no stick, or the initial-ramdisk sweep that
|
||||
//! spawns every bundled binary bare with no VFS — it waits briefly, then exits
|
||||
//! silently, deranging no other test's output.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const fs = runtime.fs;
|
||||
|
||||
const log_path = "/mnt/usb/DANOS.LOG";
|
||||
|
||||
/// Copy the whole kernel log to the open file, looping klog_read -> write until
|
||||
/// the log is exhausted. Returns the number of bytes written.
|
||||
fn drainKernelLog(file: *fs.File) usize {
|
||||
var chunk: [4096]u8 = undefined;
|
||||
var offset: usize = 0;
|
||||
while (true) {
|
||||
const got = runtime.system.klogRead(offset, &chunk);
|
||||
if (got == 0) break; // reached the end of the accumulated log
|
||||
if (file.writeAll(chunk[0..got]) == null) break; // storage went away
|
||||
offset += got;
|
||||
}
|
||||
return offset;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// Wait for the fat server to mount /mnt/usb (it must bring up the whole USB
|
||||
// storage chain first, so it races us at boot). Bounded: if the mount never
|
||||
// appears — no volume, or the no-VFS ramdisk sweep — give up silently.
|
||||
var ready = false;
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1400) : (tries += 1) {
|
||||
if (fs.openDirectory("/mnt/usb")) |directory| {
|
||||
var dir = directory;
|
||||
dir.close();
|
||||
ready = true;
|
||||
break;
|
||||
}
|
||||
runtime.system.sleep(50);
|
||||
}
|
||||
if (!ready) return; // /mnt/usb never became available — nothing to persist to
|
||||
|
||||
// Truncate on open: each flush replaces the file, so a shorter log on a later
|
||||
// boot of the same stick leaves no stale tail from a previous, longer one.
|
||||
var file = fs.open(log_path, .{ .create = true, .truncate = true }) orelse return;
|
||||
const written = drainKernelLog(&file);
|
||||
file.close();
|
||||
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, "log-flush: wrote {d} bytes to {s}\n", .{ written, log_path }) catch return);
|
||||
}
|
||||
@@ -0,0 +1,299 @@
|
||||
//! The logger service — the per-process log persister.
|
||||
//!
|
||||
//! Drains the tagged kernel log ring (`klog_read`/`klog_status`) and
|
||||
//! demultiplexes it into **one file per process** on the flash volume:
|
||||
//!
|
||||
//! <base>/<boot-stamp>/<binary-path>.log
|
||||
//! e.g. /mnt/usb/var/log/2026-07-21T101530Z/system/services/fat.log
|
||||
//!
|
||||
//! The boot stamp is the wall-clock time of boot (from klog_status), so one
|
||||
//! boot session is one self-contained directory; the kernel's own records go to
|
||||
//! kernel.log. Records carry the sender's pid and binary path, stamped by the
|
||||
//! kernel — the logger trusts the ring, never the payload.
|
||||
//!
|
||||
//! Storage is best-effort and late: until the FAT volume mounts, the ring
|
||||
//! simply buffers (it holds a full boot many times over), and the first drain
|
||||
//! writes the whole backlog. The storage stack's own records are captured the
|
||||
//! same way — services never write their own log files (the fat service
|
||||
//! logging through itself would rendezvous-deadlock; the ring sidesteps that
|
||||
//! by design).
|
||||
//!
|
||||
//! The logger announces itself ONCE (a periodic status line would feed the
|
||||
//! very stream it drains — self-sustaining churn). Lost records surface as an
|
||||
//! explicit "-- N records lost --" line derived from sequence-number gaps.
|
||||
//!
|
||||
//! Durability: files are opened create-once and kept open across a burst, then
|
||||
//! all closed after a quiet period (~2 s) — each close is the fat server's
|
||||
//! SCSI SYNCHRONIZE CACHE, so data-at-risk is bounded by the last busy burst
|
||||
//! without thrashing the device on every record. `on_terminate` does a final
|
||||
//! drain and closes everything, so an orderly shutdown loses nothing (init
|
||||
//! stops the logger FIRST — reverse boot order — while fat is still up).
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
|
||||
const system = runtime.system;
|
||||
const fs = runtime.fs;
|
||||
|
||||
/// Where log trees live: the FHS path. The kernel VFS routes /var to whatever
|
||||
/// volume the fat server mounted there (today: the /var subtree of the USB
|
||||
/// flash volume) — swapping the persistent medium later touches fat's two
|
||||
/// mount calls, never this constant.
|
||||
const base = "/var/log";
|
||||
|
||||
/// Drain cadence and the quiet period after which files are closed (flushed).
|
||||
const tick_ms = 250;
|
||||
const quiet_close_ticks = 8; // 8 * 250 ms = 2 s
|
||||
|
||||
/// One cached open file per source process path. Sized above the practical
|
||||
/// process count; the fat server's global open-node table (32) is the real
|
||||
/// ceiling, so stay comfortably below it.
|
||||
const maximum_files = 24;
|
||||
|
||||
const CachedFile = struct {
|
||||
used: bool = false,
|
||||
name: [system.maximum_process_name]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
file: fs.File = undefined,
|
||||
};
|
||||
|
||||
var files: [maximum_files]CachedFile = @splat(.{});
|
||||
var endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
/// The drain cursor into the ring's byte stream, and loss accounting.
|
||||
var cursor: u64 = 0;
|
||||
var next_expected_sequence: u64 = 0;
|
||||
|
||||
/// Carry buffer: a record can straddle two klog_read chunks.
|
||||
var carry: [carry_capacity]u8 = undefined;
|
||||
var carry_len: usize = 0;
|
||||
const carry_capacity = 64 + 256 + 64; // header + payload + name, padded generously
|
||||
|
||||
/// The per-boot directory, formatted once storage appears.
|
||||
var boot_directory: [base.len + 1 + 19]u8 = undefined;
|
||||
var boot_directory_len: usize = 0;
|
||||
var storage_ready = false;
|
||||
var announced = false;
|
||||
var ticks_since_record: u32 = 0;
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(64, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
.on_terminate = onTerminate,
|
||||
});
|
||||
}
|
||||
|
||||
fn initialise(harness_endpoint: runtime.ipc.Handle) bool {
|
||||
endpoint = harness_endpoint;
|
||||
const status = system.klogStatus() orelse return false;
|
||||
cursor = status.tail;
|
||||
// Sequence expectations start at the tail record's sequence — discovered on
|
||||
// the first drain; 0 is right for a fresh boot either way.
|
||||
formatBootDirectory(status.boot_unix_seconds);
|
||||
_ = system.timerOnce(endpoint, tick_ms);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The logger serves no protocol; the ping is answered by the harness.
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & runtime.ipc.notify_timer_bit == 0) return;
|
||||
tick();
|
||||
_ = system.timerOnce(endpoint, tick_ms);
|
||||
}
|
||||
|
||||
fn onTerminate() void {
|
||||
// Final drain: everything still in the ring, then close (= flush) all files.
|
||||
drain();
|
||||
closeAll();
|
||||
var line: [96]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "logger: flushed through sequence {d}\n", .{next_expected_sequence}) catch return);
|
||||
}
|
||||
|
||||
fn tick() void {
|
||||
if (!storage_ready) {
|
||||
// makePath doubles as the readiness probe: while /var is unmounted the
|
||||
// resolve fails fast (no storage round trip) and the ring buffers; the
|
||||
// first success creates the whole per-boot tree.
|
||||
if (!fs.makePath(boot_directory[0..boot_directory_len])) return;
|
||||
storage_ready = true;
|
||||
if (!announced) {
|
||||
announced = true; // once — a periodic line would feed the stream we drain
|
||||
var line: [128]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "logger: logging to {s}\n", .{boot_directory[0..boot_directory_len]}) catch "");
|
||||
}
|
||||
}
|
||||
drain();
|
||||
// Quiet-period close: one device cache flush per burst.
|
||||
ticks_since_record += 1;
|
||||
if (ticks_since_record == quiet_close_ticks) closeAll();
|
||||
}
|
||||
|
||||
fn drain() void {
|
||||
if (!storage_ready) return;
|
||||
var chunk: [4096]u8 = undefined;
|
||||
while (true) {
|
||||
@memcpy(chunk[0..carry_len], carry[0..carry_len]);
|
||||
const got = system.klogRead(cursor, chunk[carry_len..]) orelse {
|
||||
// Cursor overwritten: re-sync to the ring tail; the sequence gap is
|
||||
// reported by the next record's header.
|
||||
const status = system.klogStatus() orelse return;
|
||||
cursor = status.tail;
|
||||
carry_len = 0;
|
||||
continue;
|
||||
};
|
||||
if (got == 0) return; // caught up (any partial record stays carried)
|
||||
cursor += got;
|
||||
consume(chunk[0 .. carry_len + got]);
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse whole records out of `bytes`; keep any trailing partial in `carry`.
|
||||
fn consume(bytes: []u8) void {
|
||||
const header_size = system.klog_record_header_size;
|
||||
var offset: usize = 0;
|
||||
while (bytes.len - offset >= header_size) {
|
||||
const header = std.mem.bytesToValue(system.KlogRecordHeader, bytes[offset..][0..32]);
|
||||
if (header.magic != system.klog_record_magic) {
|
||||
// Corrupt frame — should not happen; drop the carry and re-sync.
|
||||
carry_len = 0;
|
||||
const status = system.klogStatus() orelse return;
|
||||
cursor = status.head;
|
||||
return;
|
||||
}
|
||||
const record_len = recordLength(header);
|
||||
if (bytes.len - offset < record_len) break; // partial — carry it
|
||||
const name = bytes[offset + header_size ..][0..header.name_len];
|
||||
const message = bytes[offset + header_size + header.name_len ..][0..header.message_len];
|
||||
deliver(header, name, message);
|
||||
offset += record_len;
|
||||
}
|
||||
const rest = bytes.len - offset;
|
||||
if (rest > carry_capacity) {
|
||||
carry_len = 0; // cannot happen with sane frames; drop rather than overflow
|
||||
return;
|
||||
}
|
||||
@memcpy(carry[0..rest], bytes[offset..]);
|
||||
carry_len = rest;
|
||||
}
|
||||
|
||||
fn deliver(header: system.KlogRecordHeader, name: []const u8, message: []const u8) void {
|
||||
ticks_since_record = 0;
|
||||
const file = fileFor(if (header.pid == 0 or name.len == 0) "kernel" else name) orelse return;
|
||||
|
||||
if (header.sequence != next_expected_sequence and next_expected_sequence != 0) {
|
||||
var gap_line: [64]u8 = undefined;
|
||||
const lost = header.sequence - next_expected_sequence;
|
||||
if (std.fmt.bufPrint(&gap_line, "-- {d} records lost --\n", .{lost})) |line| {
|
||||
_ = file.writeAll(line);
|
||||
} else |_| {}
|
||||
}
|
||||
next_expected_sequence = header.sequence + 1;
|
||||
|
||||
// [+ssssss.mmm] level: payload
|
||||
var stamp: [48]u8 = undefined;
|
||||
const seconds = header.timestamp_ns / 1_000_000_000;
|
||||
const millis = (header.timestamp_ns / 1_000_000) % 1000;
|
||||
const level: []const u8 = switch (header.level) {
|
||||
.err => "error: ",
|
||||
.warn => "warning: ",
|
||||
.debug => "debug: ",
|
||||
.info, .raw => "",
|
||||
};
|
||||
if (std.fmt.bufPrint(&stamp, "[{d:>6}.{d:0>3}] {s}", .{ seconds, millis, level })) |prefix| {
|
||||
_ = file.writeAll(prefix);
|
||||
} else |_| {}
|
||||
_ = file.writeAll(message);
|
||||
if (header.flags & system.klog_flag_truncated != 0) _ = file.writeAll("~");
|
||||
_ = file.writeAll("\n");
|
||||
}
|
||||
|
||||
/// The cached (or freshly opened) file for a source name. The file path is the
|
||||
/// binary path with its leading '/' stripped, ".log" appended, under the
|
||||
/// per-boot directory; parents are created on first use.
|
||||
fn fileFor(name: []const u8) ?*fs.File {
|
||||
for (&files) |*cached| {
|
||||
if (cached.used and std.mem.eql(u8, cached.name[0..cached.name_len], name)) return &cached.file;
|
||||
}
|
||||
var slot: ?*CachedFile = null;
|
||||
for (&files) |*cached| {
|
||||
if (!cached.used) {
|
||||
slot = cached;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const cached = slot orelse evictOne() orelse return null;
|
||||
|
||||
var path: [base.len + 1 + 19 + 1 + system.maximum_process_name + 4]u8 = undefined;
|
||||
const relative = if (name.len != 0 and name[0] == '/') name[1..] else name;
|
||||
const full = std.fmt.bufPrint(&path, "{s}/{s}.log", .{ boot_directory[0..boot_directory_len], relative }) catch return null;
|
||||
|
||||
// Parent directories: everything up to the final slash.
|
||||
if (std.mem.lastIndexOfScalar(u8, full, '/')) |last| {
|
||||
if (!fs.makePath(full[0..last])) return null;
|
||||
}
|
||||
var file = fs.open(full, .{ .create = true }) orelse return null;
|
||||
// Append: land after whatever an earlier open of this boot wrote.
|
||||
if (file.attributes()) |attributes| file.seekTo(attributes.size);
|
||||
|
||||
cached.* = .{ .used = true, .file = file };
|
||||
@memcpy(cached.name[0..name.len], name);
|
||||
cached.name_len = name.len;
|
||||
return &cached.file;
|
||||
}
|
||||
|
||||
fn evictOne() ?*CachedFile {
|
||||
// All slots busy: close the first (oldest-created) and reuse it. Simple and
|
||||
// rare — the process count sits well under the cache size.
|
||||
for (&files) |*cached| {
|
||||
if (cached.used) {
|
||||
cached.file.close();
|
||||
cached.used = false;
|
||||
return cached;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn closeAll() void {
|
||||
for (&files) |*cached| {
|
||||
if (cached.used) {
|
||||
cached.file.close();
|
||||
cached.used = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn recordLength(header: system.KlogRecordHeader) usize {
|
||||
return std.mem.alignForward(usize, system.klog_record_header_size + header.name_len + header.message_len, system.klog_record_alignment);
|
||||
}
|
||||
|
||||
/// Format the per-boot directory "<base>/YYYY-MM-DDTHHMMSSZ" from the boot
|
||||
/// wall-clock anchor. No colons — FAT names cannot carry them. A dead RTC
|
||||
/// (anchor 0) yields the 1970 epoch directory, which is still a valid,
|
||||
/// distinct-per-boot-rarely name and better than refusing to log.
|
||||
fn formatBootDirectory(boot_unix_seconds: u64) void {
|
||||
const epoch_seconds = std.time.epoch.EpochSeconds{ .secs = boot_unix_seconds };
|
||||
const year_day = epoch_seconds.getEpochDay().calculateYearDay();
|
||||
const month_day = year_day.calculateMonthDay();
|
||||
const day_seconds = epoch_seconds.getDaySeconds();
|
||||
const written = std.fmt.bufPrint(&boot_directory, "{s}/{d:0>4}-{d:0>2}-{d:0>2}T{d:0>2}{d:0>2}{d:0>2}Z", .{
|
||||
base,
|
||||
year_day.year,
|
||||
month_day.month.numeric(),
|
||||
@as(u32, month_day.day_index) + 1,
|
||||
day_seconds.getHoursIntoDay(),
|
||||
day_seconds.getMinutesIntoHour(),
|
||||
day_seconds.getSecondsIntoMinute(),
|
||||
}) catch return;
|
||||
boot_directory_len = written.len;
|
||||
}
|
||||
@@ -147,8 +147,8 @@ pub fn main(init: runtime.process.Init) void {
|
||||
const spinner = runtime.system.spawnSupervised("process-test", &.{"spinner"}, endpoint) orelse fail("spawn spinner");
|
||||
|
||||
runtime.system.sleep(100); // let the sleeper block and the spinner get a core
|
||||
if (!listed(sleeper, "process-test")) fail("sleeper not in process_enumerate");
|
||||
if (!listed(spinner, "process-test")) fail("spinner not in process_enumerate");
|
||||
if (!listed(sleeper, "/system/tests/process-test")) fail("sleeper not in process_enumerate");
|
||||
if (!listed(spinner, "/system/tests/process-test")) fail("spinner not in process_enumerate");
|
||||
|
||||
// Kills that must be refused: a kernel task (id 0), and an id that was never
|
||||
// issued — both -ESRCH. (-EPERM needs a second supervisor; the kernel-level
|
||||
@@ -167,8 +167,8 @@ pub fn main(init: runtime.process.Init) void {
|
||||
if (!runtime.system.kill(spinner)) fail("kill spinner");
|
||||
if (awaitChildExit(endpoint) != spinner) fail("spinner exit notification");
|
||||
|
||||
if (listed(sleeper, "process-test")) fail("sleeper still listed after kill");
|
||||
if (listed(spinner, "process-test")) fail("spinner still listed after kill");
|
||||
if (listed(sleeper, "/system/tests/process-test")) fail("sleeper still listed after kill");
|
||||
if (listed(spinner, "/system/tests/process-test")) fail("spinner still listed after kill");
|
||||
|
||||
// M17.2: both children were killed by us, and the reason says so — the whole
|
||||
// restart-policy input, read through the runtime like a real supervisor would.
|
||||
|
||||
+12
-12
@@ -1,17 +1,17 @@
|
||||
//! system/services/shm-client — the creating half of the shm test (docs/display-v2.md V2).
|
||||
//! It `shm_create`s a shared region, writes a known pattern into it, and hands the region's
|
||||
//! capability to `shm-server` as an `ipc_call` send_cap. The server maps that capability and
|
||||
//! system/services/shared-memory-client — the creating half of the shared-memory test (docs/display-v2.md V2).
|
||||
//! It `shared_memory_create`s a shared region, writes a known pattern into it, and hands the region's
|
||||
//! capability to `shared-memory-server` as an `ipc_call` send_cap. The server maps that capability and
|
||||
//! confirms the pattern is visible — proving cross-process shared memory over the extended
|
||||
//! capability-passing path.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const system = runtime.system;
|
||||
const shm = runtime.shm;
|
||||
const shared_memory = runtime.shared_memory;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const pattern_len = 4096;
|
||||
|
||||
/// The pattern the server checks — must match shm-server.zig.
|
||||
/// The pattern the server checks — must match shared-memory-server.zig.
|
||||
fn expected(i: usize) u8 {
|
||||
return @truncate(i *% 7 +% 3);
|
||||
}
|
||||
@@ -19,28 +19,28 @@ fn expected(i: usize) u8 {
|
||||
fn lookupServer() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.shm_test)) |h| return h;
|
||||
if (ipc.lookup(.shared_memory_test)) |h| return h;
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const region = shm.create(pattern_len) orelse {
|
||||
_ = system.write("shm: create failed\n");
|
||||
const region = shared_memory.create(pattern_len) orelse {
|
||||
_ = system.write("shared-memory: create failed\n");
|
||||
return;
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < pattern_len) : (i += 1) region.ptr[i] = expected(i);
|
||||
|
||||
const server = lookupServer() orelse {
|
||||
_ = system.write("shm: no server\n");
|
||||
_ = system.write("shared-memory: no server\n");
|
||||
return;
|
||||
};
|
||||
// A non-empty message (so it reaches on_message, not the ping path), carrying the shm
|
||||
// A non-empty message (so it reaches on_message, not the ping path), carrying the shared-memory
|
||||
// region's capability. The reply is empty; we just need the round trip.
|
||||
var reply: [64]u8 = undefined;
|
||||
_ = ipc.callCap(server, "shm", &reply, region.handle) catch {
|
||||
_ = system.write("shm: call failed\n");
|
||||
_ = ipc.callCap(server, "shared-memory", &reply, region.handle) catch {
|
||||
_ = system.write("shared-memory: call failed\n");
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
//! system/services/shared-memory-server — the receiving half of the shared-memory test (docs/display-v2.md V2).
|
||||
//! It registers under `ServiceId.shared_memory_test`; when `shared-memory-client` calls it carrying a
|
||||
//! shared-memory capability, it `shared_memory_map`s that capability and checks the client's pattern
|
||||
//! is visible through the mapping — proving the two processes share the same physical pages
|
||||
//! (not a copy). On success it prints `shared-memory: shared 4096 bytes ok`, the test's marker.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const system = runtime.system;
|
||||
const shared_memory = runtime.shared_memory;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const pattern_len = 4096;
|
||||
|
||||
/// The pattern the client writes — must match shared-memory-client.zig.
|
||||
fn expected(i: usize) u8 {
|
||||
return @truncate(i *% 7 +% 3);
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
const cap = capability orelse {
|
||||
_ = system.write("shared-memory: shared FAILED (no capability)\n");
|
||||
return 0;
|
||||
};
|
||||
const ptr = shared_memory.map(cap) orelse {
|
||||
_ = system.write("shared-memory: shared FAILED (map)\n");
|
||||
return 0;
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < pattern_len) : (i += 1) {
|
||||
if (ptr[i] != expected(i)) {
|
||||
_ = system.write("shared-memory: shared FAILED (mismatch)\n");
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
_ = system.write("shared-memory: shared 4096 bytes ok\n");
|
||||
return 0; // empty reply — the client only needs the round trip to unblock
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(64, .{ .service = .shared_memory_test, .on_message = onMessage });
|
||||
}
|
||||
@@ -1,44 +0,0 @@
|
||||
//! system/services/shm-server — the receiving half of the shm test (docs/display-v2.md V2).
|
||||
//! It registers under `ServiceId.shm_test`; when `shm-client` calls it carrying a
|
||||
//! shared-memory capability, it `shm_map`s that capability and checks the client's pattern
|
||||
//! is visible through the mapping — proving the two processes share the same physical pages
|
||||
//! (not a copy). On success it prints `shm: shared 4096 bytes ok`, the test's marker.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const system = runtime.system;
|
||||
const shm = runtime.shm;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const pattern_len = 4096;
|
||||
|
||||
/// The pattern the client writes — must match shm-client.zig.
|
||||
fn expected(i: usize) u8 {
|
||||
return @truncate(i *% 7 +% 3);
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
const cap = capability orelse {
|
||||
_ = system.write("shm: shared FAILED (no capability)\n");
|
||||
return 0;
|
||||
};
|
||||
const ptr = shm.map(cap) orelse {
|
||||
_ = system.write("shm: shared FAILED (map)\n");
|
||||
return 0;
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < pattern_len) : (i += 1) {
|
||||
if (ptr[i] != expected(i)) {
|
||||
_ = system.write("shm: shared FAILED (mismatch)\n");
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
_ = system.write("shm: shared 4096 bytes ok\n");
|
||||
return 0; // empty reply — the client only needs the round trip to unblock
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(64, .{ .service = .shm_test, .on_message = onMessage });
|
||||
}
|
||||
@@ -40,7 +40,7 @@ fn runSpawnMode() void {
|
||||
runtime.system.yield();
|
||||
}
|
||||
if (spawn_done.load(.acquire) == 1 and shared_value == sentinel) {
|
||||
write("thread-test: child ran in shared aspace ok\n");
|
||||
write("thread-test: child ran in shared address space ok\n");
|
||||
} else {
|
||||
write("thread-test: FAIL worker did not update shared memory\n");
|
||||
}
|
||||
@@ -74,6 +74,8 @@ fn detachWorker() void {
|
||||
detach_done.store(1, .release);
|
||||
}
|
||||
|
||||
fn noopWorker() void {}
|
||||
|
||||
fn runJoinMode() void {
|
||||
write("thread-test: join mode starting\n");
|
||||
|
||||
@@ -114,7 +116,19 @@ fn runJoinMode() void {
|
||||
return;
|
||||
}
|
||||
|
||||
write("thread-test: join ok\n"); // the M3 verdict marker
|
||||
// Prove join needs no per-thread kernel endpoint (M9): many spawn+join cycles. Under
|
||||
// the old per-thread-endpoint scheme these leaked handles and would exhaust the
|
||||
// 16-slot handle table well before 40; here they all succeed.
|
||||
var cycle: u32 = 0;
|
||||
while (cycle < 40) : (cycle += 1) {
|
||||
const th = runtime.Thread.spawn(.{}, noopWorker, .{}) catch {
|
||||
write("thread-test: FAIL spawn exhausted across join cycles (endpoint leak?)\n");
|
||||
return;
|
||||
};
|
||||
th.join();
|
||||
}
|
||||
|
||||
write("thread-test: join ok\n"); // the M3/M9 verdict marker
|
||||
}
|
||||
|
||||
// --- M4: futex mode ---------------------------------------------------------
|
||||
@@ -294,6 +308,167 @@ fn runIdMode() void {
|
||||
write("thread-id: ok\n"); // the M6 verdict marker
|
||||
}
|
||||
|
||||
// --- M7: alloc mode (concurrent heap allocation) ----------------------------
|
||||
|
||||
const alloc_threads: u32 = 4;
|
||||
const allocs_per_thread: u32 = 500;
|
||||
var allocs_clean = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn allocWorker(seed: u32) void {
|
||||
const gpa = runtime.allocator();
|
||||
var rng: u32 = seed | 1;
|
||||
var round: u32 = 0;
|
||||
while (round < allocs_per_thread) : (round += 1) {
|
||||
rng = rng *% 1664525 +% 1013904223; // cheap LCG for varied sizes
|
||||
const size: usize = 16 + (rng % 4080); // 16..4095 bytes
|
||||
const buf = gpa.alloc(u8, size) catch return; // OOM: don't count this thread clean
|
||||
const pattern: u8 = @truncate(seed +% round);
|
||||
@memset(buf, pattern);
|
||||
// Nothing else should touch our block; if a concurrent allocation overlapped it,
|
||||
// one of us would read the other's pattern here.
|
||||
var ok = true;
|
||||
for (buf) |b| {
|
||||
if (b != pattern) ok = false;
|
||||
}
|
||||
gpa.free(buf);
|
||||
if (!ok) return; // corruption — leave without counting clean
|
||||
}
|
||||
_ = allocs_clean.fetchAdd(1, .monotonic);
|
||||
}
|
||||
|
||||
fn runAllocMode() void {
|
||||
write("thread-alloc: starting\n");
|
||||
var threads: [alloc_threads]runtime.Thread = undefined;
|
||||
var n: u32 = 0;
|
||||
while (n < alloc_threads) : (n += 1) {
|
||||
threads[n] = runtime.Thread.spawn(.{}, allocWorker, .{n +% 1}) catch {
|
||||
write("thread-alloc: FAIL spawn\n");
|
||||
return;
|
||||
};
|
||||
}
|
||||
for (threads[0..alloc_threads]) |t| t.join();
|
||||
|
||||
// Every thread must have completed all rounds with each block intact — proof the
|
||||
// shared heap and the per-address_space mmap arena are safe under concurrent allocation.
|
||||
if (allocs_clean.load(.acquire) != alloc_threads) {
|
||||
write("thread-alloc: FAIL corruption or OOM under concurrent allocation\n");
|
||||
return;
|
||||
}
|
||||
write("thread-alloc: ok\n"); // the M7 verdict marker
|
||||
}
|
||||
|
||||
// --- M10: tls mode (per-thread FS base storage) -----------------------------
|
||||
|
||||
fn writeTlsSlot(value: u64) void {
|
||||
asm volatile ("movq %[v], %%fs:8"
|
||||
:
|
||||
: [v] "r" (value),
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
fn readTlsSlot() u64 {
|
||||
return asm volatile ("movq %%fs:8, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
:
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
var tls_written = std.atomic.Value(u32).init(0);
|
||||
var tls_ok = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn tlsWorker(marker: u64) void {
|
||||
writeTlsSlot(marker);
|
||||
_ = tls_written.fetchAdd(1, .release);
|
||||
// Wait until both threads have written their own slot. If the FS base were shared, the
|
||||
// second write would clobber the first, and the read below would return the wrong
|
||||
// marker — cross-talk. A per-thread FS base keeps each thread's slot private.
|
||||
var spins: usize = 0;
|
||||
while (tls_written.load(.acquire) < 2 and spins < 50_000_000) : (spins += 1) {
|
||||
runtime.system.yield();
|
||||
}
|
||||
if (readTlsSlot() == marker and runtime.Thread.getCurrentId() != 0) {
|
||||
_ = tls_ok.fetchAdd(1, .monotonic);
|
||||
}
|
||||
}
|
||||
|
||||
fn runTlsMode() void {
|
||||
write("thread-tls: starting\n");
|
||||
const t0 = runtime.Thread.spawn(.{}, tlsWorker, .{@as(u64, 0xAAAA_0000)}) catch {
|
||||
write("thread-tls: FAIL spawn\n");
|
||||
return;
|
||||
};
|
||||
const t1 = runtime.Thread.spawn(.{}, tlsWorker, .{@as(u64, 0xBBBB_0000)}) catch {
|
||||
write("thread-tls: FAIL spawn\n");
|
||||
return;
|
||||
};
|
||||
t0.join();
|
||||
t1.join();
|
||||
if (tls_ok.load(.acquire) == 2) {
|
||||
write("thread-tls: ok\n"); // the M10 verdict marker
|
||||
} else {
|
||||
write("thread-tls: FAIL cross-talk (FS base not per-thread)\n");
|
||||
}
|
||||
}
|
||||
|
||||
// --- M11: rwlock mode (readers/writers over an RwLock) ----------------------
|
||||
|
||||
const RwLock = runtime.Thread.RwLock;
|
||||
|
||||
var rwlock = RwLock{};
|
||||
var rw_a: u64 = 0;
|
||||
var rw_b: u64 = 0; // invariant while any lock is held: rw_a == rw_b
|
||||
var rw_stop = std.atomic.Value(u32).init(0);
|
||||
var rw_violations = std.atomic.Value(u32).init(0);
|
||||
var rw_reads = std.atomic.Value(u64).init(0);
|
||||
|
||||
fn rwWriter() void {
|
||||
var v: u64 = 1;
|
||||
while (rw_stop.load(.acquire) == 0) : (v +%= 1) {
|
||||
rwlock.lock(); // exclusive: no reader may observe the gap between the two writes
|
||||
rw_a = v;
|
||||
rw_b = v;
|
||||
rwlock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
fn rwReader() void {
|
||||
const reads: u64 = 50_000;
|
||||
var i: u64 = 0;
|
||||
while (i < reads) : (i += 1) {
|
||||
rwlock.lockShared();
|
||||
if (rw_a != rw_b) _ = rw_violations.fetchAdd(1, .monotonic); // saw a half-write!
|
||||
rwlock.unlockShared();
|
||||
}
|
||||
_ = rw_reads.fetchAdd(reads, .monotonic);
|
||||
}
|
||||
|
||||
fn runRwlockMode() void {
|
||||
write("thread-rwlock: starting\n");
|
||||
var writers: [2]runtime.Thread = undefined;
|
||||
var readers: [3]runtime.Thread = undefined;
|
||||
for (&writers) |*w| {
|
||||
w.* = runtime.Thread.spawn(.{}, rwWriter, .{}) catch {
|
||||
write("thread-rwlock: FAIL spawn\n");
|
||||
return;
|
||||
};
|
||||
}
|
||||
for (&readers) |*r| {
|
||||
r.* = runtime.Thread.spawn(.{}, rwReader, .{}) catch {
|
||||
write("thread-rwlock: FAIL spawn\n");
|
||||
return;
|
||||
};
|
||||
}
|
||||
for (readers) |r| r.join();
|
||||
rw_stop.store(1, .release); // readers done → stop the writers
|
||||
for (writers) |w| w.join();
|
||||
|
||||
if (rw_violations.load(.acquire) == 0 and rw_reads.load(.acquire) > 0) {
|
||||
write("thread-rwlock: ok\n"); // the M11 verdict marker
|
||||
} else {
|
||||
write("thread-rwlock: FAIL reader observed a half-written value\n");
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const mode = init.arguments.get(1) orelse "spawn";
|
||||
if (std.mem.eql(u8, mode, "join")) {
|
||||
@@ -304,6 +479,12 @@ pub fn main(init: runtime.process.Init) void {
|
||||
runMutexMode();
|
||||
} else if (std.mem.eql(u8, mode, "id")) {
|
||||
runIdMode();
|
||||
} else if (std.mem.eql(u8, mode, "alloc")) {
|
||||
runAllocMode();
|
||||
} else if (std.mem.eql(u8, mode, "tls")) {
|
||||
runTlsMode();
|
||||
} else if (std.mem.eql(u8, mode, "rwlock")) {
|
||||
runRwlockMode();
|
||||
} else {
|
||||
runSpawnMode();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
//! /system/tests/vfs-test — a ring-3 client that proves the kernel VFS end to
|
||||
//! end through the plain `runtime.fs` API: resolve its OWN binary under the
|
||||
//! kernel-served /system mount, check its metadata, read its ELF magic, and
|
||||
//! list /system/services. On success it heartbeats "vfstest: ok" so the kernel
|
||||
//! test can observe it; on failure it reports what went wrong.
|
||||
//!
|
||||
//! The "park" role (the fat-client-death test): open a file on the FAT volume,
|
||||
//! then hold the handle forever without closing — the kill and the fat
|
||||
//! server's release-on-death sweep are the point.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const fs = runtime.fs;
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
if (init.arguments.count > 1) {
|
||||
park();
|
||||
return;
|
||||
}
|
||||
|
||||
// Our own binary, resolved through the kernel mount table.
|
||||
const self_path = "/system/tests/vfs-test";
|
||||
var file = fs.open(self_path, .{}) orelse {
|
||||
_ = runtime.system.write("vfstest: open of own binary failed\n");
|
||||
return;
|
||||
};
|
||||
defer file.close();
|
||||
|
||||
const attributes = file.attributes() orelse {
|
||||
_ = runtime.system.write("vfstest: attributes failed\n");
|
||||
return;
|
||||
};
|
||||
if (attributes.kind != .regular or attributes.size == 0) {
|
||||
_ = runtime.system.write("vfstest: bad attributes\n");
|
||||
return;
|
||||
}
|
||||
|
||||
var header: [4]u8 = undefined;
|
||||
const n = file.read(&header) orelse 0;
|
||||
if (n != 4 or header[0] != 0x7f or header[1] != 'E' or header[2] != 'L' or header[3] != 'F') {
|
||||
_ = runtime.system.write("vfstest: ELF magic mismatch\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// The write refusal: /system is read-only by construction.
|
||||
if (file.write("x") != null or fs.open("/system/tests/new-file", .{ .create = true }) != null) {
|
||||
_ = runtime.system.write("vfstest: /system accepted a write\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Listing: /system/services contains init.
|
||||
var saw_init = false;
|
||||
if (fs.openDirectory("/system/services")) |listing| {
|
||||
var directory = listing;
|
||||
defer directory.close();
|
||||
var entry: fs.Entry = .{};
|
||||
while (directory.next(&entry)) {
|
||||
if (std.mem.eql(u8, entry.name(), "init")) saw_init = true;
|
||||
}
|
||||
}
|
||||
if (!saw_init) {
|
||||
_ = runtime.system.write("vfstest: /system/services listing missed init\n");
|
||||
return;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
_ = runtime.system.write("vfstest: ok\n");
|
||||
runtime.system.sleep(1000);
|
||||
}
|
||||
}
|
||||
|
||||
fn park() void {
|
||||
// The storage chain (usb -> block -> fat -> mounts) takes a few seconds;
|
||||
// retry until the volume appears.
|
||||
var parked: ?fs.File = null;
|
||||
var tries: u32 = 0;
|
||||
while (parked == null and tries < 1000) : (tries += 1) {
|
||||
parked = fs.open("/mnt/usb/parked", .{ .create = true });
|
||||
if (parked == null) runtime.system.sleep(20);
|
||||
}
|
||||
if (parked == null) {
|
||||
_ = runtime.system.write("vfstest: park open failed\n");
|
||||
return;
|
||||
}
|
||||
while (true) {
|
||||
_ = runtime.system.write("vfstest: parked\n");
|
||||
runtime.system.sleep(500);
|
||||
}
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
//! Pure path utilities for the VFS mount router — no IPC, no state, so they are
|
||||
//! host-testable in isolation. The router uses these to decide whether an opened
|
||||
//! path lies under a mount point and, if so, what it looks like relative to that
|
||||
//! mount.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by a
|
||||
/// path separator — return the path relative to the mount ("/" for an exact
|
||||
/// match, otherwise the tail beginning with '/'). Returns null when `path` is not
|
||||
/// under the mount, so a prefix like "/mnt/usb" never captures "/mnt/usbextra".
|
||||
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
||||
if (path.len < mount_prefix.len) return null;
|
||||
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
||||
if (path.len == mount_prefix.len) return "/";
|
||||
if (path[mount_prefix.len] != '/') return null;
|
||||
return path[mount_prefix.len..];
|
||||
}
|
||||
|
||||
/// Whether `path` is absolute (rooted at '/'). Bare names — what the flat ramfs
|
||||
/// uses — are relative and never route through a mount.
|
||||
pub fn isAbsolute(path: []const u8) bool {
|
||||
return path.len > 0 and path[0] == '/';
|
||||
}
|
||||
|
||||
test "underMount matches only at path boundaries" {
|
||||
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
|
||||
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
|
||||
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null); // not a boundary
|
||||
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null); // shorter than the prefix
|
||||
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
|
||||
try std.testing.expect(underMount("greeting", "/mnt/usb") == null); // a bare name
|
||||
}
|
||||
|
||||
test "isAbsolute distinguishes paths from bare names" {
|
||||
try std.testing.expect(isAbsolute("/mnt/usb"));
|
||||
try std.testing.expect(!isAbsolute("greeting"));
|
||||
try std.testing.expect(!isAbsolute(""));
|
||||
}
|
||||
@@ -1,62 +0,0 @@
|
||||
//! /system/services/vfs/vfs-test — a client that proves the VFS round trip end to end: open a
|
||||
//! file through the `runtime.fs` file API, write to it, seek back, read it, and compare.
|
||||
//! On success it heartbeats "vfstest: ok" so the kernel test can observe it;
|
||||
//! on failure it reports what went wrong. Shipped in the initial_ramdisk alongside vfs.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const fs = runtime.fs;
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const payload = "hello-vfs";
|
||||
|
||||
// The "park" role (the vfs-client-death test): open a file, then hold the
|
||||
// handle forever without closing — the kill and the VFS's release-on-death
|
||||
// are the point.
|
||||
if (init.arguments.count > 1) {
|
||||
var parked: ?fs.File = null;
|
||||
var tries: u32 = 0;
|
||||
while (parked == null and tries < 200) : (tries += 1) {
|
||||
parked = fs.open("parked", .{ .create = true });
|
||||
if (parked == null) runtime.system.sleep(20);
|
||||
}
|
||||
if (parked == null) {
|
||||
_ = runtime.system.write("vfstest: park open failed\n");
|
||||
return;
|
||||
}
|
||||
while (true) {
|
||||
_ = runtime.system.write("vfstest: parked\n");
|
||||
runtime.system.sleep(500);
|
||||
}
|
||||
}
|
||||
|
||||
// The VFS server may not have registered yet — retry open until it's up.
|
||||
var opened: ?fs.File = null;
|
||||
var tries: u32 = 0;
|
||||
while (opened == null and tries < 200) : (tries += 1) {
|
||||
opened = fs.open("greeting", .{ .create = true });
|
||||
if (opened == null) runtime.system.sleep(20);
|
||||
}
|
||||
var greeting = opened orelse {
|
||||
_ = runtime.system.write("vfstest: open failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if ((greeting.write(payload) orelse 0) != payload.len) {
|
||||
_ = runtime.system.write("vfstest: write failed\n");
|
||||
return;
|
||||
}
|
||||
greeting.seekTo(0);
|
||||
|
||||
var buffer: [32]u8 = undefined;
|
||||
const n = greeting.read(&buffer) orelse 0;
|
||||
greeting.close();
|
||||
|
||||
if (n == payload.len and std.mem.eql(u8, buffer[0..n], payload)) {
|
||||
while (true) {
|
||||
_ = runtime.system.write("vfstest: ok\n");
|
||||
runtime.system.sleep(1000);
|
||||
}
|
||||
}
|
||||
_ = runtime.system.write("vfstest: mismatch\n");
|
||||
}
|
||||
@@ -1,396 +0,0 @@
|
||||
//! system/services/vfs — the user-space VFS server. Shipped in the initial_ramdisk, spawned as a
|
||||
//! ring-3 process, and reached by every other process through IPC (the `runtime`
|
||||
//! file API marshals open/read/write/stat/close into calls to this server's
|
||||
//! endpoint, published under the well-known `vfs` service id).
|
||||
//!
|
||||
//! Two namespaces meet here (M5):
|
||||
//! - a small in-memory **ramfs** — opening a bare name creates it — enough to
|
||||
//! prove the round trip and to back the existing tests;
|
||||
//! - **mounted filesystems**: a mount table maps an absolute path prefix (e.g.
|
||||
//! `/mnt/usb`) to a backend server's endpoint. An open of a path under a mount
|
||||
//! is *forwarded* to that backend (which speaks this same protocol), and every
|
||||
//! later read/write/status/readdir/close on the resulting handle is relayed to
|
||||
//! it. The VFS is the router; a filesystem (FAT) is the backend.
|
||||
//!
|
||||
//! A path routes through a mount only when it is absolute and lies under a mount
|
||||
//! prefix; bare names always resolve in the flat ramfs — the backward-compat
|
||||
//! contract the `vfs` / `vfs-client-death` tests rely on.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.vfs_protocol;
|
||||
const path = @import("path.zig");
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const Node = struct {
|
||||
used: bool = false,
|
||||
name: [24]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
data: [512]u8 = undefined,
|
||||
size: usize = 0,
|
||||
};
|
||||
|
||||
const OpenFile = struct {
|
||||
used: bool = false,
|
||||
// For a local handle: an index into `nodes`. For a forwarding handle: the
|
||||
// node id the backend returned. (usize == u64 here, so it holds either.)
|
||||
node: usize = 0,
|
||||
// Non-null for a handle that forwards to a mounted backend.
|
||||
backend: ?ipc.Handle = null,
|
||||
// The client (task id — an IPC badge is one) that opened this handle. What
|
||||
// release-on-death sweeps by: a service must never depend on its clients
|
||||
// cleaning up after themselves (docs/process-lifecycle.md).
|
||||
owner: u32 = 0,
|
||||
};
|
||||
|
||||
// One mounted filesystem: an absolute path prefix and the backend endpoint that
|
||||
// serves everything under it.
|
||||
const Mount = struct {
|
||||
used: bool = false,
|
||||
prefix: [64]u8 = undefined,
|
||||
prefix_len: usize = 0,
|
||||
backend: ipc.Handle = 0,
|
||||
};
|
||||
|
||||
var nodes = [_]Node{.{}} ** 8;
|
||||
var opens = [_]OpenFile{.{}} ** 16;
|
||||
var mounts = [_]Mount{.{}} ** 8;
|
||||
|
||||
fn findNode(name: []const u8) ?usize {
|
||||
for (&nodes, 0..) |*n, i| {
|
||||
if (n.used and std.mem.eql(u8, n.name[0..n.name_len], name)) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn createNode(name: []const u8) ?usize {
|
||||
for (&nodes, 0..) |*n, i| {
|
||||
if (!n.used) {
|
||||
const l = @min(name.len, n.name.len);
|
||||
@memcpy(n.name[0..l], name[0..l]);
|
||||
n.* = .{ .used = true, .name = n.name, .name_len = l, .size = 0 };
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn openAt(id: u64) ?*OpenFile {
|
||||
if (id >= opens.len) return null;
|
||||
const o = &opens[@intCast(id)];
|
||||
return if (o.used) o else null;
|
||||
}
|
||||
|
||||
/// The mount whose prefix most specifically contains `name`, and the path
|
||||
/// relative to it. Only absolute paths route; bare names never match.
|
||||
const MountMatch = struct { backend: ipc.Handle, relative: []const u8 };
|
||||
fn longestMount(name: []const u8) ?MountMatch {
|
||||
if (!path.isAbsolute(name)) return null;
|
||||
var best: ?MountMatch = null;
|
||||
var best_len: usize = 0;
|
||||
for (&mounts) |*m| {
|
||||
if (!m.used) continue;
|
||||
const prefix = m.prefix[0..m.prefix_len];
|
||||
if (path.underMount(name, prefix)) |relative| {
|
||||
if (best == null or prefix.len >= best_len) {
|
||||
best_len = prefix.len;
|
||||
best = .{ .backend = m.backend, .relative = relative };
|
||||
}
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
/// Serialise a reply header + payload into `out`; returns the total length.
|
||||
fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize {
|
||||
@memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply));
|
||||
const n = @min(payload.len, out.len - protocol.reply_size);
|
||||
@memcpy(out[protocol.reply_size..][0..n], payload[0..n]);
|
||||
return protocol.reply_size + n;
|
||||
}
|
||||
|
||||
fn fail(out: []u8) usize {
|
||||
return writeReply(out, .{ .status = -1 }, &.{});
|
||||
}
|
||||
|
||||
/// Format one whole log line and emit it in a single `debug_write`, so lines from
|
||||
/// concurrent processes can never land in the middle of it.
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// --- mount routing ----------------------------------------------------------
|
||||
|
||||
/// Forward an open under a mount to its backend and, on success, allocate a local
|
||||
/// forwarding handle that remembers the backend's node id.
|
||||
fn forwardOpen(out: []u8, backend: ipc.Handle, relative: []const u8, flags: u32, sender: u32) usize {
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = flags };
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const rel = relative[0..@min(relative.len, protocol.maximum_payload)];
|
||||
@memcpy(message[protocol.request_size..][0..rel.len], rel);
|
||||
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(backend, message[0 .. protocol.request_size + rel.len], &reply) catch return fail(out);
|
||||
if (n < protocol.reply_size) return fail(out);
|
||||
const backend_reply = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
if (backend_reply.status != 0) return writeReply(out, .{ .status = backend_reply.status }, &.{});
|
||||
|
||||
for (&opens, 0..) |*o, i| {
|
||||
if (!o.used) {
|
||||
o.* = .{ .used = true, .node = @intCast(backend_reply.node), .backend = backend, .owner = sender };
|
||||
return writeReply(out, .{ .status = 0, .node = i }, &.{});
|
||||
}
|
||||
}
|
||||
return fail(out);
|
||||
}
|
||||
|
||||
/// Relay a read/write/status/readdir/close on a forwarding handle to the backend
|
||||
/// (the node already rewritten to the backend's id) and copy its reply out.
|
||||
fn forwardRequest(out: []u8, backend: ipc.Handle, request: protocol.Request, payload: []const u8) usize {
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const plen = @min(payload.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..plen], payload[0..plen]);
|
||||
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(backend, message[0 .. protocol.request_size + plen], &reply) catch return fail(out);
|
||||
const copy = @min(n, out.len);
|
||||
@memcpy(out[0..copy], reply[0..copy]);
|
||||
return copy;
|
||||
}
|
||||
|
||||
/// Forward a path-based operation (mkdir, unlink) under a mount to its backend and
|
||||
/// relay the reply. No handle is created — these operate by path and return only a
|
||||
/// status.
|
||||
fn forwardPath(out: []u8, backend: ipc.Handle, operation: protocol.Operation, relative: []const u8) usize {
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const rel = relative[0..@min(relative.len, protocol.maximum_payload)];
|
||||
@memcpy(message[protocol.request_size..][0..rel.len], rel);
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(backend, message[0 .. protocol.request_size + rel.len], &reply) catch return fail(out);
|
||||
const copy = @min(n, out.len);
|
||||
@memcpy(out[0..copy], reply[0..copy]);
|
||||
return copy;
|
||||
}
|
||||
|
||||
/// Forward a rename to its backend: the payload is the mount-relative old path, a
|
||||
/// 0x00 separator, then the mount-relative new path. Relays the backend's reply.
|
||||
fn forwardRename(out: []u8, backend: ipc.Handle, old_relative: []const u8, new_relative: []const u8) usize {
|
||||
const total = old_relative.len + 1 + new_relative.len;
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
if (protocol.request_size + total > message.len) return fail(out);
|
||||
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
var p = protocol.request_size;
|
||||
@memcpy(message[p..][0..old_relative.len], old_relative);
|
||||
p += old_relative.len;
|
||||
message[p] = 0;
|
||||
p += 1;
|
||||
@memcpy(message[p..][0..new_relative.len], new_relative);
|
||||
p += new_relative.len;
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(backend, message[0..p], &reply) catch return fail(out);
|
||||
const copy = @min(n, out.len);
|
||||
@memcpy(out[0..copy], reply[0..copy]);
|
||||
return copy;
|
||||
}
|
||||
|
||||
/// Best-effort close of a backend node (used when a dead client's forwarding
|
||||
/// handles are swept — the backend must not leak the vfs's opens).
|
||||
fn forwardClose(backend: ipc.Handle, backend_node: u64) void {
|
||||
const request = protocol.Request{ .operation = .close, .node = backend_node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = ipc.call(backend, std.mem.asBytes(&request), &reply) catch {};
|
||||
}
|
||||
|
||||
fn doMount(out: []u8, prefix: []const u8, backend: ipc.Handle) usize {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefix[0..m.prefix_len], prefix)) {
|
||||
m.backend = backend;
|
||||
writeLine("/system/services/vfs: remounted {s}\n", .{prefix});
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
}
|
||||
}
|
||||
for (&mounts) |*m| {
|
||||
if (!m.used) {
|
||||
const l = @min(prefix.len, m.prefix.len);
|
||||
m.used = true;
|
||||
@memcpy(m.prefix[0..l], prefix[0..l]);
|
||||
m.prefix_len = l;
|
||||
m.backend = backend;
|
||||
writeLine("/system/services/vfs: mounted {s}\n", .{prefix[0..l]});
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
}
|
||||
}
|
||||
return fail(out);
|
||||
}
|
||||
|
||||
fn doUnmount(out: []u8, prefix: []const u8) usize {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefix[0..m.prefix_len], prefix)) {
|
||||
m.used = false;
|
||||
writeLine("/system/services/vfs: unmounted {s}\n", .{prefix});
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
}
|
||||
}
|
||||
return fail(out);
|
||||
}
|
||||
|
||||
/// Release every open handle `client` held — called on that client's published
|
||||
/// exit event. Forwarding handles also tell their backend to release; local
|
||||
/// nodes (the ramfs files) stay, since ramfs contents outlive their writers.
|
||||
fn releaseClientHandles(client: u32) void {
|
||||
var released: u32 = 0;
|
||||
for (&opens) |*o| {
|
||||
if (o.used and o.owner == client) {
|
||||
if (o.backend) |backend| forwardClose(backend, o.node);
|
||||
o.used = false;
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) writeLine("/system/services/vfs: released {d} handle(s) for dead client {d}\n", .{ released, client });
|
||||
}
|
||||
|
||||
/// Handle one request from `sender`; write the reply into `out`, return its length.
|
||||
fn handle(message: []const u8, out: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
if (message.len < protocol.request_size) return fail(out);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
|
||||
switch (request.operation) {
|
||||
.mount => {
|
||||
const prefix = payload[0..@min(payload.len, request.len)];
|
||||
const backend = capability orelse return fail(out);
|
||||
return doMount(out, prefix, backend);
|
||||
},
|
||||
.unmount => {
|
||||
const prefix = payload[0..@min(payload.len, request.len)];
|
||||
return doUnmount(out, prefix);
|
||||
},
|
||||
.open => {
|
||||
const name = payload[0..@min(payload.len, request.len)];
|
||||
if (longestMount(name)) |m| return forwardOpen(out, m.backend, m.relative, request.flags, sender);
|
||||
// An absolute path with no matching mount is simply not found — only
|
||||
// bare names live in the flat ramfs. (Else /mnt/usb would be silently
|
||||
// created as a flat file when its filesystem is not yet mounted.)
|
||||
if (path.isAbsolute(name)) return fail(out);
|
||||
const ni = findNode(name) orelse createNode(name) orelse return fail(out);
|
||||
for (&opens, 0..) |*o, i| {
|
||||
if (!o.used) {
|
||||
o.* = .{ .used = true, .node = ni, .backend = null, .owner = sender };
|
||||
return writeReply(out, .{ .status = 0, .node = i }, &.{});
|
||||
}
|
||||
}
|
||||
return fail(out);
|
||||
},
|
||||
.read => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
if (of.backend) |backend| {
|
||||
var forwarded = request;
|
||||
forwarded.node = of.node;
|
||||
return forwardRequest(out, backend, forwarded, payload);
|
||||
}
|
||||
const nd = &nodes[@intCast(of.node)];
|
||||
const off: usize = @intCast(request.offset);
|
||||
if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF
|
||||
const n = @min(@min(nd.size - off, request.len), protocol.maximum_payload);
|
||||
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]);
|
||||
},
|
||||
.write => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
if (of.backend) |backend| {
|
||||
var forwarded = request;
|
||||
forwarded.node = of.node;
|
||||
return forwardRequest(out, backend, forwarded, payload);
|
||||
}
|
||||
const nd = &nodes[@intCast(of.node)];
|
||||
const off: usize = @intCast(request.offset);
|
||||
if (off > nd.data.len) return fail(out);
|
||||
const n = @min(@min(payload.len, request.len), nd.data.len - off);
|
||||
@memcpy(nd.data[off .. off + n], payload[0..n]);
|
||||
if (off + n > nd.size) nd.size = off + n;
|
||||
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{});
|
||||
},
|
||||
.status => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
if (of.backend) |backend| {
|
||||
var forwarded = request;
|
||||
forwarded.node = of.node;
|
||||
return forwardRequest(out, backend, forwarded, payload);
|
||||
}
|
||||
const st = protocol.FileStatus{ .size = nodes[@intCast(of.node)].size, .kind = @intFromEnum(protocol.NodeKind.regular) };
|
||||
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.FileStatus) }, std.mem.asBytes(&st));
|
||||
},
|
||||
.readdir => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
if (of.backend) |backend| {
|
||||
var forwarded = request;
|
||||
forwarded.node = of.node;
|
||||
return forwardRequest(out, backend, forwarded, payload);
|
||||
}
|
||||
// The flat ramfs has no directories: report EOF.
|
||||
return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
|
||||
},
|
||||
.close => {
|
||||
const of = openAt(request.node);
|
||||
if (of) |o| {
|
||||
if (o.backend) |backend| forwardClose(backend, o.node);
|
||||
o.used = false;
|
||||
}
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
.mkdir, .unlink => {
|
||||
const name = payload[0..@min(payload.len, request.len)];
|
||||
if (longestMount(name)) |m| return forwardPath(out, m.backend, request.operation, m.relative);
|
||||
// Only a mounted backend has real directories; the flat ramfs cannot
|
||||
// create or remove them (and a bare-name path is not a mount target).
|
||||
return fail(out);
|
||||
},
|
||||
.rename => {
|
||||
const both = payload[0..@min(payload.len, request.len)];
|
||||
const sep = std.mem.indexOfScalar(u8, both, 0) orelse return fail(out);
|
||||
const old_path = both[0..sep];
|
||||
const new_path = both[sep + 1 ..];
|
||||
const mo = longestMount(old_path) orelse return fail(out);
|
||||
const mn = longestMount(new_path) orelse return fail(out);
|
||||
// Both paths must live under the same mount — cross-filesystem rename is
|
||||
// not supported.
|
||||
if (mo.backend != mn.backend) return fail(out);
|
||||
return forwardRename(out, mo.backend, mo.relative, mn.relative);
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// Startup, under the harness: subscribe to the published exit events — when a
|
||||
/// client dies holding open handles, the exit notification is how the VFS learns
|
||||
/// to release them (docs/process-lifecycle.md).
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
if (!runtime.process.subscribeExits(endpoint)) {
|
||||
_ = runtime.system.write("/system/services/vfs: exit subscription failed\n");
|
||||
}
|
||||
_ = runtime.system.write("/system/services/vfs: ready\n");
|
||||
return true;
|
||||
}
|
||||
|
||||
/// A non-signal notification: the only kind the VFS subscribes to is exit events.
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & ipc.notify_exit_bit != 0) {
|
||||
releaseClientHandles(@intCast(badge & ~(ipc.notify_badge_bit | ipc.notify_exit_bit)));
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// The harness owns the loop: requests dispatch to handle(), exit events to
|
||||
// onNotification(), ping and terminate are answered for free — this service
|
||||
// gained the whole lifecycle contract by deleting its hand-rolled loop.
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.service = .vfs,
|
||||
.init = initialise,
|
||||
.on_message = handle,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
+77
-26
@@ -185,12 +185,22 @@ CASES = [
|
||||
{"name": "display-demo",
|
||||
"expect": r"display-demo: scene up[\s\S]*display-demo: ok",
|
||||
"fail": r"display-demo: (no display|create failed)|display: could not|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# Shared memory (v2 V2): shm-client creates a region, writes a pattern, and passes its
|
||||
# capability to shm-server, which maps it and confirms the same bytes — proving
|
||||
# Threaded compositor tracks a mouse (docs/threading.md, docs/display.md): the display
|
||||
# runs a mouse-listener thread alongside its compositor loop. `input-source mouse`
|
||||
# publishes pure motion -> the input service fans it to the display's listener -> the
|
||||
# listener accumulates it into a cursor position handed to the render loop over a
|
||||
# single-slot channel. `display: cursor tracking mouse ok` latches once the cursor has
|
||||
# tracked a run of that motion end to end.
|
||||
{"name": "display-cursor",
|
||||
"smp": 4,
|
||||
"expect": r"display: online \d+x\d+[\s\S]*display: cursor tracking mouse ok",
|
||||
"fail": r"display: (could not|mouse subscribe failed)|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# Shared memory (v2 V2): shared-memory-client creates a region, writes a pattern, and passes its
|
||||
# capability to shared-memory-server, which maps it and confirms the same bytes — proving
|
||||
# cross-process shared pages over the extended capability passing.
|
||||
{"name": "shm",
|
||||
"expect": r"shm: shared 4096 bytes ok",
|
||||
"fail": r"shm: (shared FAILED|create failed|no server|call failed|map)|CPU EXCEPTION|KERNEL PANIC"},
|
||||
{"name": "shared-memory",
|
||||
"expect": r"shared-memory: shared 4096 bytes ok",
|
||||
"fail": r"shared-memory: (shared FAILED|create failed|no server|call failed|map)|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# virtio-gpu driver (v2 V3): boot with an emulated virtio-gpu. The device-manager stack
|
||||
# discovers the PCI function and spawns the driver, which brings up the control virtqueue,
|
||||
# creates a 2D scanout resource backed by DMA memory, set_scanouts it, paints a test
|
||||
@@ -211,15 +221,16 @@ CASES = [
|
||||
# require all three markers to appear somewhere rather than in a fixed order.
|
||||
"expect": r"(?s)(?=.*display: scanout upgraded to virtio-gpu)(?=.*display: native present verified)(?=.*display-demo: ok)",
|
||||
"fail": r"display: native present FAILED|display: could not|display-demo: (no display|create failed)|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# Mode-set + EDID + vsync (v2 V5): same boot as display-native. After upgrading, the
|
||||
# compositor queries the driver's modes, switches to a different resolution, and confirms the
|
||||
# backend now reports it; the fenced present path makes it a vsync present. (The driver also
|
||||
# logs the EDID preferred mode during bring-up.) Reuses the display-native kernel scenario.
|
||||
# Mode-set + EDID + fenced presents (v2 V5): same boot as display-native. After upgrading,
|
||||
# the compositor queries the driver's modes, switches to a different resolution, and confirms
|
||||
# the backend now reports it; each present is fenced — completion-acknowledged and tear-free,
|
||||
# not vblank-paced (docs/display-v2.md, "Fenced is not vsync"). (The driver also logs the
|
||||
# EDID preferred mode during bring-up.) Reuses the display-native kernel scenario.
|
||||
{"name": "display-modeset",
|
||||
"build_case": "display-native",
|
||||
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||
"mem": "512M",
|
||||
"expect": r"(?s)(?=.*display: mode set to \d+x\d+, verified)(?=.*display: vsync present ok)",
|
||||
"expect": r"(?s)(?=.*display: mode set to \d+x\d+, verified)(?=.*display: fenced present ok)",
|
||||
"fail": r"display: mode set FAILED|display: mode-set self-check: |display: native present FAILED|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# Resilience: driver restart + re-attach (v2 V6). device-manager (in test-scanout-restart
|
||||
# mode) kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
|
||||
@@ -295,8 +306,8 @@ CASES = [
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M1: address-space refcount — spaces destroyed exactly
|
||||
# once per process, no leak/double-free (the foundation shared-aspace threads need).
|
||||
{"name": "aspace-refcount",
|
||||
# once per process, no leak/double-free (the foundation shared-address-space threads need).
|
||||
{"name": "address-space-refcount",
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -339,6 +350,38 @@ CASES = [
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M7: thread-safe allocation — N threads hammer the shared heap
|
||||
# (per-aspace mmap arena + locked free list) with no cross-block corruption.
|
||||
{"name": "thread-alloc",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M8: the task reaper — spawn+kill many processes; total kernel
|
||||
# stack bytes return to baseline (every dead task's stack reclaimed, no leak).
|
||||
{"name": "task-reap",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M10: per-thread fs.base — two threads keep private %fs:8 TLS
|
||||
# slots across context switches (no cross-talk).
|
||||
{"name": "thread-tls",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M11: RwLock — readers/writers across cores; a reader never
|
||||
# observes a half-written value (writers hold it exclusively).
|
||||
{"name": "thread-rwlock",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned
|
||||
# name, argv[1..] = the system_spawn argument blob) and echoes back intact.
|
||||
{"name": "args",
|
||||
@@ -384,6 +427,7 @@ CASES = [
|
||||
# open handle, and the VFS releases it (process-lifecycle.md "Who learns of a death").
|
||||
{"name": "vfs-client-death",
|
||||
"smp": 4,
|
||||
"timeout": 90, # the park client waits out the whole USB->block->fat chain
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M17.4: signals over IPC — ping, reload, terminate (clean exit), the one-shot
|
||||
@@ -404,7 +448,7 @@ CASES = [
|
||||
r"device-manager: child added[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
r"device-manager: child removed[\s\S]*"
|
||||
r"device-manager: restarting usb-xhci-bus[\s\S]*"
|
||||
r"device-manager: restarting \S*usb-xhci-bus[\s\S]*"
|
||||
r"device-manager: child added",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# USB HID end to end: boot the full tree, enumerate the xHCI, and let the
|
||||
@@ -415,7 +459,7 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
# usb-kbd/usb-mouse ride the default boot xHCI bus (see qemu_args).
|
||||
"expect": r"(?=[\s\S]*usb-hid/keyboard: ok)(?=[\s\S]*usb-hid/mouse: ok)",
|
||||
"expect": r"(?=[\s\S]*usb-hid-keyboard: ok)(?=[\s\S]*usb-hid-mouse: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# USB mass storage end to end: the boot usb-storage device (the FAT32 image,
|
||||
# which has a real 0x55AA boot sector) is enough — the manager spawns
|
||||
@@ -483,8 +527,8 @@ CASES = [
|
||||
{"name": "acpi-ps2",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"acpi: reported PNP0303[\s\S]*"
|
||||
r"device-manager: spawned ps2-bus[\s\S]*"
|
||||
"expect": r"discovery: reported PNP0303[\s\S]*"
|
||||
r"device-manager: spawned \S*ps2-bus[\s\S]*"
|
||||
r"ps2-bus: keyboard driver attached",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
|
||||
@@ -510,17 +554,18 @@ CASES = [
|
||||
r"power: entering S5",
|
||||
"fail": r"power: S5 write did not take|DANOS-TEST-RESULT: FAIL"},
|
||||
# M8: the boot log is persisted to the USB FAT volume. Reuses the orderly-
|
||||
# shutdown build (full tree + power button): init spawns log-flush at boot,
|
||||
# which copies the kernel log to /mnt/usb/DANOS.LOG once /mnt/usb is mounted
|
||||
# (first marker); then the power button drives init's own pre-teardown flush
|
||||
# (second marker), proving both triggers write the file while storage is up.
|
||||
{"name": "log-flush",
|
||||
# shutdown build (full tree + power button): the logger service announces its
|
||||
# per-boot directory once storage mounts (first marker), then the power
|
||||
# button drives the orderly stop — the logger, stopped first, final-drains
|
||||
# and reports the flush (second marker) before S5.
|
||||
{"name": "logger",
|
||||
"build_case": "orderly-shutdown",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qmp_after": {"delay": 8, "command": "system_powerdown"},
|
||||
"expect": r"log-flush: wrote \d+ bytes to /mnt/usb/DANOS\.LOG[\s\S]*"
|
||||
r"init: flushed log to /mnt/usb/DANOS\.LOG[\s\S]*"
|
||||
"expect": r"logger: logging to /var/log/\d{4}-\d{2}-\d{2}T\d{6}Z[\s\S]*"
|
||||
r"init: shutting down[\s\S]*"
|
||||
r"logger: flushed through sequence \d+[\s\S]*"
|
||||
r"power: entering S5",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers +
|
||||
@@ -529,8 +574,8 @@ CASES = [
|
||||
{"name": "acpi-report",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"acpi: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
||||
r"acpi: reported PNP0F13 \(device \d+, 1 resources\)",
|
||||
"expect": r"discovery: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
||||
r"discovery: reported PNP0F13 \(device \d+, 1 resources\)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M19.1/M19.3: the ring-3 PCI scan. pci-bus walks the ECAM through its mmio_map
|
||||
# grant and registers every function it finds; the kernel's own walk retired, so
|
||||
@@ -546,7 +591,7 @@ CASES = [
|
||||
"timeout": 60,
|
||||
"expect": r"pci-bus: (\d+) functions found[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
r"device-manager: restarting pci-bus[\s\S]*"
|
||||
r"device-manager: restarting \S*pci-bus[\s\S]*"
|
||||
r"pci-bus: \1 functions found[\s\S]*"
|
||||
r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -580,6 +625,12 @@ CASES = [
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The user-space VFS: a client opens/writes/reads a file through the rt file
|
||||
# API, which IPCs the VFS server process; the round trip must match.
|
||||
# The kernel VFS root (M-F): the mount table serves the initrd at /system —
|
||||
# path resolution, node status/read (an ELF magic), and directory listing,
|
||||
# asserted kernel-side.
|
||||
{"name": "kvfs",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
{"name": "vfs",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
@@ -179,14 +179,17 @@ def short_name_for(name, used):
|
||||
else:
|
||||
base, ext = name, ""
|
||||
upper_base, upper_ext = base.upper(), ext.upper()
|
||||
# A name fits 8.3 if it is short enough and uses valid characters; a lowercase
|
||||
# name is simply stored uppercased (FAT is case-insensitive, so the bootloader
|
||||
# and the danos driver still find it). Only genuinely non-8.3 names (too long,
|
||||
# e.g. initial-ramdisk.img) get a mangled short name plus LFN entries.
|
||||
# A name fits 8.3 if it is short enough and uses valid characters. The raw
|
||||
# 8.3 entry is always uppercase; if that loses the real name's case (e.g.
|
||||
# "init" -> "INIT"), a long-name chain carries the exact name. This matters
|
||||
# because the EFI loader *enumerates* /system to build the ramdisk — it gets
|
||||
# back whatever the directory stores, so the stored name must be exact, not
|
||||
# merely case-insensitively findable.
|
||||
fits = (1 <= len(base) <= 8 and len(ext) <= 3
|
||||
and all(c in VALID_83 for c in upper_base + upper_ext))
|
||||
if fits:
|
||||
return (upper_base.ljust(8) + upper_ext.ljust(3)).encode("ascii"), False
|
||||
exact = base == upper_base and ext == upper_ext
|
||||
return (upper_base.ljust(8) + upper_ext.ljust(3)).encode("ascii"), not exact
|
||||
# Mangle to STEM~N.EXT.
|
||||
stem = "".join(c for c in upper_base if c in VALID_83 and c != " ")[:6] or "FILE"
|
||||
index = 1
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build-time initial_ramdisk packer. Concatenates user binaries into one image the
|
||||
bootloader ferries to the kernel.
|
||||
|
||||
Usage: make-initial-ramdisk.py <out.img> [<name> <file>]...
|
||||
|
||||
Image layout (little-endian), mirroring src/user/proto/initial-ramdisk.zig:
|
||||
Header : magic u32 ("DNRD"=0x444E5244), count u32
|
||||
Entry*N : name [32]u8 (NUL-padded), offset u64, len u64
|
||||
blobs : each entry's file bytes at its offset
|
||||
"""
|
||||
import struct
|
||||
import sys
|
||||
|
||||
MAGIC = 0x444E5244
|
||||
HEADER = struct.Struct("<II") # magic, count
|
||||
ENTRY = struct.Struct("<32sQQ") # name[32], offset, len
|
||||
|
||||
|
||||
def main() -> int:
|
||||
out_path = sys.argv[1]
|
||||
rest = sys.argv[2:]
|
||||
if len(rest) % 2 != 0:
|
||||
sys.stderr.write("usage: make-initial-ramdisk.py <out.img> [<name> <file>]...\n")
|
||||
return 2
|
||||
items = [(rest[i], rest[i + 1]) for i in range(0, len(rest), 2)]
|
||||
|
||||
table_end = HEADER.size + len(items) * ENTRY.size
|
||||
entries = b""
|
||||
blobs = []
|
||||
off = table_end
|
||||
for name, path in items:
|
||||
with open(path, "rb") as f:
|
||||
data = f.read()
|
||||
entries += ENTRY.pack(name.encode()[:31], off, len(data))
|
||||
blobs.append(data)
|
||||
off += len(data)
|
||||
|
||||
with open(out_path, "wb") as f:
|
||||
f.write(HEADER.pack(MAGIC, len(items)))
|
||||
f.write(entries)
|
||||
for b in blobs:
|
||||
f.write(b)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user