reorg: split the runtime into library/kernel concern modules (C1)

The one giant `runtime` module (with a `system.zig` that was itself a dumping
ground of unrelated syscalls) is split into directly-importable, flat concern
modules under library/kernel/:

  system-call  ipc  memory  process  thread  time  logging  file-system  service  start
  (+ the device/service clients: device, device-manager, block, display, input)

system.zig is dissolved — its functions moved to their concern home (mmap ->
memory, spawn/kill/exit -> process, sleep/clock -> time, write/klog -> logging,
fs* -> file-system). `memory` merges heap+dma+shared-memory behind one flat API
(memory.allocator/dmaAlloc/sharedCreate/mmap), keeping heap's state and malloc
export single. The memory<->thread dependency cycle (heap needs Thread.Mutex,
thread needs mmap) is broken by having thread allocate its own stack via the raw
mmap syscall, so the module graph is a DAG.

This is the atomic step: all 42 internal cross-imports flip from relative to
module imports at once. `runtime.zig` and `system.zig` become thin re-export
SHIMS so the ~38 consumers keep compiling on `runtime.*` untouched; they migrate
to direct imports in C2, after which the shims are deleted (C5).

zig build + zig build test green; 14 QEMU cases pass (smoke, process,
process-kill, thread-spawn/join, logger, vfs, fat-mount, display-native,
usb-storage, virtio-gpu, device-manager, input, power-button).
This commit is contained in:
Daniel Samson
2026-07-22 22:53:13 +01:00
parent 3e69712b97
commit 60f32ee9ff
21 changed files with 636 additions and 467 deletions
+140 -19
View File
@@ -387,31 +387,15 @@ pub fn build(b: *std.Build) void {
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS // types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
// server. It never touches `boot-handoff` — user space has no business with the // server. It never touches `boot-handoff` — user space has no business with the
// loader↔kernel handoff. // loader↔kernel handoff.
const runtime_module = b.addModule("runtime", .{ // Wire-protocol modules the kernel-library client wrappers (device-manager/block/display)
.root_source_file = b.path("library/kernel/runtime.zig"), // and the services speak. The `runtime` module itself is defined below, after the
.imports = &.{ // library/kernel concern modules it shims over.
.{ .name = "abi", .module = abi_module },
.{ .name = "device-abi", .module = device_abi_module },
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
.{ .name = "input-protocol", .module = input_protocol_module },
},
});
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
// own module like the other protocol modules. Imported through the runtime.
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{ const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
.root_source_file = b.path("library/protocol/device-manager/device-manager-protocol.zig"), .root_source_file = b.path("library/protocol/device-manager/device-manager-protocol.zig"),
}); });
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
// The block protocol, so runtime.block (the block-device client) can speak it.
runtime_module.addImport("block-protocol", block_protocol_module);
// The display protocol, so runtime.display (the compositor client) and the display
// service both speak it through the runtime, like the other protocol modules.
const display_protocol_module = b.addModule("display-protocol", .{ const display_protocol_module = b.addModule("display-protocol", .{
.root_source_file = b.path("library/protocol/display/display-protocol.zig"), .root_source_file = b.path("library/protocol/display/display-protocol.zig"),
}); });
runtime_module.addImport("display-protocol", display_protocol_module); // runtime.display client speaks it
// The scanout protocol: the compositor's outbound present channel to a native scanout // The scanout protocol: the compositor's outbound present channel to a native scanout
// driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md). // driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md).
@@ -433,6 +417,143 @@ pub fn build(b: *std.Build) void {
.root_source_file = b.path("library/device/mmio/mmio.zig"), .root_source_file = b.path("library/device/mmio/mmio.zig"),
}); });
// --- library/kernel: the userspace private-ABI library (kernel32-style), split by
// concern into directly-importable modules. `system.zig` (the old dumping ground) and
// `runtime.zig` (the old aggregator) are compatibility shims re-exporting these until the
// consumers migrate to direct imports (reorg C1–C5). The graph is a DAG: memory depends on
// thread (heap needs Thread.Mutex), and thread does its own raw mmap so there is no cycle.
const system_call_module = b.addModule("system-call", .{
.root_source_file = b.path("library/kernel/system-call.zig"),
.imports = &.{.{ .name = "abi", .module = abi_module }},
});
const ipc_module = b.addModule("ipc", .{
.root_source_file = b.path("library/kernel/ipc.zig"),
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
});
const time_module = b.addModule("time", .{
.root_source_file = b.path("library/kernel/time.zig"),
.imports = &.{.{ .name = "system-call", .module = system_call_module }},
});
const thread_module = b.addModule("thread", .{
.root_source_file = b.path("library/kernel/thread.zig"),
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
});
const logging_module = b.addModule("logging", .{
.root_source_file = b.path("library/kernel/logging.zig"),
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
});
const process_module = b.addModule("process", .{
.root_source_file = b.path("library/kernel/process.zig"),
.imports = &.{
.{ .name = "abi", .module = abi_module },
.{ .name = "system-call", .module = system_call_module },
.{ .name = "ipc", .module = ipc_module },
.{ .name = "time", .module = time_module },
},
});
const file_system_module = b.addModule("file-system", .{
.root_source_file = b.path("library/kernel/file-system.zig"),
.imports = &.{
.{ .name = "abi", .module = abi_module },
.{ .name = "system-call", .module = system_call_module },
.{ .name = "ipc", .module = ipc_module },
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
},
});
const memory_module = b.addModule("memory", .{
.root_source_file = b.path("library/kernel/memory/memory.zig"),
.imports = &.{
.{ .name = "abi", .module = abi_module },
.{ .name = "system-call", .module = system_call_module },
.{ .name = "ipc", .module = ipc_module },
.{ .name = "thread", .module = thread_module },
},
});
const service_module = b.addModule("service", .{
.root_source_file = b.path("library/kernel/service.zig"),
.imports = &.{
.{ .name = "abi", .module = abi_module },
.{ .name = "ipc", .module = ipc_module },
.{ .name = "process", .module = process_module },
},
});
const start_module = b.addModule("start", .{
.root_source_file = b.path("library/kernel/start.zig"),
.imports = &.{ .{ .name = "process", .module = process_module }, .{ .name = "logging", .module = logging_module } },
});
const device_module = b.addModule("device", .{
.root_source_file = b.path("library/kernel/device.zig"),
.imports = &.{
.{ .name = "abi", .module = abi_module },
.{ .name = "device-abi", .module = device_abi_module },
.{ .name = "system-call", .module = system_call_module },
},
});
const device_manager_client_module = b.addModule("device-manager", .{
.root_source_file = b.path("library/kernel/device-manager.zig"),
.imports = &.{
.{ .name = "ipc", .module = ipc_module },
.{ .name = "time", .module = time_module },
.{ .name = "device-manager-protocol", .module = device_manager_protocol_module },
},
});
const block_client_module = b.addModule("block", .{
.root_source_file = b.path("library/kernel/block.zig"),
.imports = &.{
.{ .name = "ipc", .module = ipc_module },
.{ .name = "time", .module = time_module },
.{ .name = "block-protocol", .module = block_protocol_module },
},
});
const display_client_module = b.addModule("display", .{
.root_source_file = b.path("library/kernel/display.zig"),
.imports = &.{
.{ .name = "ipc", .module = ipc_module },
.{ .name = "time", .module = time_module },
.{ .name = "display-protocol", .module = display_protocol_module },
},
});
const input_client_module = b.addModule("input", .{
.root_source_file = b.path("library/kernel/input.zig"),
.imports = &.{
.{ .name = "ipc", .module = ipc_module },
.{ .name = "time", .module = time_module },
.{ .name = "input-protocol", .module = input_protocol_module },
},
});
// system.zig — compatibility shim for runtime.system.*, re-exporting the concern modules.
const system_module = b.addModule("system", .{
.root_source_file = b.path("library/kernel/system.zig"),
.imports = &.{
.{ .name = "memory", .module = memory_module },
.{ .name = "process", .module = process_module },
.{ .name = "time", .module = time_module },
.{ .name = "logging", .module = logging_module },
.{ .name = "file-system", .module = file_system_module },
},
});
// runtime.zig — compatibility shim re-exporting every concern module under the old runtime.* names.
const runtime_module = b.addModule("runtime", .{
.root_source_file = b.path("library/kernel/runtime.zig"),
.imports = &.{
.{ .name = "system", .module = system_module },
.{ .name = "ipc", .module = ipc_module },
.{ .name = "memory", .module = memory_module },
.{ .name = "process", .module = process_module },
.{ .name = "service", .module = service_module },
.{ .name = "thread", .module = thread_module },
.{ .name = "time", .module = time_module },
.{ .name = "logging", .module = logging_module },
.{ .name = "file-system", .module = file_system_module },
.{ .name = "start", .module = start_module },
.{ .name = "device", .module = device_module },
.{ .name = "device-manager", .module = device_manager_client_module },
.{ .name = "input", .module = input_client_module },
.{ .name = "block", .module = block_client_module },
.{ .name = "display", .module = display_client_module },
},
});
// A device driver's view of its claimed PCI function: config-space header fields, BAR // A device driver's view of its claimed PCI function: config-space header fields, BAR
// decode + map, and the capability walk (library/device/pci/pci.zig). The generic PCI // decode + map, and the capability walk (library/device/pci/pci.zig). The generic PCI
// mechanics every leaf PCI driver used to re-derive inline. Imports runtime (device // mechanics every leaf PCI driver used to re-derive inline. Imports runtime (device
+3 -3
View File
@@ -8,8 +8,8 @@
//! limit — the same handoff usb-storage uses toward the controller. //! limit — the same handoff usb-storage uses toward the controller.
const std = @import("std"); const std = @import("std");
const ipc = @import("ipc.zig"); const ipc = @import("ipc");
const system = @import("system.zig"); const time = @import("time");
const block_protocol = @import("block-protocol"); const block_protocol = @import("block-protocol");
pub const Geometry = struct { block_size: u32, block_count: u64 }; pub const Geometry = struct { block_size: u32, block_count: u64 };
@@ -73,7 +73,7 @@ pub fn open() ?Device {
// not sit a further minute pretending otherwise. // not sit a further minute pretending otherwise.
while (attempts < 600) : (attempts += 1) { while (attempts < 600) : (attempts += 1) {
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle }; if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
system.sleep(50); time.sleepMillis(50);
} }
return null; return null;
} }
+3 -3
View File
@@ -10,8 +10,8 @@
//! be — copied byte-for-byte into each bus and class driver — lives here once. //! be — copied byte-for-byte into each bus and class driver — lives here once.
const std = @import("std"); const std = @import("std");
const ipc = @import("ipc.zig"); const ipc = @import("ipc");
const system = @import("system.zig"); const time = @import("time");
const device_manager_protocol = @import("device-manager-protocol"); const device_manager_protocol = @import("device-manager-protocol");
/// What kind of driver is announcing itself (a bus that reports children, or a /// What kind of driver is announcing itself (a bus that reports children, or a
@@ -36,7 +36,7 @@ pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
var attempts: u32 = 0; var attempts: u32 = 0;
const manager = while (attempts < lookup_attempts) : (attempts += 1) { const manager = while (attempts < lookup_attempts) : (attempts += 1) {
if (ipc.lookup(.device_manager)) |handle| break handle; if (ipc.lookup(.device_manager)) |handle| break handle;
system.sleep(lookup_pause_ms); time.sleepMillis(lookup_pause_ms);
} else { } else {
std.log.info("no device manager to hello", .{}); std.log.info("no device manager to hello", .{});
return null; return null;
+1 -1
View File
@@ -6,7 +6,7 @@
const std = @import("std"); const std = @import("std");
const abi = @import("abi"); const abi = @import("abi");
const device_abi = @import("device-abi"); const device_abi = @import("device-abi");
const sc = @import("system-call.zig"); const sc = @import("system-call");
pub const DeviceDescriptor = device_abi.DeviceDescriptor; pub const DeviceDescriptor = device_abi.DeviceDescriptor;
pub const ResourceDescriptor = device_abi.ResourceDescriptor; pub const ResourceDescriptor = device_abi.ResourceDescriptor;
+3 -3
View File
@@ -4,8 +4,8 @@
//! reply marshalling. See system/services/display/ and docs/display.md. //! reply marshalling. See system/services/display/ and docs/display.md.
const std = @import("std"); const std = @import("std");
const ipc = @import("ipc.zig"); const ipc = @import("ipc");
const system = @import("system.zig"); const time = @import("time");
const display_protocol = @import("display-protocol"); const display_protocol = @import("display-protocol");
/// The display's current mode, as `info()` reports it. /// The display's current mode, as `info()` reports it.
@@ -29,7 +29,7 @@ fn service() ?ipc.Handle {
handle = h; handle = h;
return h; return h;
} }
system.sleep(50); time.sleepMillis(50);
} }
return null; return null;
} }
@@ -11,8 +11,9 @@
//! shape, unlike the POSIX fd model the old shim emulated. //! shape, unlike the POSIX fd model the old shim emulated.
const std = @import("std"); const std = @import("std");
const ipc = @import("ipc.zig"); const abi = @import("abi");
const system = @import("system.zig"); const sc = @import("system-call");
const ipc = @import("ipc");
const vfs_protocol = @import("vfs-protocol"); const vfs_protocol = @import("vfs-protocol");
/// The kind of a filesystem node — re-exported so a caller need not import the /// The kind of a filesystem node — re-exported so a caller need not import the
@@ -76,7 +77,7 @@ const Route = union(enum) {
fn resolve(path: []const u8, flags: usize) ?Route { fn resolve(path: []const u8, flags: usize) ?Route {
var out: [224]u8 = undefined; var out: [224]u8 = undefined;
const route = system.fsResolve(path, flags, &out) orelse return null; const route = fsResolve(path, flags, &out) orelse return null;
switch (route) { switch (route) {
.kernel => |token| return .{ .kernel = token }, .kernel => |token| return .{ .kernel = token },
.backend => |b| { .backend => |b| {
@@ -118,7 +119,7 @@ pub const File = struct {
/// null on error. /// null on error.
pub fn read(self: *File, buffer: []u8) ?usize { pub fn read(self: *File, buffer: []u8) ?usize {
const h = self.backend orelse { const h = self.backend orelse {
const n = system.fsNodeRead(self.node, self.offset, buffer) orelse return null; const n = fsNodeRead(self.node, self.offset, buffer) orelse return null;
self.offset += n; self.offset += n;
return n; return n;
}; };
@@ -164,8 +165,8 @@ pub const File = struct {
/// This file's metadata. /// This file's metadata.
pub fn attributes(self: *File) ?Attributes { pub fn attributes(self: *File) ?Attributes {
const h = self.backend orelse { const h = self.backend orelse {
const a = system.fsNodeStatus(self.node) orelse return null; const a = fsNodeStatus(self.node) orelse return null;
return .{ .size = a.size, .kind = if (a.kind == system.file_kind_directory) .directory else .regular, .mtime = a.mtime }; return .{ .size = a.size, .kind = if (a.kind == file_kind_directory) .directory else .regular, .mtime = a.mtime };
}; };
const request = vfs_protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 }; const request = vfs_protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
var buffer: [@sizeOf(vfs_protocol.FileStatus)]u8 = undefined; var buffer: [@sizeOf(vfs_protocol.FileStatus)]u8 = undefined;
@@ -234,14 +235,14 @@ pub const Directory = struct {
/// on error. /// on error.
pub fn next(self: *Directory, entry: *Entry) bool { pub fn next(self: *Directory, entry: *Entry) bool {
const h = self.backend orelse { const h = self.backend orelse {
var buffer: [@sizeOf(system.DirectoryEntryHeader) + 64]u8 = undefined; var buffer: [@sizeOf(DirectoryEntryHeader) + 64]u8 = undefined;
const n = system.fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false; const n = fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false;
if (n < @sizeOf(system.DirectoryEntryHeader)) return false; // end if (n < @sizeOf(DirectoryEntryHeader)) return false; // end
const header = std.mem.bytesToValue(system.DirectoryEntryHeader, buffer[0..@sizeOf(system.DirectoryEntryHeader)]); const header = std.mem.bytesToValue(DirectoryEntryHeader, buffer[0..@sizeOf(DirectoryEntryHeader)]);
entry.kind = if (header.kind == system.file_kind_directory) .directory else .regular; entry.kind = if (header.kind == file_kind_directory) .directory else .regular;
entry.size = header.size; entry.size = header.size;
const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len); const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len);
@memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(system.DirectoryEntryHeader)..][0..nlen]); @memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(DirectoryEntryHeader)..][0..nlen]);
entry.name_len = nlen; entry.name_len = nlen;
self.cursor += 1; self.cursor += 1;
return true; return true;
@@ -343,12 +344,84 @@ pub fn rename(old_path: []const u8, new_path: []const u8) bool {
/// the kernel VFS then routes everything under `target` to that backend. /// the kernel VFS then routes everything under `target` to that backend.
/// Possession of the endpoint handle is the capability. Returns true on success. /// Possession of the endpoint handle is the capability. Returns true on success.
pub fn mount(target: []const u8, backend: ipc.Handle) bool { pub fn mount(target: []const u8, backend: ipc.Handle) bool {
return system.fsMount(target, backend, ""); return fsMount(target, backend, "");
} }
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches /// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two /// the backend as `rewrite` + the mount-relative tail. How one volume serves two
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree). /// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool { pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
return system.fsMount(target, backend, rewrite); return fsMount(target, backend, rewrite);
}
// --- raw filesystem syscalls, formerly in the system.zig dumping ground ---
pub const FileAttributes = abi.FileAttributes;
pub const DirectoryEntryHeader = abi.DirectoryEntryHeader;
pub const file_kind_regular = abi.file_kind_regular;
pub const file_kind_directory = abi.file_kind_directory;
/// Where fs_resolve routed a path: served by the kernel (a permanent node token for
/// `fs_node`) or by a user-space filesystem backend (an endpoint handle plus the rewritten
/// mount-relative path returned in the caller's buffer).
pub const FsRoute = union(enum) {
kernel: u64,
backend: struct { handle: usize, path_len: usize },
};
/// Route `path` through the kernel VFS. For a backend route the rewritten mount-relative
/// path lands in `out` (behind a kernel-written length prefix, already stripped here:
/// out[0..path_len] is the path).
pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute {
var rax: usize = undefined;
var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle
asm volatile ("syscall"
: [rax] "={rax}" (rax),
[rdx] "+{rdx}" (rdx),
: [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)),
[a0] "{rdi}" (@intFromPtr(path.ptr)),
[a1] "{rsi}" (path.len),
[a3] "{r10}" (@intFromPtr(out.ptr)),
[a4] "{r8}" (out.len),
: .{ .rcx = true, .r11 = true, .memory = true });
if (@as(isize, @bitCast(rax)) < 0) return null;
if (rax == abi.fs_route_kernel) return .{ .kernel = rdx };
if (rax != abi.fs_route_backend) return null;
const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8);
if (path_len + 2 > out.len) return null;
std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]);
return .{ .backend = .{ .handle = rdx, .path_len = path_len } };
}
/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read).
pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len);
if (@as(isize, @bitCast(r)) < 0) return null;
return r;
}
/// A kernel-served node's metadata (fs_node status).
pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes {
var attrs: abi.FileAttributes = undefined;
const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attrs), @sizeOf(abi.FileAttributes));
if (@as(isize, @bitCast(r)) < 0) return null;
return attrs;
}
/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills `out` with
/// [DirectoryEntryHeader][name]; returns total bytes (0 = end).
pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize {
const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len);
if (@as(isize, @bitCast(r)) < 0) return null;
return r;
}
/// Mount a userspace filesystem's endpoint at `prefix`, with an optional backend-side
/// `rewrite` prefix ("" = none). Possession of the endpoint handle is the capability.
pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool {
return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0;
}
pub fn fsUnmount(prefix: []const u8) bool {
return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0;
} }
+3 -3
View File
@@ -23,8 +23,8 @@
const std = @import("std"); const std = @import("std");
const abi = @import("abi"); const abi = @import("abi");
const ipc = @import("ipc.zig"); const ipc = @import("ipc");
const system = @import("system.zig"); const time = @import("time");
const input_protocol = @import("input-protocol"); const input_protocol = @import("input-protocol");
pub const DeviceKind = input_protocol.DeviceKind; pub const DeviceKind = input_protocol.DeviceKind;
@@ -51,7 +51,7 @@ fn lookupService() ?ipc.Handle {
var attempts: usize = 0; var attempts: usize = 0;
while (attempts < 100) : (attempts += 1) { while (attempts < 100) : (attempts += 1) {
if (ipc.lookup(.input)) |handle| return handle; if (ipc.lookup(.input)) |handle| return handle;
system.sleep(50); time.sleepMillis(50);
} }
return null; return null;
} }
+1 -1
View File
@@ -4,7 +4,7 @@
//! added with the first server binary. //! added with the first server binary.
const abi = @import("abi"); const abi = @import("abi");
const sc = @import("system-call.zig"); const sc = @import("system-call");
/// A small-int handle into the calling process's handle table. /// A small-int handle into the calling process's handle table.
pub const Handle = usize; pub const Handle = usize;
-51
View File
@@ -1,51 +0,0 @@
//! The per-process logger: std.log wired to the tagged kernel log ring.
//!
//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped
//! logger); this backend formats the line into a fixed buffer and emits ONE
//! `debug_write` record carrying the level. The kernel stamps the record with
//! the sender's pid and task name (its binary path) — the process does NOT put
//! its own name in the payload; attribution is the kernel's, structural and
//! unforgeable. Serial shows the kernel-rendered `<path>: message` line, and
//! the logger service demultiplexes the ring into one file per process.
//!
//! Installed for every user binary by the root shim (library/runtime/root.zig)
//! via `std_options`; a program can override by declaring its own
//! `pub const std_options`.
const std = @import("std");
const system = @import("system.zig");
fn levelOf(comptime level: std.log.Level) system.KlogLevel {
return switch (level) {
.err => .err,
.warn => .warn,
.info => .info,
.debug => .debug,
};
}
pub fn logFn(
comptime level: std.log.Level,
comptime scope: @EnumLiteral(),
comptime format: []const u8,
args: anytype,
) void {
// One record = one line = at most klog_maximum_message bytes of payload.
// On overflow keep what fits and end with "~" so the record is still a
// whole line (the kernel would split an embedded rest anyway).
var buffer: [256]u8 = undefined;
const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") ";
const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: {
buffer[buffer.len - 1] = '~';
break :truncated buffer[0..];
};
_ = system.writeRecord(levelOf(level), line);
}
/// The std.Options the root shim installs unless the program overrides it.
/// Debug level: filtering is the log *reader's* job here — the ring is cheap,
/// serial is a dev convenience, and the logger service keeps everything.
pub const default_options: std.Options = .{
.log_level = .debug,
.logFn = logFn,
};
+97
View File
@@ -0,0 +1,97 @@
//! The per-process logger: std.log wired to the tagged kernel log ring.
//!
//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped
//! logger); this backend formats the line into a fixed buffer and emits ONE
//! `debug_write` record carrying the level. The kernel stamps the record with
//! the sender's pid and task name (its binary path) — the process does NOT put
//! its own name in the payload; attribution is the kernel's, structural and
//! unforgeable. Serial shows the kernel-rendered `<path>: message` line, and
//! the logger service demultiplexes the ring into one file per process.
//!
//! Installed for every user binary by the root shim (library/runtime/root.zig)
//! via `std_options`; a program can override by declaring its own
//! `pub const std_options`.
const std = @import("std");
const abi = @import("abi");
const sc = @import("system-call");
// --- the tagged log ring: raw wrappers + record types, formerly in the system.zig dump ---
/// A log record's level and the ring's framing types (re-exported from the shared ABI so
/// callers and the logger service don't import `abi` themselves).
pub const KlogLevel = abi.KlogLevel;
pub const KlogStatus = abi.KlogStatus;
pub const KlogRecordHeader = abi.KlogRecordHeader;
pub const klog_record_header_size = abi.klog_record_header_size;
pub const klog_record_alignment = abi.klog_record_alignment;
pub const klog_record_magic = abi.klog_record_magic;
pub const klog_flag_truncated = abi.klog_flag_truncated;
pub const klog_maximum_message = abi.klog_maximum_message;
pub const maximum_process_name = abi.maximum_process_name;
/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary output goes
/// through std.log -> writeRecord). The kernel stamps the record with this process's id
/// and name. Returns the byte count, or a wrapped -1.
pub fn write(message: []const u8) usize {
return writeRecord(.raw, message);
}
/// Emit one leveled record into the tagged kernel log ring. The kernel stamps
/// pid/name/sequence/timestamp; the payload should be a single line.
pub fn writeRecord(level: KlogLevel, message: []const u8) usize {
return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level));
}
/// Copy framed records out of the tagged kernel log ring starting at stream `offset` into
/// `out`. Returns the byte count (0 = caught up), or null when `offset` fell behind the
/// ring's tail or lies past its head (re-sync via `klogStatus`).
pub fn klogRead(offset: u64, out: []u8) ?usize {
const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
if (@as(isize, @bitCast(r)) < 0) return null;
return r;
}
/// The log ring's live cursors (oldest retained offset, end of stream, next sequence)
/// plus the wall-clock time of boot — how a log reader starts, detects loss, and names a
/// per-boot log directory.
pub fn klogStatus() ?KlogStatus {
var status: KlogStatus = undefined;
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null;
return status;
}
fn levelOf(comptime level: std.log.Level) KlogLevel {
return switch (level) {
.err => .err,
.warn => .warn,
.info => .info,
.debug => .debug,
};
}
pub fn logFn(
comptime level: std.log.Level,
comptime scope: @EnumLiteral(),
comptime format: []const u8,
args: anytype,
) void {
// One record = one line = at most klog_maximum_message bytes of payload.
// On overflow keep what fits and end with "~" so the record is still a
// whole line (the kernel would split an embedded rest anyway).
var buffer: [256]u8 = undefined;
const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") ";
const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: {
buffer[buffer.len - 1] = '~';
break :truncated buffer[0..];
};
_ = writeRecord(levelOf(level), line);
}
/// The std.Options the root shim installs unless the program overrides it.
/// Debug level: filtering is the log *reader's* job here — the ring is cheap,
/// serial is a dev convenience, and the logger service keeps everything.
pub const default_options: std.Options = .{
.log_level = .debug,
.logFn = logFn,
};
@@ -5,7 +5,7 @@
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md. //! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
const abi = @import("abi"); const abi = @import("abi");
const sc = @import("system-call.zig"); const sc = @import("system-call");
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are /// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
/// opt-in for specific hardware — see `abi`. /// opt-in for specific hardware — see `abi`.
@@ -19,8 +19,8 @@
const std = @import("std"); const std = @import("std");
const builtin = @import("builtin"); const builtin = @import("builtin");
const abi = @import("abi"); const abi = @import("abi");
const system_calls = @import("system.zig"); const sc = @import("system-call");
const Mutex = @import("thread.zig").Thread.Mutex; const Mutex = @import("thread").Thread.Mutex;
const page_size = abi.page_size; const page_size = abi.page_size;
@@ -63,8 +63,8 @@ fn payloadOf(block: *Block) [*]u8 {
/// grants usually are adjacent). Returns false if the kernel is out of memory. /// grants usually are adjacent). Returns false if the kernel is out of memory.
fn grow(minimum_bytes: usize) bool { fn grow(minimum_bytes: usize) bool {
const bytes = alignUp(@max(minimum_bytes, chunk), page_size); const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE); const ret = sc.systemCall2(.mmap, bytes, abi.prot_read | abi.prot_write);
if (system_calls.mmapFailed(ret)) return false; if (ret > ~@as(usize, 0) - 4095) return false; // a wrapped -errno lands in the top page
const block: *Block = @ptrFromInt(ret); const block: *Block = @ptrFromInt(ret);
block.size = bytes; block.size = bytes;
+48
View File
@@ -0,0 +1,48 @@
//! library/kernel/memory — the process's memory interface: the heap allocator, DMA-capable
//! buffers, shared-memory regions, and the raw `mmap` grant they all sit on. One flat module
//! (formerly runtime.heap / runtime.dma / runtime.shared_memory, plus the `mmap` wrappers that
//! lived in the system.zig dumping ground). Its private files are heap.zig, dma.zig, and
//! shared-memory.zig — imported only here, so the heap's state and C symbols exist once.
const abi = @import("abi");
const sc = @import("system-call");
const heap = @import("heap.zig");
const dma = @import("dma.zig");
const shared = @import("shared-memory.zig");
// --- the heap: a std.mem.Allocator over a first-fit free list (C malloc/free are also
// exported from heap.zig, compiled once here) ---
pub const allocator = heap.allocator;
// --- the raw grant every allocation sits on ---
pub const PROT_READ: usize = abi.prot_read;
pub const PROT_WRITE: usize = abi.prot_write;
pub const PROT_EXEC: usize = abi.prot_exec;
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable memory and
/// return the base virtual address. On failure returns a value in the top page (`mmapFailed`).
pub fn mmap(len: usize, prot: usize) usize {
return sc.systemCall2(.mmap, len, prot);
}
/// Release a range previously handed out by `mmap`.
pub fn munmap(base: usize, len: usize) usize {
return sc.systemCall2(.munmap, base, len);
}
/// Whether an `mmap` return value is an error (a wrapped -errno lands in the top page).
pub inline fn mmapFailed(ret: usize) bool {
return ret > ~@as(usize, 0) - 4095;
}
// --- DMA-capable buffers: physically contiguous, pinned, uncacheable, physical address known ---
pub const DmaRegion = dma.Region;
pub const dma_coherent = dma.coherent;
pub const dma_write_combining = dma.write_combining;
pub const dma_below_4g = dma.below_4g;
pub const dmaAlloc = dma.alloc;
pub const dmaFree = dma.free;
// --- shared-memory regions: a capability handed to another process over an ipc_call send_cap ---
pub const SharedRegion = shared.Region;
pub const sharedCreate = shared.create;
pub const sharedMap = shared.map;
pub const sharedPhysical = shared.physical;
@@ -6,8 +6,8 @@
//! generalization of capability passing from endpoints to memory objects. //! generalization of capability passing from endpoints to memory objects.
const abi = @import("abi"); const abi = @import("abi");
const sc = @import("system-call.zig"); const sc = @import("system-call");
const ipc = @import("ipc.zig"); const ipc = @import("ipc");
inline fn failed(r: usize) bool { inline fn failed(r: usize) bool {
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
+80 -5
View File
@@ -7,9 +7,9 @@
const std = @import("std"); const std = @import("std");
const abi = @import("abi"); const abi = @import("abi");
const sc = @import("system-call.zig"); const sc = @import("system-call");
const ipc = @import("ipc.zig"); const ipc = @import("ipc");
const system = @import("system.zig"); const time = @import("time");
/// Everything a program receives at entry. Passed to /// Everything a program receives at entry. Passed to
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep /// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
@@ -112,14 +112,14 @@ pub fn sendSignal(id: u32, signal: Signal) bool {
/// (arm `system.timerOnce`, keep serving) instead of calling this. /// (arm `system.timerOnce`, keep serving) instead of calling this.
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void { pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
_ = sendSignal(id, .terminate); _ = sendSignal(id, .terminate);
_ = system.timerOnce(exit_endpoint, deadline_ms); _ = time.timerOnce(exit_endpoint, deadline_ms);
var receive: [8]u8 = undefined; var receive: [8]u8 = undefined;
while (true) { while (true) {
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null); const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
if (got.isChildExit() and got.childProcessId() == id) return; if (got.isChildExit() and got.childProcessId() == id) return;
if (got.isTimer()) break; // the deadline passed first — escalate if (got.isTimer()) break; // the deadline passed first — escalate
} }
_ = system.kill(id); _ = kill(id);
while (true) { while (true) {
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null); const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
if (got.isChildExit() and got.childProcessId() == id) return; if (got.isChildExit() and got.childProcessId() == id) return;
@@ -135,3 +135,78 @@ pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
pub fn subscribeExits(endpoint: usize) bool { pub fn subscribeExits(endpoint: usize) bool {
return sc.systemCall1(.process_subscribe, endpoint) == 0; return sc.systemCall1(.process_subscribe, endpoint) == 0;
} }
// --- raw process syscalls, formerly in the system.zig dumping ground ---
/// One `processes` entry — re-exported from the shared ABI so a program can declare its
/// snapshot buffer without importing `abi` itself.
pub const ProcessDescriptor = abi.ProcessDescriptor;
/// Give up the rest of this quantum.
pub fn yield() void {
_ = sc.systemCall0(.yield);
}
/// End the process. Never returns.
pub fn exit(code: usize) noreturn {
_ = sc.systemCall1(.exit, code);
unreachable; // the kernel never returns from exit
}
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3 process,
/// returning the child's process id (or null). argv[0] is `name`, and the caller becomes
/// its **supervisor** — the only process allowed to `kill` it.
pub fn spawn(name: []const u8) ?u32 {
return spawnSupervised(name, &.{}, null);
}
/// Like `spawn`, but hands the child argv[1..] (argv[0] is still `name`).
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
return spawnSupervised(name, arguments, null);
}
/// The full spawn: argv[1..] for the child, and an optional endpoint the kernel notifies
/// when the child ends (any way — clean exit, fault, or `kill`), delivered via
/// `ipc.replyWait` as a child-exit badge (`ipc.Received.isChildExit`/`childProcessId`), so
/// one endpoint can supervise many children. Returns the child's process id, or null.
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
var blob: [256]u8 = undefined;
var len: usize = 0;
for (arguments, 0..) |argument, i| {
if (i != 0) {
if (len >= blob.len) return null;
blob[len] = 0;
len += 1;
}
if (len + argument.len > blob.len) return null;
@memcpy(blob[len..][0..argument.len], argument);
len += argument.len;
}
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
return @intCast(r);
}
/// Snapshot the process table into `out` and return the total number of live processes
/// (which may exceed `out.len`; call again with a larger buffer). Kernel tasks are
/// included, with an empty name. The primitive `ps` is built on.
pub fn processes(out: []ProcessDescriptor) usize {
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
}
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
pub fn isProcessRunning(name: []const u8) bool {
var table: [32]ProcessDescriptor = undefined;
const total = processes(&table);
for (table[0..@min(total, table.len)]) |descriptor| {
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
}
return false;
}
/// End process `id`. Only its supervisor — the process that spawned it — may; anyone else
/// gets false, as does a stale or unknown id. Delivery is prompt but asynchronous, like a
/// signal. True means the kill is accepted and irrevocable.
pub fn kill(id: u32) bool {
return sc.systemCall1(.process_kill, id) == 0;
}
+44 -67
View File
@@ -1,74 +1,51 @@
//! danos user-space runtime library — a nascent libc. Every user binary (init, //! runtime.zig — a **compatibility shim** for the runtime split (reorg C1–C5).
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
//! shim. It is compiled into each binary (inheriting its `.large` code model and
//! freestanding target), so all user programs share one implementation.
//! //!
//! A user binary only defines a `pub fn main() void` or //! `library/runtime` became `library/kernel`, and the one giant `runtime` module is being
//! `pub fn main(init: runtime.process.Init) void` (arguments arrive via `init`). //! split into directly-importable concern modules (`ipc`, `memory`, `process`, `time`,
//! The panic handler and the `_start` entry pull live in the shared compilation //! `logging`, `file-system`, `service`, `thread`, `start`, plus the device/service clients).
//! root, library/runtime/root.zig, which build.zig wires around every program — //! This file re-exports those modules under the old `runtime.*` names so the ~38 consumers
//! nothing to declare per source file. //! keep compiling until each is migrated to direct imports. Deleted in step C5.
pub const system = @import("system.zig"); pub const system = @import("system"); // the system.zig compatibility shim
pub const log = @import("log.zig"); pub const ipc = @import("ipc");
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls pub const memory = @import("memory");
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md). pub const allocator = memory.allocator;
pub const time = @import("time.zig"); pub const process = @import("process");
pub const heap = @import("heap.zig"); pub const service = @import("service");
pub const ipc = @import("ipc.zig"); pub const Thread = @import("thread").Thread;
pub const start = @import("start.zig"); pub const time = @import("time");
pub const log = @import("logging");
/// Client for talking to the device manager (the hello handshake a supervised pub const logging = @import("logging");
/// driver owes at startup). See library/runtime/device-manager.zig. The wire pub const fs = @import("file-system");
/// protocol itself is the library/protocol/device-manager module, imported pub const start = @import("start");
/// directly by drivers and services that speak it.
pub const device_manager = @import("device-manager.zig");
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
/// service. See library/runtime/input.zig and system/services/input/.
pub const input = @import("input.zig");
/// POSIX-style file API: open/read/write/lseek/stat/close.
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
/// Device access for drivers: enumerate/claim/mmioMap.
pub const device = @import("device.zig");
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
pub const dma = @import("dma.zig");
/// Shared cacheable memory: create a region + capability, pass the capability to another
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shared-memory.zig
/// and docs/display-v2.md.
pub const shared_memory = @import("shared-memory.zig");
// The USB class-driver client moved to its domain home, library/device/usb (module
// "usb"): it is bus-family logic, not core runtime, and re-exporting it here compiled it
// into every binary. USB class drivers import it directly with @import("usb").
/// Block-device client: read/write a block device (a USB stick, via
/// usb-storage). See library/runtime/block.zig.
pub const block = @import("block.zig");
/// Display-service client: query the mode, and (from D3) create layers, draw, and
/// present frames. See library/runtime/display.zig and system/services/display/.
pub const display = @import("display.zig");
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
/// layer danos programs use directly, and where the operations that later become
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
pub const fs = @import("fs.zig");
/// Re-exported so the root shim (root.zig) can install it as the panic handler.
pub const panic = start.panic; pub const panic = start.panic;
/// Process entry types: the `Init` handed to `main`, and its `Arguments`. // Device / service clients (relocated to library/device and library/client in C3/C4).
pub const process = @import("process.zig"); pub const device = @import("device");
pub const device_manager = @import("device-manager");
pub const input = @import("input");
pub const block = @import("block");
pub const display = @import("display");
/// Threads: `runtime.Thread`, std.Thread-shaped, over the private thread ABI /// The heap as a namespace (`runtime.heap.allocator()`), plus `runtime.allocator`.
/// (docs/threading.md). A binary must be built multi-threaded to spawn. pub const heap = struct {
pub const Thread = @import("thread.zig").Thread; pub const allocator = memory.allocator;
};
/// The service harness: one replyWait loop folding requests, signals, and /// `runtime.dma.*` mapped onto the flat `memory` API (memory groups heap+dma+shared-memory).
/// notifications into callbacks (docs/process-lifecycle.md). pub const dma = struct {
pub const service = @import("service.zig"); pub const Region = memory.DmaRegion;
pub const coherent = memory.dma_coherent;
pub const write_combining = memory.dma_write_combining;
pub const below_4g = memory.dma_below_4g;
pub const alloc = memory.dmaAlloc;
pub const free = memory.dmaFree;
};
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code. /// `runtime.shared_memory.*` mapped onto the flat `memory` API.
pub const allocator = heap.allocator; pub const shared_memory = struct {
pub const Region = memory.SharedRegion;
pub const create = memory.sharedCreate;
pub const map = memory.sharedMap;
pub const physical = memory.sharedPhysical;
};
+2 -2
View File
@@ -13,8 +13,8 @@
//! is the diagnosis (see docs/ipc.md). //! is the diagnosis (see docs/ipc.md).
const abi = @import("abi"); const abi = @import("abi");
const ipc = @import("ipc.zig"); const ipc = @import("ipc");
const process = @import("process.zig"); const process = @import("process");
pub const Callbacks = struct { pub const Callbacks = struct {
/// Called once with the service's endpoint before the loop starts — the /// Called once with the service's endpoint before the loop starts — the
+5 -5
View File
@@ -4,8 +4,8 @@
//! the whole runtime is linked in. //! the whole runtime is linked in.
const std = @import("std"); const std = @import("std");
const system = @import("system.zig"); const logging = @import("logging");
const process = @import("process.zig"); const process = @import("process");
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V /// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the /// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
@@ -31,7 +31,7 @@ export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
.count = stack[0], .count = stack[0],
.vector = @ptrCast(stack + 1), .vector = @ptrCast(stack + 1),
} }; } };
system.exit(callMain(init)); process.exit(callMain(init));
} }
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig: /// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
@@ -68,7 +68,7 @@ fn callMain(init: process.Init) u8 {
const payload = @call(.auto, root.main, call_arguments) catch |err| { const payload = @call(.auto, root.main, call_arguments) catch |err| {
var buffer: [128]u8 = undefined; var buffer: [128]u8 = undefined;
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n"; const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
_ = system.write(line); _ = logging.write(line);
return 1; // distinct from panic's 127 return 1; // distinct from panic's 127
}; };
if (@TypeOf(payload) == void) return 0; if (@TypeOf(payload) == void) return 0;
@@ -82,6 +82,6 @@ fn callMain(init: process.Init) u8 {
/// No runtime to unwind into — report a panic as a nonzero exit code. /// No runtime to unwind into — report a panic as a nonzero exit code.
pub const panic = std.debug.FullPanic(struct { pub const panic = std.debug.FullPanic(struct {
fn panic(_: []const u8, _: ?usize) noreturn { fn panic(_: []const u8, _: ?usize) noreturn {
system.exit(127); process.exit(127);
} }
}.panic); }.panic);
+60 -260
View File
@@ -1,266 +1,66 @@
//! Typed system_call surface for user space — thin wrappers over the raw `system_call` //! system.zig — a **compatibility shim**, not the real home of anything anymore.
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single //!
//! source of truth shared with the kernel dispatcher. //! The runtime's syscall surface used to be dumped here in one file. It has been split by
//! concern into `time` / `logging` / `process` / `file-system` / `memory`. This re-exports
//! the old flat `system.*` names from those homes so consumers that still write
//! `runtime.system.write` (etc.) keep compiling until they migrate to the concern modules.
//! Deleted once nothing references `runtime.system` (reorg step C5).
const std = @import("std"); const memory = @import("memory");
const abi = @import("abi"); const process = @import("process");
const sc = @import("system-call.zig"); const time = @import("time");
const logging = @import("logging");
const file_system = @import("file-system");
/// `mmap` protection flags (matching the usual C bit values). Grants are always // memory
/// readable+writable today; the kernel does not yet honour finer prot. pub const PROT_READ = memory.PROT_READ;
pub const PROT_READ: usize = abi.prot_read; pub const PROT_WRITE = memory.PROT_WRITE;
pub const PROT_WRITE: usize = abi.prot_write; pub const PROT_EXEC = memory.PROT_EXEC;
pub const PROT_EXEC: usize = abi.prot_exec; pub const mmap = memory.mmap;
pub const munmap = memory.munmap;
pub const mmapFailed = memory.mmapFailed;
/// One `processes` entry — re-exported from the shared ABI so a user program can // process
/// declare its snapshot buffer without importing `abi` itself. pub const ProcessDescriptor = process.ProcessDescriptor;
pub const ProcessDescriptor = abi.ProcessDescriptor; pub const yield = process.yield;
pub const exit = process.exit;
pub const spawn = process.spawn;
pub const spawnWithArguments = process.spawnWithArguments;
pub const spawnSupervised = process.spawnSupervised;
pub const processes = process.processes;
pub const isProcessRunning = process.isProcessRunning;
pub const kill = process.kill;
/// Give up the rest of this quantum. // time
pub fn yield() void { pub const clock = time.clock;
_ = sc.systemCall0(.yield); pub const wallClock = time.wallClock;
} pub const sleep = time.sleepMillis;
pub const timerOnce = time.timerOnce;
/// The tagged-log level of a record — re-exported so runtime.log and the logger // logging
/// service don't import `abi` themselves. pub const write = logging.write;
pub const KlogLevel = abi.KlogLevel; pub const writeRecord = logging.writeRecord;
pub const KlogStatus = abi.KlogStatus; pub const klogRead = logging.klogRead;
pub const KlogRecordHeader = abi.KlogRecordHeader; pub const klogStatus = logging.klogStatus;
pub const klog_record_header_size = abi.klog_record_header_size; pub const KlogLevel = logging.KlogLevel;
pub const klog_record_alignment = abi.klog_record_alignment; pub const KlogStatus = logging.KlogStatus;
pub const klog_record_magic = abi.klog_record_magic; pub const KlogRecordHeader = logging.KlogRecordHeader;
pub const klog_flag_truncated = abi.klog_flag_truncated; pub const klog_record_header_size = logging.klog_record_header_size;
pub const klog_maximum_message = abi.klog_maximum_message; pub const klog_record_alignment = logging.klog_record_alignment;
pub const maximum_process_name = abi.maximum_process_name; pub const klog_record_magic = logging.klog_record_magic;
pub const FileAttributes = abi.FileAttributes; pub const klog_flag_truncated = logging.klog_flag_truncated;
pub const DirectoryEntryHeader = abi.DirectoryEntryHeader; pub const klog_maximum_message = logging.klog_maximum_message;
pub const file_kind_regular = abi.file_kind_regular; pub const maximum_process_name = logging.maximum_process_name;
pub const file_kind_directory = abi.file_kind_directory;
/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary // file-system
/// output goes through std.log -> writeRecord). The kernel stamps the record pub const FsRoute = file_system.FsRoute;
/// with this process's id and name. Returns the byte count, or a wrapped -1. pub const fsResolve = file_system.fsResolve;
pub fn write(message: []const u8) usize { pub const fsNodeRead = file_system.fsNodeRead;
return writeRecord(.raw, message); pub const fsNodeStatus = file_system.fsNodeStatus;
} pub const fsNodeReaddir = file_system.fsNodeReaddir;
pub const fsMount = file_system.fsMount;
/// Emit one leveled record into the tagged kernel log ring. The kernel stamps pub const fsUnmount = file_system.fsUnmount;
/// pid/name/sequence/timestamp; the payload should be a single line (embedded pub const FileAttributes = file_system.FileAttributes;
/// newlines split into further records). pub const DirectoryEntryHeader = file_system.DirectoryEntryHeader;
pub fn writeRecord(level: KlogLevel, message: []const u8) usize { pub const file_kind_regular = file_system.file_kind_regular;
return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level)); pub const file_kind_directory = file_system.file_kind_directory;
}
/// Block the caller for `ms` milliseconds.
pub fn sleep(ms: usize) void {
_ = sc.systemCall1(.sleep, ms);
}
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
/// and restart backoff are built from.
pub fn timerOnce(endpoint: usize, ms: u64) bool {
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
}
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
///
/// const deadline = clock() + timeout_ns;
/// while (clock() < deadline) { ... }
pub fn clock() u64 {
return @intCast(sc.systemCall0(.clock));
}
/// Wall-clock time in Unix epoch seconds (UTC) — the real date/time, from the RTC.
/// Unlike `clock` (monotonic since boot), this tracks calendar time, so it is what a
/// filesystem stamps as a file's modification time. Formatting it into a calendar
/// date/timezone is user-space policy layered on top.
pub fn wallClock() u64 {
return @intCast(sc.systemCall0(.wall_clock));
}
/// Copy bytes out of the tagged kernel log ring — framed records of everything
/// every process (and the kernel) has emitted — starting at stream offset
/// `offset`, into `out`. Returns the byte count (0 = caught up), or null when
/// `offset` fell behind the ring's tail (those records were overwritten) or
/// lies past its head; re-sync via `klogStatus`. A reader parses
/// [KlogRecordHeader][name][message] frames (8-byte aligned) from the bytes.
pub fn klogRead(offset: u64, out: []u8) ?usize {
const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
if (@as(isize, @bitCast(r)) < 0) return null;
return r;
}
/// The log ring's live cursors (oldest retained offset, end of stream, next
/// sequence number) plus the wall-clock time of boot — how a log reader starts,
/// detects loss, and names a per-boot log directory.
pub fn klogStatus() ?KlogStatus {
var status: KlogStatus = undefined;
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null;
return status;
}
/// Where fs_resolve routed a path: served by the kernel (a permanent node
/// token for fs_node) or by a userspace filesystem backend (an endpoint handle
/// plus the rewritten mount-relative path, returned in the caller's buffer).
pub const FsRoute = union(enum) {
kernel: u64,
backend: struct { handle: usize, path_len: usize },
};
/// Route `path` through the kernel VFS. For a backend route the rewritten
/// mount-relative path lands in `out` (behind a kernel-written length prefix,
/// already stripped here: out[0..path_len] is the path).
pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute {
var rax: usize = undefined;
var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle
asm volatile ("syscall"
: [rax] "={rax}" (rax),
[rdx] "+{rdx}" (rdx),
: [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)),
[a0] "{rdi}" (@intFromPtr(path.ptr)),
[a1] "{rsi}" (path.len),
[a3] "{r10}" (@intFromPtr(out.ptr)),
[a4] "{r8}" (out.len),
: .{ .rcx = true, .r11 = true, .memory = true });
if (@as(isize, @bitCast(rax)) < 0) return null;
if (rax == abi.fs_route_kernel) return .{ .kernel = rdx };
if (rax != abi.fs_route_backend) return null;
const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8);
if (path_len + 2 > out.len) return null;
std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]);
return .{ .backend = .{ .handle = rdx, .path_len = path_len } };
}
/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read).
pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len);
if (@as(isize, @bitCast(r)) < 0) return null;
return r;
}
/// A kernel-served node's metadata (fs_node status).
pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes {
var attributes: abi.FileAttributes = undefined;
const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attributes), @sizeOf(abi.FileAttributes));
if (@as(isize, @bitCast(r)) < 0) return null;
return attributes;
}
/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills
/// `out` with [DirectoryEntryHeader][name]; returns total bytes (0 = end).
pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize {
const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len);
if (@as(isize, @bitCast(r)) < 0) return null;
return r;
}
/// Mount a userspace filesystem's endpoint at `prefix`, with an optional
/// backend-side `rewrite` prefix ("" = none). Possession of the endpoint
/// handle is the capability.
pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool {
return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0;
}
pub fn fsUnmount(prefix: []const u8) bool {
return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0;
}
/// End the process. Never returns.
pub fn exit(code: usize) noreturn {
_ = sc.systemCall1(.exit, code);
unreachable; // the kernel never returns from exit
}
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
/// process, returning the child's process id (or null on failure). The child's
/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process
/// allowed to `kill` it. This is how a supervisor (the device manager) launches a
/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the
/// POSIX layer later).
pub fn spawn(name: []const u8) ?u32 {
return spawnSupervised(name, &.{}, null);
}
/// Like `spawn`, but hands the child command-line arguments: they arrive as
/// argv[1..] on its System V entry stack (argv[0] is still `name`).
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
return spawnSupervised(name, arguments, null);
}
/// The full spawn: command-line arguments for the child, and an optional endpoint
/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends
/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via
/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in
/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can
/// supervise many children. Arguments are marshalled to the kernel as one
/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the
/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or
/// null on failure.
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
var blob: [256]u8 = undefined;
var len: usize = 0;
for (arguments, 0..) |argument, i| {
if (i != 0) {
if (len >= blob.len) return null;
blob[len] = 0;
len += 1;
}
if (len + argument.len > blob.len) return null;
@memcpy(blob[len..][0..argument.len], argument);
len += argument.len;
}
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
return @intCast(r);
}
/// Snapshot the process table into `out` (up to its length) and return the total
/// number of live processes — which may exceed `out.len`; call again with a larger
/// buffer for the full listing. Kernel tasks are included, with an empty name.
/// The primitive `ps` is built on.
pub fn processes(out: []abi.ProcessDescriptor) usize {
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
}
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
pub fn isProcessRunning(name: []const u8) bool {
var table: [32]ProcessDescriptor = undefined;
const total = processes(&table);
for (table[0..@min(total, table.len)]) |descriptor| {
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
}
return false;
}
/// End process `id`. Only its supervisor — the process that spawned it — may;
/// anyone else gets false, as does a stale or unknown id (ids are never reused).
/// Delivery is prompt but asynchronous, like a signal: a target caught running on
/// another core dies at its next system call or timer tick. True means the kill
/// is accepted and irrevocable; the exit notification (if an endpoint was given
/// at spawn) confirms completion.
pub fn kill(id: u32) bool {
return sc.systemCall1(.process_kill, id) == 0;
}
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
/// memory and return the base virtual address. On failure returns a value in the
/// top page (see `mmapFailed`). The user heap grows through this call.
pub fn mmap(len: usize, prot: usize) usize {
return sc.systemCall2(.mmap, len, prot);
}
/// Release a range previously handed out by `mmap`.
pub fn munmap(base: usize, len: usize) usize {
return sc.systemCall2(.munmap, base, len);
}
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
/// -errno, which lands in the top page — no real grant base is ever that high).
pub inline fn mmapFailed(ret: usize) bool {
return ret > ~@as(usize, 0) - 4095;
}
+17 -6
View File
@@ -13,8 +13,19 @@
const std = @import("std"); const std = @import("std");
const builtin = @import("builtin"); const builtin = @import("builtin");
const abi = @import("abi"); const abi = @import("abi");
const sc = @import("system-call.zig"); const sc = @import("system-call");
const system = @import("system.zig");
// A thread allocates its own stack straight from the mmap syscall (not through the
// `memory` module) so `memory`'s heap can depend on this module's Mutex without a cycle.
inline fn mmapStack(len: usize) usize {
return sc.systemCall2(.mmap, len, abi.prot_read | abi.prot_write);
}
inline fn mmapFailed(ret: usize) bool {
return ret > ~@as(usize, 0) - 4095;
}
inline fn munmapStack(base: usize, len: usize) void {
_ = sc.systemCall2(.munmap, base, len);
}
/// True in a real danos binary; false when this module is compiled for host unit tests. /// True in a real danos binary; false when this module is compiled for host unit tests.
/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state /// The `Futex` seam and the test blocks below branch on it so the lock/condvar state
@@ -66,8 +77,8 @@ pub const Thread = struct {
} }
}; };
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE); const base = mmapStack(config.stack_size);
if (system.mmapFailed(base)) return error.SystemResources; if (mmapFailed(base)) return error.SystemResources;
// Top of the thread's own stack, downward: the closure, then a small per-thread TLS // Top of the thread's own stack, downward: the closure, then a small per-thread TLS
// block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is // block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is
@@ -88,7 +99,7 @@ pub const Thread = struct {
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr); const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr);
if (threadSpawnFailed(tid)) { if (threadSpawnFailed(tid)) {
_ = system.munmap(base, config.stack_size); munmapStack(base, config.stack_size);
return error.SystemResources; return error.SystemResources;
} }
return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size }; return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size };
@@ -99,7 +110,7 @@ pub const Thread = struct {
/// child-exit notification on it is this thread's. /// child-exit notification on it is this thread's.
pub fn join(self: Thread) void { pub fn join(self: Thread) void {
_ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited _ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited
_ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack munmapStack(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
} }
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is /// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
+35 -17
View File
@@ -11,7 +11,33 @@
//! CLOCK_REALTIME) layered on top later. //! CLOCK_REALTIME) layered on top later.
const std = @import("std"); const std = @import("std");
const system = @import("system.zig"); const sc = @import("system-call");
// --- raw syscall wrappers, formerly in the system.zig dumping ground ---
/// Monotonic nanoseconds since boot — the raw reading; `now()` wraps it in an `Instant`.
/// Never runs backward. Not wall-clock time (see `wallClock`).
pub fn clock() u64 {
return @intCast(sc.systemCall0(.clock));
}
/// Wall-clock time in Unix epoch seconds (UTC), from the RTC — the real date/time, what a
/// filesystem stamps as an mtime. Unlike `clock` (monotonic since boot), this is calendar time.
pub fn wallClock() u64 {
return @intCast(sc.systemCall0(.wall_clock));
}
/// Block the caller for `ms` milliseconds — the raw, coarse, allocation-free form.
pub fn sleepMillis(ms: u64) void {
_ = sc.systemCall1(.sleep, ms);
}
/// Arm a one-shot timer: after `ms` the kernel posts a timer notification
/// (`ipc.Received.isTimer`) to `endpoint` (a handle from `ipc.createIpcEndpoint`). Unlike
/// `sleep`, does not block — a service keeps serving IPC while the deadline is pending.
pub fn timerOnce(endpoint: usize, ms: u64) bool {
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
}
const nanos_per_micro: u64 = 1_000; const nanos_per_micro: u64 = 1_000;
const nanos_per_milli: u64 = 1_000_000; const nanos_per_milli: u64 = 1_000_000;
@@ -90,32 +116,27 @@ pub const Instant = struct {
/// The current monotonic time. /// The current monotonic time.
pub fn now() Instant { pub fn now() Instant {
return .{ .ns = system.clock() }; return .{ .ns = clock() };
} }
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that /// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
/// want a plain integer instead of an `Instant`. /// want a plain integer instead of an `Instant`.
pub fn monotonicNanos() u64 { pub fn monotonicNanos() u64 {
return system.clock(); return clock();
} }
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is /// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as /// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
/// "unavailable" instead of assuming the clock advances. /// "unavailable" instead of assuming the clock advances.
pub fn available() bool { pub fn available() bool {
return system.clock() != 0; return clock() != 0;
} }
/// Block the caller for at least `d`, rounded up to the kernel's millisecond /// Block the caller for at least `d`, rounded up to the kernel's millisecond
/// granularity. For sub-millisecond precision the scheduler cannot express, use /// granularity. For sub-millisecond precision the scheduler cannot express, use
/// `spin`. /// `spin`. (The raw millisecond form is `sleepMillis`.)
pub fn sleep(d: Duration) void { pub fn sleep(d: Duration) void {
system.sleep(d.ceilMillis()); sleepMillis(d.ceilMillis());
}
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
pub fn sleepMillis(ms: u64) void {
system.sleep(ms);
} }
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU /// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
@@ -126,13 +147,10 @@ pub fn spin(d: Duration) void {
while (!deadline.reached()) {} while (!deadline.reached()) {}
} }
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`): /// The ergonomic `Duration` form of `timerOnce`: arm a one-shot timer against `endpoint`
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there. /// for `d` (rounded up to milliseconds). Returns false if the timer could not be armed.
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
/// false if the timer could not be armed. See `system.timerOnce`.
pub fn after(endpoint: usize, d: Duration) bool { pub fn after(endpoint: usize, d: Duration) bool {
return system.timerOnce(endpoint, d.ceilMillis()); return timerOnce(endpoint, d.ceilMillis());
} }
test "Duration unit conversions round toward zero" { test "Duration unit conversions round toward zero" {