diff --git a/build.zig b/build.zig index b0f5294..c44c04b 100644 --- a/build.zig +++ b/build.zig @@ -387,31 +387,15 @@ pub fn build(b: *std.Build) void { // types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS // server. It never touches `boot-handoff` — user space has no business with the // loader↔kernel handoff. - const runtime_module = b.addModule("runtime", .{ - .root_source_file = b.path("library/kernel/runtime.zig"), - .imports = &.{ - .{ .name = "abi", .module = abi_module }, - .{ .name = "device-abi", .module = device_abi_module }, - .{ .name = "vfs-protocol", .module = vfs_protocol_module }, - .{ .name = "input-protocol", .module = input_protocol_module }, - }, - }); - - // The device-manager protocol: hello + (M18.2) tree reports, exposed as its - // own module like the other protocol modules. Imported through the runtime. + // Wire-protocol modules the kernel-library client wrappers (device-manager/block/display) + // and the services speak. The `runtime` module itself is defined below, after the + // library/kernel concern modules it shims over. const device_manager_protocol_module = b.addModule("device-manager-protocol", .{ .root_source_file = b.path("library/protocol/device-manager/device-manager-protocol.zig"), }); - runtime_module.addImport("device-manager-protocol", device_manager_protocol_module); - // The block protocol, so runtime.block (the block-device client) can speak it. - runtime_module.addImport("block-protocol", block_protocol_module); - - // The display protocol, so runtime.display (the compositor client) and the display - // service both speak it through the runtime, like the other protocol modules. const display_protocol_module = b.addModule("display-protocol", .{ .root_source_file = b.path("library/protocol/display/display-protocol.zig"), }); - runtime_module.addImport("display-protocol", display_protocol_module); // runtime.display client speaks it // The scanout protocol: the compositor's outbound present channel to a native scanout // driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md). @@ -433,6 +417,143 @@ pub fn build(b: *std.Build) void { .root_source_file = b.path("library/device/mmio/mmio.zig"), }); + // --- library/kernel: the userspace private-ABI library (kernel32-style), split by + // concern into directly-importable modules. `system.zig` (the old dumping ground) and + // `runtime.zig` (the old aggregator) are compatibility shims re-exporting these until the + // consumers migrate to direct imports (reorg C1–C5). The graph is a DAG: memory depends on + // thread (heap needs Thread.Mutex), and thread does its own raw mmap so there is no cycle. + const system_call_module = b.addModule("system-call", .{ + .root_source_file = b.path("library/kernel/system-call.zig"), + .imports = &.{.{ .name = "abi", .module = abi_module }}, + }); + const ipc_module = b.addModule("ipc", .{ + .root_source_file = b.path("library/kernel/ipc.zig"), + .imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } }, + }); + const time_module = b.addModule("time", .{ + .root_source_file = b.path("library/kernel/time.zig"), + .imports = &.{.{ .name = "system-call", .module = system_call_module }}, + }); + const thread_module = b.addModule("thread", .{ + .root_source_file = b.path("library/kernel/thread.zig"), + .imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } }, + }); + const logging_module = b.addModule("logging", .{ + .root_source_file = b.path("library/kernel/logging.zig"), + .imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } }, + }); + const process_module = b.addModule("process", .{ + .root_source_file = b.path("library/kernel/process.zig"), + .imports = &.{ + .{ .name = "abi", .module = abi_module }, + .{ .name = "system-call", .module = system_call_module }, + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "time", .module = time_module }, + }, + }); + const file_system_module = b.addModule("file-system", .{ + .root_source_file = b.path("library/kernel/file-system.zig"), + .imports = &.{ + .{ .name = "abi", .module = abi_module }, + .{ .name = "system-call", .module = system_call_module }, + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "vfs-protocol", .module = vfs_protocol_module }, + }, + }); + const memory_module = b.addModule("memory", .{ + .root_source_file = b.path("library/kernel/memory/memory.zig"), + .imports = &.{ + .{ .name = "abi", .module = abi_module }, + .{ .name = "system-call", .module = system_call_module }, + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "thread", .module = thread_module }, + }, + }); + const service_module = b.addModule("service", .{ + .root_source_file = b.path("library/kernel/service.zig"), + .imports = &.{ + .{ .name = "abi", .module = abi_module }, + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "process", .module = process_module }, + }, + }); + const start_module = b.addModule("start", .{ + .root_source_file = b.path("library/kernel/start.zig"), + .imports = &.{ .{ .name = "process", .module = process_module }, .{ .name = "logging", .module = logging_module } }, + }); + const device_module = b.addModule("device", .{ + .root_source_file = b.path("library/kernel/device.zig"), + .imports = &.{ + .{ .name = "abi", .module = abi_module }, + .{ .name = "device-abi", .module = device_abi_module }, + .{ .name = "system-call", .module = system_call_module }, + }, + }); + const device_manager_client_module = b.addModule("device-manager", .{ + .root_source_file = b.path("library/kernel/device-manager.zig"), + .imports = &.{ + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "time", .module = time_module }, + .{ .name = "device-manager-protocol", .module = device_manager_protocol_module }, + }, + }); + const block_client_module = b.addModule("block", .{ + .root_source_file = b.path("library/kernel/block.zig"), + .imports = &.{ + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "time", .module = time_module }, + .{ .name = "block-protocol", .module = block_protocol_module }, + }, + }); + const display_client_module = b.addModule("display", .{ + .root_source_file = b.path("library/kernel/display.zig"), + .imports = &.{ + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "time", .module = time_module }, + .{ .name = "display-protocol", .module = display_protocol_module }, + }, + }); + const input_client_module = b.addModule("input", .{ + .root_source_file = b.path("library/kernel/input.zig"), + .imports = &.{ + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "time", .module = time_module }, + .{ .name = "input-protocol", .module = input_protocol_module }, + }, + }); + // system.zig — compatibility shim for runtime.system.*, re-exporting the concern modules. + const system_module = b.addModule("system", .{ + .root_source_file = b.path("library/kernel/system.zig"), + .imports = &.{ + .{ .name = "memory", .module = memory_module }, + .{ .name = "process", .module = process_module }, + .{ .name = "time", .module = time_module }, + .{ .name = "logging", .module = logging_module }, + .{ .name = "file-system", .module = file_system_module }, + }, + }); + // runtime.zig — compatibility shim re-exporting every concern module under the old runtime.* names. + const runtime_module = b.addModule("runtime", .{ + .root_source_file = b.path("library/kernel/runtime.zig"), + .imports = &.{ + .{ .name = "system", .module = system_module }, + .{ .name = "ipc", .module = ipc_module }, + .{ .name = "memory", .module = memory_module }, + .{ .name = "process", .module = process_module }, + .{ .name = "service", .module = service_module }, + .{ .name = "thread", .module = thread_module }, + .{ .name = "time", .module = time_module }, + .{ .name = "logging", .module = logging_module }, + .{ .name = "file-system", .module = file_system_module }, + .{ .name = "start", .module = start_module }, + .{ .name = "device", .module = device_module }, + .{ .name = "device-manager", .module = device_manager_client_module }, + .{ .name = "input", .module = input_client_module }, + .{ .name = "block", .module = block_client_module }, + .{ .name = "display", .module = display_client_module }, + }, + }); + // A device driver's view of its claimed PCI function: config-space header fields, BAR // decode + map, and the capability walk (library/device/pci/pci.zig). The generic PCI // mechanics every leaf PCI driver used to re-derive inline. Imports runtime (device diff --git a/library/kernel/block.zig b/library/kernel/block.zig index ea7bfea..fc93e09 100644 --- a/library/kernel/block.zig +++ b/library/kernel/block.zig @@ -8,8 +8,8 @@ //! limit — the same handoff usb-storage uses toward the controller. const std = @import("std"); -const ipc = @import("ipc.zig"); -const system = @import("system.zig"); +const ipc = @import("ipc"); +const time = @import("time"); const block_protocol = @import("block-protocol"); pub const Geometry = struct { block_size: u32, block_count: u64 }; @@ -73,7 +73,7 @@ pub fn open() ?Device { // not sit a further minute pretending otherwise. while (attempts < 600) : (attempts += 1) { if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle }; - system.sleep(50); + time.sleepMillis(50); } return null; } diff --git a/library/kernel/device-manager.zig b/library/kernel/device-manager.zig index 518d521..10fd2d4 100644 --- a/library/kernel/device-manager.zig +++ b/library/kernel/device-manager.zig @@ -10,8 +10,8 @@ //! be — copied byte-for-byte into each bus and class driver — lives here once. const std = @import("std"); -const ipc = @import("ipc.zig"); -const system = @import("system.zig"); +const ipc = @import("ipc"); +const time = @import("time"); const device_manager_protocol = @import("device-manager-protocol"); /// What kind of driver is announcing itself (a bus that reports children, or a @@ -36,7 +36,7 @@ pub fn hello(role: Role, device_id: u64) ?ipc.Handle { var attempts: u32 = 0; const manager = while (attempts < lookup_attempts) : (attempts += 1) { if (ipc.lookup(.device_manager)) |handle| break handle; - system.sleep(lookup_pause_ms); + time.sleepMillis(lookup_pause_ms); } else { std.log.info("no device manager to hello", .{}); return null; diff --git a/library/kernel/device.zig b/library/kernel/device.zig index 3bb5217..b50703d 100644 --- a/library/kernel/device.zig +++ b/library/kernel/device.zig @@ -6,7 +6,7 @@ const std = @import("std"); const abi = @import("abi"); const device_abi = @import("device-abi"); -const sc = @import("system-call.zig"); +const sc = @import("system-call"); pub const DeviceDescriptor = device_abi.DeviceDescriptor; pub const ResourceDescriptor = device_abi.ResourceDescriptor; diff --git a/library/kernel/display.zig b/library/kernel/display.zig index f4b7ee5..0971444 100644 --- a/library/kernel/display.zig +++ b/library/kernel/display.zig @@ -4,8 +4,8 @@ //! reply marshalling. See system/services/display/ and docs/display.md. const std = @import("std"); -const ipc = @import("ipc.zig"); -const system = @import("system.zig"); +const ipc = @import("ipc"); +const time = @import("time"); const display_protocol = @import("display-protocol"); /// The display's current mode, as `info()` reports it. @@ -29,7 +29,7 @@ fn service() ?ipc.Handle { handle = h; return h; } - system.sleep(50); + time.sleepMillis(50); } return null; } diff --git a/library/kernel/fs.zig b/library/kernel/file-system.zig similarity index 77% rename from library/kernel/fs.zig rename to library/kernel/file-system.zig index cf1ec71..0b4e79e 100644 --- a/library/kernel/fs.zig +++ b/library/kernel/file-system.zig @@ -11,8 +11,9 @@ //! shape, unlike the POSIX fd model the old shim emulated. const std = @import("std"); -const ipc = @import("ipc.zig"); -const system = @import("system.zig"); +const abi = @import("abi"); +const sc = @import("system-call"); +const ipc = @import("ipc"); const vfs_protocol = @import("vfs-protocol"); /// The kind of a filesystem node — re-exported so a caller need not import the @@ -76,7 +77,7 @@ const Route = union(enum) { fn resolve(path: []const u8, flags: usize) ?Route { var out: [224]u8 = undefined; - const route = system.fsResolve(path, flags, &out) orelse return null; + const route = fsResolve(path, flags, &out) orelse return null; switch (route) { .kernel => |token| return .{ .kernel = token }, .backend => |b| { @@ -118,7 +119,7 @@ pub const File = struct { /// null on error. pub fn read(self: *File, buffer: []u8) ?usize { const h = self.backend orelse { - const n = system.fsNodeRead(self.node, self.offset, buffer) orelse return null; + const n = fsNodeRead(self.node, self.offset, buffer) orelse return null; self.offset += n; return n; }; @@ -164,8 +165,8 @@ pub const File = struct { /// This file's metadata. pub fn attributes(self: *File) ?Attributes { const h = self.backend orelse { - const a = system.fsNodeStatus(self.node) orelse return null; - return .{ .size = a.size, .kind = if (a.kind == system.file_kind_directory) .directory else .regular, .mtime = a.mtime }; + const a = fsNodeStatus(self.node) orelse return null; + return .{ .size = a.size, .kind = if (a.kind == file_kind_directory) .directory else .regular, .mtime = a.mtime }; }; const request = vfs_protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 }; var buffer: [@sizeOf(vfs_protocol.FileStatus)]u8 = undefined; @@ -234,14 +235,14 @@ pub const Directory = struct { /// on error. pub fn next(self: *Directory, entry: *Entry) bool { const h = self.backend orelse { - var buffer: [@sizeOf(system.DirectoryEntryHeader) + 64]u8 = undefined; - const n = system.fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false; - if (n < @sizeOf(system.DirectoryEntryHeader)) return false; // end - const header = std.mem.bytesToValue(system.DirectoryEntryHeader, buffer[0..@sizeOf(system.DirectoryEntryHeader)]); - entry.kind = if (header.kind == system.file_kind_directory) .directory else .regular; + var buffer: [@sizeOf(DirectoryEntryHeader) + 64]u8 = undefined; + const n = fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false; + if (n < @sizeOf(DirectoryEntryHeader)) return false; // end + const header = std.mem.bytesToValue(DirectoryEntryHeader, buffer[0..@sizeOf(DirectoryEntryHeader)]); + entry.kind = if (header.kind == file_kind_directory) .directory else .regular; entry.size = header.size; const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len); - @memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(system.DirectoryEntryHeader)..][0..nlen]); + @memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(DirectoryEntryHeader)..][0..nlen]); entry.name_len = nlen; self.cursor += 1; return true; @@ -343,12 +344,84 @@ pub fn rename(old_path: []const u8, new_path: []const u8) bool { /// the kernel VFS then routes everything under `target` to that backend. /// Possession of the endpoint handle is the capability. Returns true on success. pub fn mount(target: []const u8, backend: ipc.Handle) bool { - return system.fsMount(target, backend, ""); + return fsMount(target, backend, ""); } /// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches /// the backend as `rewrite` + the mount-relative tail. How one volume serves two /// mounts ("/mnt/usb" from its root, "/var" from its /var subtree). pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool { - return system.fsMount(target, backend, rewrite); + return fsMount(target, backend, rewrite); +} + +// --- raw filesystem syscalls, formerly in the system.zig dumping ground --- + +pub const FileAttributes = abi.FileAttributes; +pub const DirectoryEntryHeader = abi.DirectoryEntryHeader; +pub const file_kind_regular = abi.file_kind_regular; +pub const file_kind_directory = abi.file_kind_directory; + +/// Where fs_resolve routed a path: served by the kernel (a permanent node token for +/// `fs_node`) or by a user-space filesystem backend (an endpoint handle plus the rewritten +/// mount-relative path returned in the caller's buffer). +pub const FsRoute = union(enum) { + kernel: u64, + backend: struct { handle: usize, path_len: usize }, +}; + +/// Route `path` through the kernel VFS. For a backend route the rewritten mount-relative +/// path lands in `out` (behind a kernel-written length prefix, already stripped here: +/// out[0..path_len] is the path). +pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute { + var rax: usize = undefined; + var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle + asm volatile ("syscall" + : [rax] "={rax}" (rax), + [rdx] "+{rdx}" (rdx), + : [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)), + [a0] "{rdi}" (@intFromPtr(path.ptr)), + [a1] "{rsi}" (path.len), + [a3] "{r10}" (@intFromPtr(out.ptr)), + [a4] "{r8}" (out.len), + : .{ .rcx = true, .r11 = true, .memory = true }); + if (@as(isize, @bitCast(rax)) < 0) return null; + if (rax == abi.fs_route_kernel) return .{ .kernel = rdx }; + if (rax != abi.fs_route_backend) return null; + const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8); + if (path_len + 2 > out.len) return null; + std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]); + return .{ .backend = .{ .handle = rdx, .path_len = path_len } }; +} + +/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read). +pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize { + const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len); + if (@as(isize, @bitCast(r)) < 0) return null; + return r; +} + +/// A kernel-served node's metadata (fs_node status). +pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes { + var attrs: abi.FileAttributes = undefined; + const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attrs), @sizeOf(abi.FileAttributes)); + if (@as(isize, @bitCast(r)) < 0) return null; + return attrs; +} + +/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills `out` with +/// [DirectoryEntryHeader][name]; returns total bytes (0 = end). +pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize { + const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len); + if (@as(isize, @bitCast(r)) < 0) return null; + return r; +} + +/// Mount a userspace filesystem's endpoint at `prefix`, with an optional backend-side +/// `rewrite` prefix ("" = none). Possession of the endpoint handle is the capability. +pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool { + return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0; +} + +pub fn fsUnmount(prefix: []const u8) bool { + return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0; } diff --git a/library/kernel/input.zig b/library/kernel/input.zig index 882176d..cff95fe 100644 --- a/library/kernel/input.zig +++ b/library/kernel/input.zig @@ -23,8 +23,8 @@ const std = @import("std"); const abi = @import("abi"); -const ipc = @import("ipc.zig"); -const system = @import("system.zig"); +const ipc = @import("ipc"); +const time = @import("time"); const input_protocol = @import("input-protocol"); pub const DeviceKind = input_protocol.DeviceKind; @@ -51,7 +51,7 @@ fn lookupService() ?ipc.Handle { var attempts: usize = 0; while (attempts < 100) : (attempts += 1) { if (ipc.lookup(.input)) |handle| return handle; - system.sleep(50); + time.sleepMillis(50); } return null; } diff --git a/library/kernel/ipc.zig b/library/kernel/ipc.zig index 44561fa..813b0ef 100644 --- a/library/kernel/ipc.zig +++ b/library/kernel/ipc.zig @@ -4,7 +4,7 @@ //! added with the first server binary. const abi = @import("abi"); -const sc = @import("system-call.zig"); +const sc = @import("system-call"); /// A small-int handle into the calling process's handle table. pub const Handle = usize; diff --git a/library/kernel/log.zig b/library/kernel/log.zig deleted file mode 100644 index 5ae921d..0000000 --- a/library/kernel/log.zig +++ /dev/null @@ -1,51 +0,0 @@ -//! The per-process logger: std.log wired to the tagged kernel log ring. -//! -//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped -//! logger); this backend formats the line into a fixed buffer and emits ONE -//! `debug_write` record carrying the level. The kernel stamps the record with -//! the sender's pid and task name (its binary path) — the process does NOT put -//! its own name in the payload; attribution is the kernel's, structural and -//! unforgeable. Serial shows the kernel-rendered `: message` line, and -//! the logger service demultiplexes the ring into one file per process. -//! -//! Installed for every user binary by the root shim (library/runtime/root.zig) -//! via `std_options`; a program can override by declaring its own -//! `pub const std_options`. - -const std = @import("std"); -const system = @import("system.zig"); - -fn levelOf(comptime level: std.log.Level) system.KlogLevel { - return switch (level) { - .err => .err, - .warn => .warn, - .info => .info, - .debug => .debug, - }; -} - -pub fn logFn( - comptime level: std.log.Level, - comptime scope: @EnumLiteral(), - comptime format: []const u8, - args: anytype, -) void { - // One record = one line = at most klog_maximum_message bytes of payload. - // On overflow keep what fits and end with "~" so the record is still a - // whole line (the kernel would split an embedded rest anyway). - var buffer: [256]u8 = undefined; - const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") "; - const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: { - buffer[buffer.len - 1] = '~'; - break :truncated buffer[0..]; - }; - _ = system.writeRecord(levelOf(level), line); -} - -/// The std.Options the root shim installs unless the program overrides it. -/// Debug level: filtering is the log *reader's* job here — the ring is cheap, -/// serial is a dev convenience, and the logger service keeps everything. -pub const default_options: std.Options = .{ - .log_level = .debug, - .logFn = logFn, -}; diff --git a/library/kernel/logging.zig b/library/kernel/logging.zig new file mode 100644 index 0000000..9e3f1e7 --- /dev/null +++ b/library/kernel/logging.zig @@ -0,0 +1,97 @@ +//! The per-process logger: std.log wired to the tagged kernel log ring. +//! +//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped +//! logger); this backend formats the line into a fixed buffer and emits ONE +//! `debug_write` record carrying the level. The kernel stamps the record with +//! the sender's pid and task name (its binary path) — the process does NOT put +//! its own name in the payload; attribution is the kernel's, structural and +//! unforgeable. Serial shows the kernel-rendered `: message` line, and +//! the logger service demultiplexes the ring into one file per process. +//! +//! Installed for every user binary by the root shim (library/runtime/root.zig) +//! via `std_options`; a program can override by declaring its own +//! `pub const std_options`. + +const std = @import("std"); +const abi = @import("abi"); +const sc = @import("system-call"); + +// --- the tagged log ring: raw wrappers + record types, formerly in the system.zig dump --- + +/// A log record's level and the ring's framing types (re-exported from the shared ABI so +/// callers and the logger service don't import `abi` themselves). +pub const KlogLevel = abi.KlogLevel; +pub const KlogStatus = abi.KlogStatus; +pub const KlogRecordHeader = abi.KlogRecordHeader; +pub const klog_record_header_size = abi.klog_record_header_size; +pub const klog_record_alignment = abi.klog_record_alignment; +pub const klog_record_magic = abi.klog_record_magic; +pub const klog_flag_truncated = abi.klog_flag_truncated; +pub const klog_maximum_message = abi.klog_maximum_message; +pub const maximum_process_name = abi.maximum_process_name; + +/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary output goes +/// through std.log -> writeRecord). The kernel stamps the record with this process's id +/// and name. Returns the byte count, or a wrapped -1. +pub fn write(message: []const u8) usize { + return writeRecord(.raw, message); +} + +/// Emit one leveled record into the tagged kernel log ring. The kernel stamps +/// pid/name/sequence/timestamp; the payload should be a single line. +pub fn writeRecord(level: KlogLevel, message: []const u8) usize { + return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level)); +} + +/// Copy framed records out of the tagged kernel log ring starting at stream `offset` into +/// `out`. Returns the byte count (0 = caught up), or null when `offset` fell behind the +/// ring's tail or lies past its head (re-sync via `klogStatus`). +pub fn klogRead(offset: u64, out: []u8) ?usize { + const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len); + if (@as(isize, @bitCast(r)) < 0) return null; + return r; +} + +/// The log ring's live cursors (oldest retained offset, end of stream, next sequence) +/// plus the wall-clock time of boot — how a log reader starts, detects loss, and names a +/// per-boot log directory. +pub fn klogStatus() ?KlogStatus { + var status: KlogStatus = undefined; + if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null; + return status; +} + +fn levelOf(comptime level: std.log.Level) KlogLevel { + return switch (level) { + .err => .err, + .warn => .warn, + .info => .info, + .debug => .debug, + }; +} + +pub fn logFn( + comptime level: std.log.Level, + comptime scope: @EnumLiteral(), + comptime format: []const u8, + args: anytype, +) void { + // One record = one line = at most klog_maximum_message bytes of payload. + // On overflow keep what fits and end with "~" so the record is still a + // whole line (the kernel would split an embedded rest anyway). + var buffer: [256]u8 = undefined; + const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") "; + const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: { + buffer[buffer.len - 1] = '~'; + break :truncated buffer[0..]; + }; + _ = writeRecord(levelOf(level), line); +} + +/// The std.Options the root shim installs unless the program overrides it. +/// Debug level: filtering is the log *reader's* job here — the ring is cheap, +/// serial is a dev convenience, and the logger service keeps everything. +pub const default_options: std.Options = .{ + .log_level = .debug, + .logFn = logFn, +}; diff --git a/library/kernel/dma.zig b/library/kernel/memory/dma.zig similarity index 98% rename from library/kernel/dma.zig rename to library/kernel/memory/dma.zig index 4f3b96d..5d22078 100644 --- a/library/kernel/dma.zig +++ b/library/kernel/memory/dma.zig @@ -5,7 +5,7 @@ //! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md. const abi = @import("abi"); -const sc = @import("system-call.zig"); +const sc = @import("system-call"); /// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are /// opt-in for specific hardware — see `abi`. diff --git a/library/kernel/heap.zig b/library/kernel/memory/heap.zig similarity index 96% rename from library/kernel/heap.zig rename to library/kernel/memory/heap.zig index 0e525b7..3b76048 100644 --- a/library/kernel/heap.zig +++ b/library/kernel/memory/heap.zig @@ -19,8 +19,8 @@ const std = @import("std"); const builtin = @import("builtin"); const abi = @import("abi"); -const system_calls = @import("system.zig"); -const Mutex = @import("thread.zig").Thread.Mutex; +const sc = @import("system-call"); +const Mutex = @import("thread").Thread.Mutex; const page_size = abi.page_size; @@ -63,8 +63,8 @@ fn payloadOf(block: *Block) [*]u8 { /// grants usually are adjacent). Returns false if the kernel is out of memory. fn grow(minimum_bytes: usize) bool { const bytes = alignUp(@max(minimum_bytes, chunk), page_size); - const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE); - if (system_calls.mmapFailed(ret)) return false; + const ret = sc.systemCall2(.mmap, bytes, abi.prot_read | abi.prot_write); + if (ret > ~@as(usize, 0) - 4095) return false; // a wrapped -errno lands in the top page const block: *Block = @ptrFromInt(ret); block.size = bytes; diff --git a/library/kernel/memory/memory.zig b/library/kernel/memory/memory.zig new file mode 100644 index 0000000..a8f0db9 --- /dev/null +++ b/library/kernel/memory/memory.zig @@ -0,0 +1,48 @@ +//! library/kernel/memory — the process's memory interface: the heap allocator, DMA-capable +//! buffers, shared-memory regions, and the raw `mmap` grant they all sit on. One flat module +//! (formerly runtime.heap / runtime.dma / runtime.shared_memory, plus the `mmap` wrappers that +//! lived in the system.zig dumping ground). Its private files are heap.zig, dma.zig, and +//! shared-memory.zig — imported only here, so the heap's state and C symbols exist once. + +const abi = @import("abi"); +const sc = @import("system-call"); +const heap = @import("heap.zig"); +const dma = @import("dma.zig"); +const shared = @import("shared-memory.zig"); + +// --- the heap: a std.mem.Allocator over a first-fit free list (C malloc/free are also +// exported from heap.zig, compiled once here) --- +pub const allocator = heap.allocator; + +// --- the raw grant every allocation sits on --- +pub const PROT_READ: usize = abi.prot_read; +pub const PROT_WRITE: usize = abi.prot_write; +pub const PROT_EXEC: usize = abi.prot_exec; + +/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable memory and +/// return the base virtual address. On failure returns a value in the top page (`mmapFailed`). +pub fn mmap(len: usize, prot: usize) usize { + return sc.systemCall2(.mmap, len, prot); +} +/// Release a range previously handed out by `mmap`. +pub fn munmap(base: usize, len: usize) usize { + return sc.systemCall2(.munmap, base, len); +} +/// Whether an `mmap` return value is an error (a wrapped -errno lands in the top page). +pub inline fn mmapFailed(ret: usize) bool { + return ret > ~@as(usize, 0) - 4095; +} + +// --- DMA-capable buffers: physically contiguous, pinned, uncacheable, physical address known --- +pub const DmaRegion = dma.Region; +pub const dma_coherent = dma.coherent; +pub const dma_write_combining = dma.write_combining; +pub const dma_below_4g = dma.below_4g; +pub const dmaAlloc = dma.alloc; +pub const dmaFree = dma.free; + +// --- shared-memory regions: a capability handed to another process over an ipc_call send_cap --- +pub const SharedRegion = shared.Region; +pub const sharedCreate = shared.create; +pub const sharedMap = shared.map; +pub const sharedPhysical = shared.physical; diff --git a/library/kernel/shared-memory.zig b/library/kernel/memory/shared-memory.zig similarity index 97% rename from library/kernel/shared-memory.zig rename to library/kernel/memory/shared-memory.zig index 9251e01..8ca400b 100644 --- a/library/kernel/shared-memory.zig +++ b/library/kernel/memory/shared-memory.zig @@ -6,8 +6,8 @@ //! generalization of capability passing from endpoints to memory objects. const abi = @import("abi"); -const sc = @import("system-call.zig"); -const ipc = @import("ipc.zig"); +const sc = @import("system-call"); +const ipc = @import("ipc"); inline fn failed(r: usize) bool { return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page diff --git a/library/kernel/process.zig b/library/kernel/process.zig index 7f59770..edffd6c 100644 --- a/library/kernel/process.zig +++ b/library/kernel/process.zig @@ -7,9 +7,9 @@ const std = @import("std"); const abi = @import("abi"); -const sc = @import("system-call.zig"); -const ipc = @import("ipc.zig"); -const system = @import("system.zig"); +const sc = @import("system-call"); +const ipc = @import("ipc"); +const time = @import("time"); /// Everything a program receives at entry. Passed to /// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep @@ -112,14 +112,14 @@ pub fn sendSignal(id: u32, signal: Signal) bool { /// (arm `system.timerOnce`, keep serving) instead of calling this. pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void { _ = sendSignal(id, .terminate); - _ = system.timerOnce(exit_endpoint, deadline_ms); + _ = time.timerOnce(exit_endpoint, deadline_ms); var receive: [8]u8 = undefined; while (true) { const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null); if (got.isChildExit() and got.childProcessId() == id) return; if (got.isTimer()) break; // the deadline passed first — escalate } - _ = system.kill(id); + _ = kill(id); while (true) { const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null); if (got.isChildExit() and got.childProcessId() == id) return; @@ -135,3 +135,78 @@ pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void { pub fn subscribeExits(endpoint: usize) bool { return sc.systemCall1(.process_subscribe, endpoint) == 0; } + +// --- raw process syscalls, formerly in the system.zig dumping ground --- + +/// One `processes` entry — re-exported from the shared ABI so a program can declare its +/// snapshot buffer without importing `abi` itself. +pub const ProcessDescriptor = abi.ProcessDescriptor; + +/// Give up the rest of this quantum. +pub fn yield() void { + _ = sc.systemCall0(.yield); +} + +/// End the process. Never returns. +pub fn exit(code: usize) noreturn { + _ = sc.systemCall1(.exit, code); + unreachable; // the kernel never returns from exit +} + +/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3 process, +/// returning the child's process id (or null). argv[0] is `name`, and the caller becomes +/// its **supervisor** — the only process allowed to `kill` it. +pub fn spawn(name: []const u8) ?u32 { + return spawnSupervised(name, &.{}, null); +} + +/// Like `spawn`, but hands the child argv[1..] (argv[0] is still `name`). +pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 { + return spawnSupervised(name, arguments, null); +} + +/// The full spawn: argv[1..] for the child, and an optional endpoint the kernel notifies +/// when the child ends (any way — clean exit, fault, or `kill`), delivered via +/// `ipc.replyWait` as a child-exit badge (`ipc.Received.isChildExit`/`childProcessId`), so +/// one endpoint can supervise many children. Returns the child's process id, or null. +pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 { + var blob: [256]u8 = undefined; + var len: usize = 0; + for (arguments, 0..) |argument, i| { + if (i != 0) { + if (len >= blob.len) return null; + blob[len] = 0; + len += 1; + } + if (len + argument.len > blob.len) return null; + @memcpy(blob[len..][0..argument.len], argument); + len += argument.len; + } + const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap); + if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno + return @intCast(r); +} + +/// Snapshot the process table into `out` and return the total number of live processes +/// (which may exceed `out.len`; call again with a larger buffer). Kernel tasks are +/// included, with an empty name. The primitive `ps` is built on. +pub fn processes(out: []ProcessDescriptor) usize { + return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len); +} + +/// Whether a process spawned under `name` (its argv[0]) is currently alive. +pub fn isProcessRunning(name: []const u8) bool { + var table: [32]ProcessDescriptor = undefined; + const total = processes(&table); + for (table[0..@min(total, table.len)]) |descriptor| { + if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true; + } + return false; +} + +/// End process `id`. Only its supervisor — the process that spawned it — may; anyone else +/// gets false, as does a stale or unknown id. Delivery is prompt but asynchronous, like a +/// signal. True means the kill is accepted and irrevocable. +pub fn kill(id: u32) bool { + return sc.systemCall1(.process_kill, id) == 0; +} diff --git a/library/kernel/runtime.zig b/library/kernel/runtime.zig index b8336cc..c6990bf 100644 --- a/library/kernel/runtime.zig +++ b/library/kernel/runtime.zig @@ -1,74 +1,51 @@ -//! danos user-space runtime library — a nascent libc. Every user binary (init, -//! and later the VFS server + device drivers) imports this as `@import("runtime")`: -//! system_call wrappers, the C-convention heap, IPC helpers, and the process start -//! shim. It is compiled into each binary (inheriting its `.large` code model and -//! freestanding target), so all user programs share one implementation. +//! runtime.zig — a **compatibility shim** for the runtime split (reorg C1–C5). //! -//! A user binary only defines a `pub fn main() void` or -//! `pub fn main(init: runtime.process.Init) void` (arguments arrive via `init`). -//! The panic handler and the `_start` entry pull live in the shared compilation -//! root, library/runtime/root.zig, which build.zig wires around every program — -//! nothing to declare per source file. +//! `library/runtime` became `library/kernel`, and the one giant `runtime` module is being +//! split into directly-importable concern modules (`ipc`, `memory`, `process`, `time`, +//! `logging`, `file-system`, `service`, `thread`, `start`, plus the device/service clients). +//! This file re-exports those modules under the old `runtime.*` names so the ~38 consumers +//! keep compiling until each is migrated to direct imports. Deleted in step C5. -pub const system = @import("system.zig"); -pub const log = @import("log.zig"); -/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls -/// — an `Instant`/`Duration` front door, no time service (docs/timers.md). -pub const time = @import("time.zig"); -pub const heap = @import("heap.zig"); -pub const ipc = @import("ipc.zig"); -pub const start = @import("start.zig"); - -/// Client for talking to the device manager (the hello handshake a supervised -/// driver owes at startup). See library/runtime/device-manager.zig. The wire -/// protocol itself is the library/protocol/device-manager module, imported -/// directly by drivers and services that speak it. -pub const device_manager = @import("device-manager.zig"); -/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input -/// service. See library/runtime/input.zig and system/services/input/. -pub const input = @import("input.zig"); -/// POSIX-style file API: open/read/write/lseek/stat/close. -/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd. -/// Device access for drivers: enumerate/claim/mmioMap. -pub const device = @import("device.zig"); -/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers. -pub const dma = @import("dma.zig"); - -/// Shared cacheable memory: create a region + capability, pass the capability to another -/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shared-memory.zig -/// and docs/display-v2.md. -pub const shared_memory = @import("shared-memory.zig"); - -// The USB class-driver client moved to its domain home, library/device/usb (module -// "usb"): it is bus-family logic, not core runtime, and re-exporting it here compiled it -// into every binary. USB class drivers import it directly with @import("usb"). - -/// Block-device client: read/write a block device (a USB stick, via -/// usb-storage). See library/runtime/block.zig. -pub const block = @import("block.zig"); - -/// Display-service client: query the mode, and (from D3) create layers, draw, and -/// present frames. See library/runtime/display.zig and system/services/display/. -pub const display = @import("display.zig"); - -/// The danos-native file API (open/read/write/list over the user-space VFS) — the -/// layer danos programs use directly, and where the operations that later become -/// `std.os.danos` are staged. See docs/zig-self-hosting.md. -pub const fs = @import("fs.zig"); - -/// Re-exported so the root shim (root.zig) can install it as the panic handler. +pub const system = @import("system"); // the system.zig compatibility shim +pub const ipc = @import("ipc"); +pub const memory = @import("memory"); +pub const allocator = memory.allocator; +pub const process = @import("process"); +pub const service = @import("service"); +pub const Thread = @import("thread").Thread; +pub const time = @import("time"); +pub const log = @import("logging"); +pub const logging = @import("logging"); +pub const fs = @import("file-system"); +pub const start = @import("start"); pub const panic = start.panic; -/// Process entry types: the `Init` handed to `main`, and its `Arguments`. -pub const process = @import("process.zig"); +// Device / service clients (relocated to library/device and library/client in C3/C4). +pub const device = @import("device"); +pub const device_manager = @import("device-manager"); +pub const input = @import("input"); +pub const block = @import("block"); +pub const display = @import("display"); -/// Threads: `runtime.Thread`, std.Thread-shaped, over the private thread ABI -/// (docs/threading.md). A binary must be built multi-threaded to spawn. -pub const Thread = @import("thread.zig").Thread; +/// The heap as a namespace (`runtime.heap.allocator()`), plus `runtime.allocator`. +pub const heap = struct { + pub const allocator = memory.allocator; +}; -/// The service harness: one replyWait loop folding requests, signals, and -/// notifications into callbacks (docs/process-lifecycle.md). -pub const service = @import("service.zig"); +/// `runtime.dma.*` mapped onto the flat `memory` API (memory groups heap+dma+shared-memory). +pub const dma = struct { + pub const Region = memory.DmaRegion; + pub const coherent = memory.dma_coherent; + pub const write_combining = memory.dma_write_combining; + pub const below_4g = memory.dma_below_4g; + pub const alloc = memory.dmaAlloc; + pub const free = memory.dmaFree; +}; -/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code. -pub const allocator = heap.allocator; +/// `runtime.shared_memory.*` mapped onto the flat `memory` API. +pub const shared_memory = struct { + pub const Region = memory.SharedRegion; + pub const create = memory.sharedCreate; + pub const map = memory.sharedMap; + pub const physical = memory.sharedPhysical; +}; diff --git a/library/kernel/service.zig b/library/kernel/service.zig index 8926754..0749733 100644 --- a/library/kernel/service.zig +++ b/library/kernel/service.zig @@ -13,8 +13,8 @@ //! is the diagnosis (see docs/ipc.md). const abi = @import("abi"); -const ipc = @import("ipc.zig"); -const process = @import("process.zig"); +const ipc = @import("ipc"); +const process = @import("process"); pub const Callbacks = struct { /// Called once with the service's endpoint before the loop starts — the diff --git a/library/kernel/start.zig b/library/kernel/start.zig index aae6848..0ee968b 100644 --- a/library/kernel/start.zig +++ b/library/kernel/start.zig @@ -4,8 +4,8 @@ //! the whole runtime is linked in. const std = @import("std"); -const system = @import("system.zig"); -const process = @import("process.zig"); +const logging = @import("logging"); +const process = @import("process"); /// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V /// process-entry block it built: argc, argv pointers, NULL, envp terminator, the @@ -31,7 +31,7 @@ export fn rt_start(stack: [*]const u64) callconv(.c) noreturn { .count = stack[0], .vector = @ptrCast(stack + 1), } }; - system.exit(callMain(init)); + process.exit(callMain(init)); } /// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig: @@ -68,7 +68,7 @@ fn callMain(init: process.Init) u8 { const payload = @call(.auto, root.main, call_arguments) catch |err| { var buffer: [128]u8 = undefined; const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n"; - _ = system.write(line); + _ = logging.write(line); return 1; // distinct from panic's 127 }; if (@TypeOf(payload) == void) return 0; @@ -82,6 +82,6 @@ fn callMain(init: process.Init) u8 { /// No runtime to unwind into — report a panic as a nonzero exit code. pub const panic = std.debug.FullPanic(struct { fn panic(_: []const u8, _: ?usize) noreturn { - system.exit(127); + process.exit(127); } }.panic); diff --git a/library/kernel/system.zig b/library/kernel/system.zig index ffaf7fb..a1ca604 100644 --- a/library/kernel/system.zig +++ b/library/kernel/system.zig @@ -1,266 +1,66 @@ -//! Typed system_call surface for user space — thin wrappers over the raw `system_call` -//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single -//! source of truth shared with the kernel dispatcher. +//! system.zig — a **compatibility shim**, not the real home of anything anymore. +//! +//! The runtime's syscall surface used to be dumped here in one file. It has been split by +//! concern into `time` / `logging` / `process` / `file-system` / `memory`. This re-exports +//! the old flat `system.*` names from those homes so consumers that still write +//! `runtime.system.write` (etc.) keep compiling until they migrate to the concern modules. +//! Deleted once nothing references `runtime.system` (reorg step C5). -const std = @import("std"); -const abi = @import("abi"); -const sc = @import("system-call.zig"); +const memory = @import("memory"); +const process = @import("process"); +const time = @import("time"); +const logging = @import("logging"); +const file_system = @import("file-system"); -/// `mmap` protection flags (matching the usual C bit values). Grants are always -/// readable+writable today; the kernel does not yet honour finer prot. -pub const PROT_READ: usize = abi.prot_read; -pub const PROT_WRITE: usize = abi.prot_write; -pub const PROT_EXEC: usize = abi.prot_exec; +// memory +pub const PROT_READ = memory.PROT_READ; +pub const PROT_WRITE = memory.PROT_WRITE; +pub const PROT_EXEC = memory.PROT_EXEC; +pub const mmap = memory.mmap; +pub const munmap = memory.munmap; +pub const mmapFailed = memory.mmapFailed; -/// One `processes` entry — re-exported from the shared ABI so a user program can -/// declare its snapshot buffer without importing `abi` itself. -pub const ProcessDescriptor = abi.ProcessDescriptor; +// process +pub const ProcessDescriptor = process.ProcessDescriptor; +pub const yield = process.yield; +pub const exit = process.exit; +pub const spawn = process.spawn; +pub const spawnWithArguments = process.spawnWithArguments; +pub const spawnSupervised = process.spawnSupervised; +pub const processes = process.processes; +pub const isProcessRunning = process.isProcessRunning; +pub const kill = process.kill; -/// Give up the rest of this quantum. -pub fn yield() void { - _ = sc.systemCall0(.yield); -} +// time +pub const clock = time.clock; +pub const wallClock = time.wallClock; +pub const sleep = time.sleepMillis; +pub const timerOnce = time.timerOnce; -/// The tagged-log level of a record — re-exported so runtime.log and the logger -/// service don't import `abi` themselves. -pub const KlogLevel = abi.KlogLevel; -pub const KlogStatus = abi.KlogStatus; -pub const KlogRecordHeader = abi.KlogRecordHeader; -pub const klog_record_header_size = abi.klog_record_header_size; -pub const klog_record_alignment = abi.klog_record_alignment; -pub const klog_record_magic = abi.klog_record_magic; -pub const klog_flag_truncated = abi.klog_flag_truncated; -pub const klog_maximum_message = abi.klog_maximum_message; -pub const maximum_process_name = abi.maximum_process_name; -pub const FileAttributes = abi.FileAttributes; -pub const DirectoryEntryHeader = abi.DirectoryEntryHeader; -pub const file_kind_regular = abi.file_kind_regular; -pub const file_kind_directory = abi.file_kind_directory; +// logging +pub const write = logging.write; +pub const writeRecord = logging.writeRecord; +pub const klogRead = logging.klogRead; +pub const klogStatus = logging.klogStatus; +pub const KlogLevel = logging.KlogLevel; +pub const KlogStatus = logging.KlogStatus; +pub const KlogRecordHeader = logging.KlogRecordHeader; +pub const klog_record_header_size = logging.klog_record_header_size; +pub const klog_record_alignment = logging.klog_record_alignment; +pub const klog_record_magic = logging.klog_record_magic; +pub const klog_flag_truncated = logging.klog_flag_truncated; +pub const klog_maximum_message = logging.klog_maximum_message; +pub const maximum_process_name = logging.maximum_process_name; -/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary -/// output goes through std.log -> writeRecord). The kernel stamps the record -/// with this process's id and name. Returns the byte count, or a wrapped -1. -pub fn write(message: []const u8) usize { - return writeRecord(.raw, message); -} - -/// Emit one leveled record into the tagged kernel log ring. The kernel stamps -/// pid/name/sequence/timestamp; the payload should be a single line (embedded -/// newlines split into further records). -pub fn writeRecord(level: KlogLevel, message: []const u8) usize { - return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level)); -} - -/// Block the caller for `ms` milliseconds. -pub fn sleep(ms: usize) void { - _ = sc.systemCall1(.sleep, ms); -} - -/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer -/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of -/// docs/process-lifecycle.md — a service arms a deadline and keeps serving, -/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines, -/// and restart backoff are built from. -pub fn timerOnce(endpoint: usize, ms: u64) bool { - return sc.systemCall2(.timer_bind, endpoint, ms) == 0; -} - -/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It -/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that -/// is a user-space service layered on top). Deadline pattern for a bounded poll loop: -/// -/// const deadline = clock() + timeout_ns; -/// while (clock() < deadline) { ... } -pub fn clock() u64 { - return @intCast(sc.systemCall0(.clock)); -} - -/// Wall-clock time in Unix epoch seconds (UTC) — the real date/time, from the RTC. -/// Unlike `clock` (monotonic since boot), this tracks calendar time, so it is what a -/// filesystem stamps as a file's modification time. Formatting it into a calendar -/// date/timezone is user-space policy layered on top. -pub fn wallClock() u64 { - return @intCast(sc.systemCall0(.wall_clock)); -} - -/// Copy bytes out of the tagged kernel log ring — framed records of everything -/// every process (and the kernel) has emitted — starting at stream offset -/// `offset`, into `out`. Returns the byte count (0 = caught up), or null when -/// `offset` fell behind the ring's tail (those records were overwritten) or -/// lies past its head; re-sync via `klogStatus`. A reader parses -/// [KlogRecordHeader][name][message] frames (8-byte aligned) from the bytes. -pub fn klogRead(offset: u64, out: []u8) ?usize { - const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len); - if (@as(isize, @bitCast(r)) < 0) return null; - return r; -} - -/// The log ring's live cursors (oldest retained offset, end of stream, next -/// sequence number) plus the wall-clock time of boot — how a log reader starts, -/// detects loss, and names a per-boot log directory. -pub fn klogStatus() ?KlogStatus { - var status: KlogStatus = undefined; - if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null; - return status; -} - -/// Where fs_resolve routed a path: served by the kernel (a permanent node -/// token for fs_node) or by a userspace filesystem backend (an endpoint handle -/// plus the rewritten mount-relative path, returned in the caller's buffer). -pub const FsRoute = union(enum) { - kernel: u64, - backend: struct { handle: usize, path_len: usize }, -}; - -/// Route `path` through the kernel VFS. For a backend route the rewritten -/// mount-relative path lands in `out` (behind a kernel-written length prefix, -/// already stripped here: out[0..path_len] is the path). -pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute { - var rax: usize = undefined; - var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle - asm volatile ("syscall" - : [rax] "={rax}" (rax), - [rdx] "+{rdx}" (rdx), - : [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)), - [a0] "{rdi}" (@intFromPtr(path.ptr)), - [a1] "{rsi}" (path.len), - [a3] "{r10}" (@intFromPtr(out.ptr)), - [a4] "{r8}" (out.len), - : .{ .rcx = true, .r11 = true, .memory = true }); - if (@as(isize, @bitCast(rax)) < 0) return null; - if (rax == abi.fs_route_kernel) return .{ .kernel = rdx }; - if (rax != abi.fs_route_backend) return null; - const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8); - if (path_len + 2 > out.len) return null; - std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]); - return .{ .backend = .{ .handle = rdx, .path_len = path_len } }; -} - -/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read). -pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize { - const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len); - if (@as(isize, @bitCast(r)) < 0) return null; - return r; -} - -/// A kernel-served node's metadata (fs_node status). -pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes { - var attributes: abi.FileAttributes = undefined; - const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attributes), @sizeOf(abi.FileAttributes)); - if (@as(isize, @bitCast(r)) < 0) return null; - return attributes; -} - -/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills -/// `out` with [DirectoryEntryHeader][name]; returns total bytes (0 = end). -pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize { - const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len); - if (@as(isize, @bitCast(r)) < 0) return null; - return r; -} - -/// Mount a userspace filesystem's endpoint at `prefix`, with an optional -/// backend-side `rewrite` prefix ("" = none). Possession of the endpoint -/// handle is the capability. -pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool { - return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0; -} - -pub fn fsUnmount(prefix: []const u8) bool { - return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0; -} - -/// End the process. Never returns. -pub fn exit(code: usize) noreturn { - _ = sc.systemCall1(.exit, code); - unreachable; // the kernel never returns from exit -} - -/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3 -/// process, returning the child's process id (or null on failure). The child's -/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process -/// allowed to `kill` it. This is how a supervisor (the device manager) launches a -/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the -/// POSIX layer later). -pub fn spawn(name: []const u8) ?u32 { - return spawnSupervised(name, &.{}, null); -} - -/// Like `spawn`, but hands the child command-line arguments: they arrive as -/// argv[1..] on its System V entry stack (argv[0] is still `name`). -pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 { - return spawnSupervised(name, arguments, null); -} - -/// The full spawn: command-line arguments for the child, and an optional endpoint -/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends -/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via -/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in -/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can -/// supervise many children. Arguments are marshalled to the kernel as one -/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the -/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or -/// null on failure. -pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 { - var blob: [256]u8 = undefined; - var len: usize = 0; - for (arguments, 0..) |argument, i| { - if (i != 0) { - if (len >= blob.len) return null; - blob[len] = 0; - len += 1; - } - if (len + argument.len > blob.len) return null; - @memcpy(blob[len..][0..argument.len], argument); - len += argument.len; - } - const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap); - if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno - return @intCast(r); -} - -/// Snapshot the process table into `out` (up to its length) and return the total -/// number of live processes — which may exceed `out.len`; call again with a larger -/// buffer for the full listing. Kernel tasks are included, with an empty name. -/// The primitive `ps` is built on. -pub fn processes(out: []abi.ProcessDescriptor) usize { - return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len); -} - -/// Whether a process spawned under `name` (its argv[0]) is currently alive. -pub fn isProcessRunning(name: []const u8) bool { - var table: [32]ProcessDescriptor = undefined; - const total = processes(&table); - for (table[0..@min(total, table.len)]) |descriptor| { - if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true; - } - return false; -} - -/// End process `id`. Only its supervisor — the process that spawned it — may; -/// anyone else gets false, as does a stale or unknown id (ids are never reused). -/// Delivery is prompt but asynchronous, like a signal: a target caught running on -/// another core dies at its next system call or timer tick. True means the kill -/// is accepted and irrevocable; the exit notification (if an endpoint was given -/// at spawn) confirms completion. -pub fn kill(id: u32) bool { - return sc.systemCall1(.process_kill, id) == 0; -} - -/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable -/// memory and return the base virtual address. On failure returns a value in the -/// top page (see `mmapFailed`). The user heap grows through this call. -pub fn mmap(len: usize, prot: usize) usize { - return sc.systemCall2(.mmap, len, prot); -} - -/// Release a range previously handed out by `mmap`. -pub fn munmap(base: usize, len: usize) usize { - return sc.systemCall2(.munmap, base, len); -} - -/// Whether an `mmap` return value is an error (the kernel returns a wrapped -/// -errno, which lands in the top page — no real grant base is ever that high). -pub inline fn mmapFailed(ret: usize) bool { - return ret > ~@as(usize, 0) - 4095; -} +// file-system +pub const FsRoute = file_system.FsRoute; +pub const fsResolve = file_system.fsResolve; +pub const fsNodeRead = file_system.fsNodeRead; +pub const fsNodeStatus = file_system.fsNodeStatus; +pub const fsNodeReaddir = file_system.fsNodeReaddir; +pub const fsMount = file_system.fsMount; +pub const fsUnmount = file_system.fsUnmount; +pub const FileAttributes = file_system.FileAttributes; +pub const DirectoryEntryHeader = file_system.DirectoryEntryHeader; +pub const file_kind_regular = file_system.file_kind_regular; +pub const file_kind_directory = file_system.file_kind_directory; diff --git a/library/kernel/thread.zig b/library/kernel/thread.zig index 5d2da4a..5e83583 100644 --- a/library/kernel/thread.zig +++ b/library/kernel/thread.zig @@ -13,8 +13,19 @@ const std = @import("std"); const builtin = @import("builtin"); const abi = @import("abi"); -const sc = @import("system-call.zig"); -const system = @import("system.zig"); +const sc = @import("system-call"); + +// A thread allocates its own stack straight from the mmap syscall (not through the +// `memory` module) so `memory`'s heap can depend on this module's Mutex without a cycle. +inline fn mmapStack(len: usize) usize { + return sc.systemCall2(.mmap, len, abi.prot_read | abi.prot_write); +} +inline fn mmapFailed(ret: usize) bool { + return ret > ~@as(usize, 0) - 4095; +} +inline fn munmapStack(base: usize, len: usize) void { + _ = sc.systemCall2(.munmap, base, len); +} /// True in a real danos binary; false when this module is compiled for host unit tests. /// The `Futex` seam and the test blocks below branch on it so the lock/condvar state @@ -66,8 +77,8 @@ pub const Thread = struct { } }; - const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE); - if (system.mmapFailed(base)) return error.SystemResources; + const base = mmapStack(config.stack_size); + if (mmapFailed(base)) return error.SystemResources; // Top of the thread's own stack, downward: the closure, then a small per-thread TLS // block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is @@ -88,7 +99,7 @@ pub const Thread = struct { const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr); if (threadSpawnFailed(tid)) { - _ = system.munmap(base, config.stack_size); + munmapStack(base, config.stack_size); return error.SystemResources; } return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size }; @@ -99,7 +110,7 @@ pub const Thread = struct { /// child-exit notification on it is this thread's. pub fn join(self: Thread) void { _ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited - _ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack + munmapStack(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack } /// Relinquish the right to join: never wait for or reclaim this thread. Its stack is diff --git a/library/kernel/time.zig b/library/kernel/time.zig index 9ae4908..ec44581 100644 --- a/library/kernel/time.zig +++ b/library/kernel/time.zig @@ -11,7 +11,33 @@ //! CLOCK_REALTIME) layered on top later. const std = @import("std"); -const system = @import("system.zig"); +const sc = @import("system-call"); + +// --- raw syscall wrappers, formerly in the system.zig dumping ground --- + +/// Monotonic nanoseconds since boot — the raw reading; `now()` wraps it in an `Instant`. +/// Never runs backward. Not wall-clock time (see `wallClock`). +pub fn clock() u64 { + return @intCast(sc.systemCall0(.clock)); +} + +/// Wall-clock time in Unix epoch seconds (UTC), from the RTC — the real date/time, what a +/// filesystem stamps as an mtime. Unlike `clock` (monotonic since boot), this is calendar time. +pub fn wallClock() u64 { + return @intCast(sc.systemCall0(.wall_clock)); +} + +/// Block the caller for `ms` milliseconds — the raw, coarse, allocation-free form. +pub fn sleepMillis(ms: u64) void { + _ = sc.systemCall1(.sleep, ms); +} + +/// Arm a one-shot timer: after `ms` the kernel posts a timer notification +/// (`ipc.Received.isTimer`) to `endpoint` (a handle from `ipc.createIpcEndpoint`). Unlike +/// `sleep`, does not block — a service keeps serving IPC while the deadline is pending. +pub fn timerOnce(endpoint: usize, ms: u64) bool { + return sc.systemCall2(.timer_bind, endpoint, ms) == 0; +} const nanos_per_micro: u64 = 1_000; const nanos_per_milli: u64 = 1_000_000; @@ -90,32 +116,27 @@ pub const Instant = struct { /// The current monotonic time. pub fn now() Instant { - return .{ .ns = system.clock() }; + return .{ .ns = clock() }; } /// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that /// want a plain integer instead of an `Instant`. pub fn monotonicNanos() u64 { - return system.clock(); + return clock(); } /// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is /// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as /// "unavailable" instead of assuming the clock advances. pub fn available() bool { - return system.clock() != 0; + return clock() != 0; } /// Block the caller for at least `d`, rounded up to the kernel's millisecond /// granularity. For sub-millisecond precision the scheduler cannot express, use -/// `spin`. +/// `spin`. (The raw millisecond form is `sleepMillis`.) pub fn sleep(d: Duration) void { - system.sleep(d.ceilMillis()); -} - -/// Block the caller for `ms` milliseconds — the coarse, allocation-free form. -pub fn sleepMillis(ms: u64) void { - system.sleep(ms); + sleepMillis(d.ceilMillis()); } /// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU @@ -126,13 +147,10 @@ pub fn spin(d: Duration) void { while (!deadline.reached()) {} } -/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`): -/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there. -/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same -/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns -/// false if the timer could not be armed. See `system.timerOnce`. +/// The ergonomic `Duration` form of `timerOnce`: arm a one-shot timer against `endpoint` +/// for `d` (rounded up to milliseconds). Returns false if the timer could not be armed. pub fn after(endpoint: usize, d: Duration) bool { - return system.timerOnce(endpoint, d.ceilMillis()); + return timerOnce(endpoint, d.ceilMillis()); } test "Duration unit conversions round toward zero" {