reorg: rename library/runtime -> library/kernel (groundwork)
First step of splitting the runtime dumping ground. Pure directory rename (git mv library/runtime library/kernel) + the six build.zig path references repointed. The module is still named "runtime" for now; the next commits split it into concern modules (ipc, memory, process, time, logging, file-system, ...), dissolve system.zig, and delete the runtime aggregator. The library/kernel name follows the kernel32 model: it is the userspace library that wraps the private kernel ABI, distinct from system/kernel/ (the kernel). zig build green.
This commit is contained in:
@@ -0,0 +1,79 @@
|
||||
//! Block-device client: the helper a filesystem uses to read and write a block
|
||||
//! device (a USB stick, via usb-storage) without hand-rolling the block-protocol
|
||||
//! IPC. Layered over `ipc` and the shared `block-protocol` wire format, like
|
||||
//! `runtime.usb` over the transfer protocol.
|
||||
//!
|
||||
//! Transfers name a caller-owned DMA buffer by physical address (from
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
pub const Device = struct {
|
||||
endpoint: ipc.Handle,
|
||||
|
||||
/// The device's block size and total block count.
|
||||
pub fn geometry(self: Device) ?Geometry {
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (n < block_protocol.reply_size) return null;
|
||||
const result = std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]);
|
||||
if (result.status != 0) return null;
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.read, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Write `count` blocks starting at `lba` from the DMA buffer at `physical`.
|
||||
pub fn write(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.write, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||
pub fn flush(self: Device) bool {
|
||||
return self.transfer(.flush, 0, 0, 0);
|
||||
}
|
||||
|
||||
fn transfer(self: Device, operation: block_protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (n < block_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// One lookup attempt, no waiting — for a server that retries on its own
|
||||
/// timer (the fat service) instead of blocking its harness in here.
|
||||
pub fn tryOpen() ?Device {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Look up the block device, retrying generously while the USB storage chain
|
||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||
pub fn open() ?Device {
|
||||
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||
// tens of seconds under emulation.
|
||||
var attempts: usize = 0;
|
||||
// 30 s covers the slowest observed healthy chain (a flaky QEMU enumeration
|
||||
// completed at ~24 s); a machine whose stick genuinely failed setup should
|
||||
// not sit a further minute pretending otherwise.
|
||||
while (attempts < 600) : (attempts += 1) {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
//! Client side of the device-manager protocol (docs/device-manager.md): the
|
||||
//! calls a supervised driver makes *to* the manager, as opposed to
|
||||
//! `device-manager-protocol.zig`, which is the wire contract both sides share.
|
||||
//!
|
||||
//! Today this is just the hello handshake every spawned driver owes. The
|
||||
//! manager arms a hello deadline when it spawns a driver
|
||||
//! (system/services/device-manager/device-manager.zig): a driver that stays
|
||||
//! silent past it is assumed wedged before `main` and stopped, so every driver
|
||||
//! calls `hello` early in its startup. The retry-lookup-call-check this used to
|
||||
//! be — copied byte-for-byte into each bus and class driver — lives here once.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
|
||||
/// What kind of driver is announcing itself (a bus that reports children, or a
|
||||
/// leaf device). Re-exported so callers name it without importing the protocol.
|
||||
pub const Role = device_manager_protocol.Role;
|
||||
|
||||
/// Endpoint lookups while the manager is still registering, and the pause
|
||||
/// between them: 100 x 20 ms = ~2 s, comfortably inside the manager's 3 s hello
|
||||
/// deadline (device-manager.zig `hello_deadline_ms`).
|
||||
const lookup_attempts: u32 = 100;
|
||||
const lookup_pause_ms: u64 = 20;
|
||||
|
||||
/// Say hello to the device manager and return its endpoint, or null if there is
|
||||
/// no manager (best-effort standalone bring-up) or it refused the handshake
|
||||
/// (version mismatch, unknown sender). Bus drivers keep the returned handle to
|
||||
/// report children through; a driver that runs fine unsupervised discards it
|
||||
/// with `_ =`, and a driver that requires supervision bails on null.
|
||||
///
|
||||
/// Logs the outcome itself (attribution is the kernel's, via the process's
|
||||
/// binary path), so callers stay a single line.
|
||||
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
||||
var attempts: u32 = 0;
|
||||
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||
system.sleep(lookup_pause_ms);
|
||||
} else {
|
||||
std.log.info("no device manager to hello", .{});
|
||||
return null;
|
||||
};
|
||||
|
||||
const message = device_manager_protocol.Hello{ .role = @intFromEnum(role), .device_id = device_id };
|
||||
var reply: [device_manager_protocol.reply_size]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&message), &reply) catch {
|
||||
std.log.info("hello call failed", .{});
|
||||
return null;
|
||||
};
|
||||
if (length < device_manager_protocol.reply_size or
|
||||
std.mem.bytesToValue(device_manager_protocol.HelloReply, reply[0..device_manager_protocol.reply_size]).status != 0)
|
||||
{
|
||||
std.log.info("hello refused", .{});
|
||||
return null;
|
||||
}
|
||||
std.log.info("hello acknowledged", .{});
|
||||
return manager;
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||
//! mapping registers or routing an IRQ.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||
pub const DeviceClass = device_abi.DeviceClass;
|
||||
pub const ResourceKind = device_abi.ResourceKind;
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||
}
|
||||
|
||||
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||
pub fn claim(id: u64) bool {
|
||||
return !failed(sc.systemCall1(.device_claim, id));
|
||||
}
|
||||
|
||||
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||
/// `device_id` into this address space; returns the register base virtual address.
|
||||
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||
pub const no_parent = device_abi.no_parent;
|
||||
|
||||
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. Set this on
|
||||
/// descriptors passed to `register` unless the child really is one.
|
||||
pub const no_pci_class = device_abi.no_pci_class;
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||
/// that class of device can `claim` it — that is how a bus hands off a device.
|
||||
///
|
||||
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
||||
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
||||
/// anything else, because a device descriptor is a licence to map physical memory and
|
||||
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||
/// through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
||||
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
||||
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
||||
}
|
||||
|
||||
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
||||
/// status register holds its line asserted) — the kernel left the line masked
|
||||
/// precisely because it could not do that for you. Skip it and the interrupt never
|
||||
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
||||
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
||||
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
||||
}
|
||||
|
||||
/// The Message-Signalled Interrupt address/data a driver programs into its device's
|
||||
/// MSI capability. The device raises the interrupt by writing `data` to `address`.
|
||||
pub const Msi = struct { address: u64, data: u32 };
|
||||
|
||||
/// Set up MSI for a claimed device: the kernel allocates a per-device edge-triggered
|
||||
/// vector, binds it to `endpoint` (delivered like `irqBind`, but with no mask and no
|
||||
/// `irqAck` cycle), and returns the (address, data) to write into the device's MSI
|
||||
/// capability — found by mmio_mapping the device's ECAM config space (resource 0) and
|
||||
/// walking its capability list. Returns null on failure. Two return values (address in
|
||||
/// rax, data in rdx), so a hand-written stub.
|
||||
pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined;
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.msi_bind)),
|
||||
[a0] "{rdi}" (device_id),
|
||||
[a1] "{rsi}" (endpoint),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .address = rax, .data = @intCast(rdx) };
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||
/// access is a syscall, which is fine for the low-rate hardware that needs it. Returns
|
||||
/// null if the capability check fails (device not claimed, wrong resource, out of
|
||||
/// range). A device that decodes no data returns all-ones, which is a valid value, not
|
||||
/// a failure.
|
||||
pub fn ioRead(device_id: u64, resource_index: u64, offset: u64, width: u8) ?u32 {
|
||||
const r = sc.systemCall4(.io_read, device_id, resource_index, offset, width);
|
||||
return if (failed(r)) null else @intCast(r);
|
||||
}
|
||||
|
||||
/// Write `value` (its low `width` bytes, 1/2/4) to a port in a claimed device's
|
||||
/// `io_port` resource, at byte `offset`. Same capability gate as `ioRead`.
|
||||
pub fn ioWrite(device_id: u64, resource_index: u64, offset: u64, width: u8, value: u32) bool {
|
||||
return !failed(sc.systemCall5(.io_write, device_id, resource_index, offset, width, value));
|
||||
}
|
||||
|
||||
/// Find DeviceDescription by hid
|
||||
///
|
||||
/// Utility function for driver development
|
||||
pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const u8) ?DeviceDescriptor {
|
||||
const total = enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (@as([]DeviceDescriptor, buffer[0..n])) |d| {
|
||||
const hid_haystack = d.hid[0..@intCast(d.hid_len)];
|
||||
if (std.mem.eql(u8, hid_haystack, hid_needle)) {
|
||||
return d;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||
//! shape: a cached `.display` lookup with a boot-race retry, then extern-struct request/
|
||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const display_protocol = @import("display-protocol");
|
||||
|
||||
/// The display's current mode, as `info()` reports it.
|
||||
pub const Info = struct {
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32, // bytes per row (may exceed width*4; see docs/framebuffer.md)
|
||||
format: u32, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||
};
|
||||
|
||||
/// The service endpoint, looked up once and cached.
|
||||
var handle: ?ipc.Handle = null;
|
||||
|
||||
/// Look up the display service, retrying while it comes up (a client races its
|
||||
/// registration at boot). Returns the endpoint, or null if it never appears.
|
||||
fn service() ?ipc.Handle {
|
||||
if (handle) |h| return h;
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.display)) |h| {
|
||||
handle = h;
|
||||
return h;
|
||||
}
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Send one request, receive its reply; true on a zero status. `out` receives the reply
|
||||
/// so callers can read `info`/`layer` fields on success.
|
||||
fn transact(request: display_protocol.Request, out: *display_protocol.Reply) bool {
|
||||
const h = service() orelse return false;
|
||||
var req = request;
|
||||
var reply: [display_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&req), &reply) catch return false;
|
||||
if (len < display_protocol.reply_size) return false;
|
||||
out.* = std.mem.bytesToValue(display_protocol.Reply, reply[0..display_protocol.reply_size]);
|
||||
return out.status == 0;
|
||||
}
|
||||
|
||||
/// The display's current mode, or null if the service never came up.
|
||||
pub fn info() ?Info {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
if (!transact(.{ .operation = @intFromEnum(display_protocol.Operation.info) }, &reply)) return null;
|
||||
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||
}
|
||||
|
||||
/// Composite the dirty layers and flush the frame to the screen.
|
||||
pub fn present() bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(display_protocol.Operation.present) }, &reply);
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = display_protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const h = service() orelse return 0;
|
||||
var request = display_protocol.Request{ .operation = @intFromEnum(display_protocol.Operation.get_modes) };
|
||||
var reply: [display_protocol.modes_reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (len < display_protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(display_protocol.ModesReply, reply[0..display_protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, display_protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(display_protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
|
||||
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||
var mode: ?Info = null;
|
||||
|
||||
fn cachedInfo() ?Info {
|
||||
if (mode) |m| return m;
|
||||
const i = info() orelse return null;
|
||||
mode = i;
|
||||
return i;
|
||||
}
|
||||
|
||||
/// The native pixel value for an 8-bit-per-channel colour, in the display's format. A
|
||||
/// client packs colours through this so it never has to know the byte order itself.
|
||||
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||
const format = if (cachedInfo()) |i| i.format else 0;
|
||||
return display_protocol.pack(format, r, g, b);
|
||||
}
|
||||
|
||||
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||
/// into by command. Create with `createLayer`; drawing and moves take effect on the next
|
||||
/// `present`. Coordinates are signed (a layer may sit partly off-screen).
|
||||
pub const Layer = struct {
|
||||
id: u32,
|
||||
|
||||
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(display_protocol.Operation.fill_rect),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
.colour = colour,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||
/// layer at (`x`, `y`). The tile rides inline in the request, so `w*h*4` must fit
|
||||
/// `display_protocol.maximum_payload`.
|
||||
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
var request = display_protocol.Request{
|
||||
.operation = @intFromEnum(display_protocol.Operation.blit_tile),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
};
|
||||
const header = std.mem.asBytes(&request);
|
||||
if (header.len + pixels.len > display_protocol.message_maximum) return false;
|
||||
var buffer: [display_protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(buffer[0..header.len], header);
|
||||
@memcpy(buffer[header.len..][0..pixels.len], pixels);
|
||||
const h_svc = service() orelse return false;
|
||||
var reply: [display_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h_svc, buffer[0 .. header.len + pixels.len], &reply) catch return false;
|
||||
if (len < display_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(display_protocol.Reply, reply[0..display_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Move / restack / show or hide the layer.
|
||||
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(display_protocol.Operation.configure_layer),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.z = z,
|
||||
.visible = if (visible) 1 else 0,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(display_protocol.Operation.damage),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Release the layer and its surface.
|
||||
pub fn destroy(self: Layer) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(display_protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||
}
|
||||
};
|
||||
|
||||
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
if (!transact(.{
|
||||
.operation = @intFromEnum(display_protocol.Operation.create_layer),
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
.z = z,
|
||||
.visible = 1,
|
||||
}, &reply)) return null;
|
||||
return .{ .id = reply.layer };
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
|
||||
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
|
||||
//! physically contiguous, at a physical address the driver knows, uncacheable, and
|
||||
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
|
||||
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||
/// opt-in for specific hardware — see `abi`.
|
||||
pub const coherent: usize = abi.dma_coherent;
|
||||
pub const write_combining: usize = abi.dma_write_combining;
|
||||
pub const below_4g: usize = abi.dma_below_4g;
|
||||
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||
/// to program into the device's descriptor-ring / base registers.
|
||||
pub const Region = struct {
|
||||
virtual: usize,
|
||||
physical: usize,
|
||||
};
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||
/// needs a hand-written stub.
|
||||
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: physical address
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||
[a0] "{rdi}" (len),
|
||||
[a1] "{rsi}" (flags),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .virtual = rax, .physical = rdx };
|
||||
}
|
||||
|
||||
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||
pub fn free(virtual: usize, len: usize) void {
|
||||
_ = sc.systemCall2(.dma_free, virtual, len);
|
||||
}
|
||||
@@ -0,0 +1,354 @@
|
||||
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
|
||||
//! lists files through the kernel VFS root (resolve + redirect), each call
|
||||
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
|
||||
//! danos programs use directly; it is also where the file operations that later
|
||||
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
|
||||
//! the old POSIX `unistd` shim — a compatibility spelling danos does not need yet.
|
||||
//!
|
||||
//! Handles are *values*, not entries in a global descriptor table: a `File` /
|
||||
//! `Directory` owns its VFS node id and (for files) a byte offset. So there is no
|
||||
//! per-process fd limit and no shared table to synchronise — the danos-native
|
||||
//! shape, unlike the POSIX fd model the old shim emulated.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||
/// wire protocol.
|
||||
pub const Kind = vfs_protocol.NodeKind;
|
||||
|
||||
/// A node's metadata (the answer to a status request).
|
||||
pub const Attributes = struct {
|
||||
size: u64,
|
||||
kind: Kind,
|
||||
/// Modification time — Unix epoch seconds, UTC. 0 if the filesystem has none.
|
||||
mtime: u64 = 0,
|
||||
};
|
||||
|
||||
// Map a wire `NodeKind` value to the enum, defaulting anything unrecognised to
|
||||
// `.regular` (the server is trusted, but a value outside the enum would be
|
||||
// illegal to `@enumFromInt` directly).
|
||||
fn kindFromWire(value: u32) Kind {
|
||||
return switch (value) {
|
||||
@intFromEnum(Kind.directory) => .directory,
|
||||
@intFromEnum(Kind.character_device) => .character_device,
|
||||
@intFromEnum(Kind.block_device) => .block_device,
|
||||
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
||||
@intFromEnum(Kind.fifo) => .fifo,
|
||||
@intFromEnum(Kind.socket) => .socket,
|
||||
else => .regular,
|
||||
};
|
||||
}
|
||||
|
||||
/// How to open a path.
|
||||
pub const OpenOptions = struct {
|
||||
/// Create the file if it does not exist.
|
||||
create: bool = false,
|
||||
/// Open a directory node (for listing) rather than a file.
|
||||
directory: bool = false,
|
||||
/// Truncate an existing file to zero length on open (O_TRUNC) — replace its
|
||||
/// contents rather than overwriting in place.
|
||||
truncate: bool = false,
|
||||
|
||||
fn wireFlags(self: OpenOptions) u32 {
|
||||
var f: u32 = 0;
|
||||
if (self.create) f |= vfs_protocol.create;
|
||||
if (self.directory) f |= vfs_protocol.directory;
|
||||
if (self.truncate) f |= vfs_protocol.truncate;
|
||||
return f;
|
||||
}
|
||||
};
|
||||
|
||||
// The route to a path: the kernel resolves (fs_resolve) and either serves the
|
||||
// node itself (the initrd at /system — a permanent token) or redirects us to
|
||||
// the owning filesystem backend's endpoint, to which we speak the vfs-protocol
|
||||
// rendezvous directly with the rewritten mount-relative path.
|
||||
const Route = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: ipc.Handle, path: [224]u8, path_len: usize },
|
||||
|
||||
fn backendPath(self: *const Route) []const u8 {
|
||||
return self.backend.path[0..self.backend.path_len];
|
||||
}
|
||||
};
|
||||
|
||||
fn resolve(path: []const u8, flags: usize) ?Route {
|
||||
var out: [224]u8 = undefined;
|
||||
const route = system.fsResolve(path, flags, &out) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .kernel = token },
|
||||
.backend => |b| {
|
||||
var r: Route = .{ .backend = .{ .handle = b.handle, .path = undefined, .path_len = b.path_len } };
|
||||
@memcpy(r.backend.path[0..b.path_len], out[0..b.path_len]);
|
||||
return r;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
const Result = struct { reply: vfs_protocol.Reply, payload: []u8 };
|
||||
|
||||
// One request/reply round trip: [Request header][send payload] -> backend ->
|
||||
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||
fn transact(h: ipc.Handle, request: vfs_protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
var message: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..vfs_protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, vfs_protocol.maximum_payload);
|
||||
@memcpy(message[vfs_protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. vfs_protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < vfs_protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(vfs_protocol.Reply, rbuf[0..vfs_protocol.reply_size]);
|
||||
const rpl = @min(n - vfs_protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[vfs_protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
/// An open file: a VFS node plus a byte cursor. Read and write advance the cursor.
|
||||
pub const File = struct {
|
||||
node: u64,
|
||||
offset: u64 = 0,
|
||||
/// The owning backend's endpoint, or null for a kernel-served node (the
|
||||
/// read-only /system tree), whose `node` is a permanent fs_node token.
|
||||
backend: ?ipc.Handle = null,
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
|
||||
/// null on error.
|
||||
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||
const h = self.backend orelse {
|
||||
const n = system.fsNodeRead(self.node, self.offset, buffer) orelse return null;
|
||||
self.offset += n;
|
||||
return n;
|
||||
};
|
||||
const want: u32 = @intCast(@min(buffer.len, vfs_protocol.maximum_payload));
|
||||
const request = vfs_protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(h, request, &.{}, buffer) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write `data` at the current offset; returns the count written. A single
|
||||
/// call is capped at the VFS payload size, so the return may be short — use
|
||||
/// `writeAll` to write the whole slice. Null on error (kernel-served nodes
|
||||
/// are read-only).
|
||||
pub fn write(self: *File, data: []const u8) ?usize {
|
||||
const h = self.backend orelse return null;
|
||||
const want: u32 = @intCast(@min(data.len, vfs_protocol.maximum_payload));
|
||||
const request = vfs_protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(h, request, data[0..want], &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write all of `data`, looping past the per-call payload cap. Returns the
|
||||
/// total written, or null if a write failed before any progress.
|
||||
pub fn writeAll(self: *File, data: []const u8) ?usize {
|
||||
var written: usize = 0;
|
||||
while (written < data.len) {
|
||||
const n = self.write(data[written..]) orelse return if (written == 0) null else written;
|
||||
if (n == 0) return written; // no forward progress; stop rather than spin
|
||||
written += n;
|
||||
}
|
||||
return written;
|
||||
}
|
||||
|
||||
/// Move the read/write cursor to an absolute byte position.
|
||||
pub fn seekTo(self: *File, position: u64) void {
|
||||
self.offset = position;
|
||||
}
|
||||
|
||||
/// This file's metadata.
|
||||
pub fn attributes(self: *File) ?Attributes {
|
||||
const h = self.backend orelse {
|
||||
const a = system.fsNodeStatus(self.node) orelse return null;
|
||||
return .{ .size = a.size, .kind = if (a.kind == system.file_kind_directory) .directory else .regular, .mtime = a.mtime };
|
||||
};
|
||||
const request = vfs_protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(vfs_protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return null;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(vfs_protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(vfs_protocol.FileStatus, buffer[0..@sizeOf(vfs_protocol.FileStatus)]);
|
||||
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||
}
|
||||
|
||||
/// Release the backend's open handle for this file. Kernel-served node
|
||||
/// tokens are permanent — nothing to release.
|
||||
pub fn close(self: *File) void {
|
||||
const h = self.backend orelse return;
|
||||
const request = vfs_protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(h, request, &.{}, &.{});
|
||||
}
|
||||
};
|
||||
|
||||
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
|
||||
pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||
const route = resolve(path, options.wireFlags()) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .node = token, .backend = null },
|
||||
.backend => |b| {
|
||||
const relative = route.backendPath();
|
||||
const request = vfs_protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() };
|
||||
const r = transact(b.handle, request, relative, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node, .backend = b.handle };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// A path's metadata without keeping it open (open -> status -> close).
|
||||
pub fn attributes(path: []const u8) ?Attributes {
|
||||
var file = open(path, .{}) orelse return null;
|
||||
defer file.close();
|
||||
return file.attributes();
|
||||
}
|
||||
|
||||
/// Whether `path` resolves — handy as a readiness check (e.g. waiting for a mount
|
||||
/// to come up before writing to it).
|
||||
pub fn exists(path: []const u8) bool {
|
||||
return attributes(path) != null;
|
||||
}
|
||||
|
||||
/// One entry returned by `Directory.next`.
|
||||
pub const Entry = struct {
|
||||
kind: Kind = .regular,
|
||||
size: u64 = 0,
|
||||
name_buffer: [64]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
|
||||
pub fn name(self: *const Entry) []const u8 {
|
||||
return self.name_buffer[0..self.name_len];
|
||||
}
|
||||
};
|
||||
|
||||
/// An open directory being listed, cursor-advanced by `next`.
|
||||
pub const Directory = struct {
|
||||
node: u64,
|
||||
cursor: u64 = 0,
|
||||
backend: ?ipc.Handle = null,
|
||||
|
||||
/// Fill `entry` with the next directory entry; false at end of directory or
|
||||
/// on error.
|
||||
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||
const h = self.backend orelse {
|
||||
var buffer: [@sizeOf(system.DirectoryEntryHeader) + 64]u8 = undefined;
|
||||
const n = system.fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false;
|
||||
if (n < @sizeOf(system.DirectoryEntryHeader)) return false; // end
|
||||
const header = std.mem.bytesToValue(system.DirectoryEntryHeader, buffer[0..@sizeOf(system.DirectoryEntryHeader)]);
|
||||
entry.kind = if (header.kind == system.file_kind_directory) .directory else .regular;
|
||||
entry.size = header.size;
|
||||
const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(system.DirectoryEntryHeader)..][0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
};
|
||||
const request = vfs_protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return false;
|
||||
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||
if (r.payload.len < vfs_protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(vfs_protocol.DirectoryEntry, r.payload[0..vfs_protocol.directory_entry_size]);
|
||||
entry.kind = kindFromWire(header.kind);
|
||||
entry.size = header.size;
|
||||
const source = r.payload[vfs_protocol.directory_entry_size..];
|
||||
const nlen = @min(@min(@as(usize, header.name_len), source.len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], source[0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release the backend's open handle for this directory.
|
||||
pub fn close(self: *Directory) void {
|
||||
var f = File{ .node = self.node, .backend = self.backend };
|
||||
f.close();
|
||||
}
|
||||
};
|
||||
|
||||
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
|
||||
pub fn openDirectory(path: []const u8) ?Directory {
|
||||
const file = open(path, .{ .directory = true }) orelse return null;
|
||||
return .{ .node = file.node, .backend = file.backend };
|
||||
}
|
||||
|
||||
// A path-based request that returns only a status (mkdir, unlink). Kernel-served
|
||||
// paths (the read-only /system) refuse mutation by construction: the resolve
|
||||
// must land on a backend.
|
||||
fn pathOperation(operation: vfs_protocol.Operation, path: []const u8) bool {
|
||||
const route = resolve(path, 0) orelse return false;
|
||||
if (route != .backend) return false;
|
||||
const relative = route.backendPath();
|
||||
const request = vfs_protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
const r = transact(route.backend.handle, request, relative, &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Create a directory at `path` (its parent must already exist). Returns true on
|
||||
/// success. Only works under a mounted filesystem that supports directories.
|
||||
pub fn makeDirectory(path: []const u8) bool {
|
||||
return pathOperation(.mkdir, path);
|
||||
}
|
||||
|
||||
/// Create every missing directory along `path` (mkdir -p). Probes each prefix
|
||||
/// with `exists` first — a FAT mkdir of an existing name is refused, and the
|
||||
/// probe keeps the common "already there" case cheap. Returns true when the
|
||||
/// whole path exists afterwards.
|
||||
pub fn makePath(path: []const u8) bool {
|
||||
var end: usize = 0;
|
||||
while (end < path.len) {
|
||||
end += 1;
|
||||
while (end < path.len and path[end] != '/') end += 1;
|
||||
const prefix = path[0..end];
|
||||
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
||||
// Best-effort per prefix: components at or above a mount point ("/mnt")
|
||||
// are router names, not filesystem nodes — they neither exist as nodes
|
||||
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
||||
if (!exists(prefix)) _ = makeDirectory(prefix);
|
||||
}
|
||||
return exists(path);
|
||||
}
|
||||
|
||||
/// Remove the file at `path`. Returns true on success. Directories are refused
|
||||
/// (a separate directory-removal would have to check emptiness).
|
||||
pub fn remove(path: []const u8) bool {
|
||||
return pathOperation(.unlink, path);
|
||||
}
|
||||
|
||||
/// Rename `old_path` to `new_path`. Both must resolve to the SAME filesystem
|
||||
/// backend (same-directory, 8.3-name rename only for now). Returns true on
|
||||
/// success.
|
||||
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
const old_route = resolve(old_path, 0) orelse return false;
|
||||
const new_route = resolve(new_path, 0) orelse return false;
|
||||
if (old_route != .backend or new_route != .backend) return false;
|
||||
if (old_route.backend.handle != new_route.backend.handle) return false; // cross-filesystem
|
||||
const old_relative = old_route.backendPath();
|
||||
const new_relative = new_route.backendPath();
|
||||
const total = old_relative.len + 1 + new_relative.len;
|
||||
if (total > vfs_protocol.maximum_payload) return false;
|
||||
var payload: [vfs_protocol.maximum_payload]u8 = undefined;
|
||||
@memcpy(payload[0..old_relative.len], old_relative);
|
||||
payload[old_relative.len] = 0;
|
||||
@memcpy(payload[old_relative.len + 1 ..][0..new_relative.len], new_relative);
|
||||
const request = vfs_protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const r = transact(old_route.backend.handle, request, payload[0..total], &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||
/// the kernel VFS then routes everything under `target` to that backend.
|
||||
/// Possession of the endpoint handle is the capability. Returns true on success.
|
||||
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||
return system.fsMount(target, backend, "");
|
||||
}
|
||||
|
||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||
return system.fsMount(target, backend, rewrite);
|
||||
}
|
||||
@@ -0,0 +1,215 @@
|
||||
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
|
||||
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
|
||||
//! and Zig `std` containers share one heap.
|
||||
//!
|
||||
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||
//! that changes on this side of the system_call boundary is where memory comes from
|
||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||
//! itself, and the kernel picks the base address.
|
||||
//!
|
||||
//! 16-byte maximum alignment, exactly like the kernel heap. The free list is guarded by
|
||||
//! a `Thread.Mutex` **only in multi-threaded binaries** (`addThreadedUserBinary`): the
|
||||
//! guard is gated on `builtin.single_threaded`, so an ordinary single-threaded binary
|
||||
//! compiles it out and pays nothing, while a threaded one can allocate safely from
|
||||
//! several threads at once (docs/threading-plan.md M7). The lock lives at the two
|
||||
//! free-list mutators — `rawAlloc`/`rawFree` — which every entry point funnels through.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const system_calls = @import("system.zig");
|
||||
const Mutex = @import("thread.zig").Thread.Mutex;
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// Guards `free_list`. A no-op in single-threaded builds (compiled out); a real futex
|
||||
/// mutex in threaded ones. Uncontended acquisition is a single CAS — no syscall.
|
||||
var heap_mutex: Mutex = .{};
|
||||
|
||||
inline fn lockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.lock();
|
||||
}
|
||||
inline fn unlockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.unlock();
|
||||
}
|
||||
|
||||
/// A block header, at the start of every block; while free it also links the
|
||||
/// free list via `next`.
|
||||
const Block = extern struct {
|
||||
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||
next: ?*Block, // free-list link (only meaningful while free)
|
||||
};
|
||||
|
||||
const header_size = @sizeOf(Block); // 16
|
||||
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||
const chunk = 64 * 1024;
|
||||
|
||||
var free_list: ?*Block = null;
|
||||
|
||||
fn alignUp(value: usize, alignment: usize) usize {
|
||||
return (value + alignment - 1) & ~(alignment - 1);
|
||||
}
|
||||
|
||||
fn payloadOf(block: *Block) [*]u8 {
|
||||
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||
}
|
||||
|
||||
/// Ask the kernel for more pages and add them as a free block. Because each
|
||||
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||
if (system_calls.mmapFailed(ret)) return false;
|
||||
|
||||
const block: *Block = @ptrFromInt(ret);
|
||||
block.size = bytes;
|
||||
insertFree(block); // coalesces if this grant is adjacent to a prior one
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Insert a block into the address-ordered free list, coalescing with the
|
||||
/// physically adjacent free blocks on either side.
|
||||
fn insertFree(block: *Block) void {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |c| : (current = c.next) {
|
||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||
previous = c;
|
||||
}
|
||||
|
||||
block.next = current;
|
||||
if (previous) |p| p.next = block else free_list = block;
|
||||
|
||||
// Merge forward into `current` if they're contiguous.
|
||||
if (current) |c| {
|
||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||
block.size += c.size;
|
||||
block.next = c.next;
|
||||
}
|
||||
}
|
||||
// Merge `previous` forward into `block` if they're contiguous.
|
||||
if (previous) |p| {
|
||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||
p.size += block.size;
|
||||
p.next = block.next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory. Holds the heap lock
|
||||
/// across the free-list search and any `grow` (which also touches the free list).
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < 2) : (attempts += 1) {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |block| : ({
|
||||
previous = block;
|
||||
current = block.next;
|
||||
}) {
|
||||
if (block.size < need) continue;
|
||||
|
||||
if (block.size >= need + minimum_block) {
|
||||
// Split: carve `need` off the front, leave the rest free.
|
||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||
rest.size = block.size - need;
|
||||
rest.next = block.next;
|
||||
if (previous) |p| p.next = rest else free_list = rest;
|
||||
block.size = need;
|
||||
} else {
|
||||
// Take the whole block.
|
||||
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||
}
|
||||
return payloadOf(block);
|
||||
}
|
||||
|
||||
// Nothing fit: grow and try once more.
|
||||
if (!grow(need)) return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
// --- C ABI: the global implicit heap ---------------------------------------
|
||||
// `extern "C"` symbols so future C code links the same malloc/free directly.
|
||||
|
||||
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||
if (size == 0) return null;
|
||||
const p = rawAlloc(size) orelse return null;
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||
const p = ptr orelse return;
|
||||
rawFree(@ptrCast(p));
|
||||
}
|
||||
|
||||
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
|
||||
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
|
||||
if (total == 0) return null;
|
||||
const p = rawAlloc(total) orelse return null;
|
||||
@memset(p[0..total], 0);
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
|
||||
const p = ptr orelse return malloc(size);
|
||||
if (size == 0) {
|
||||
rawFree(@ptrCast(p));
|
||||
return null;
|
||||
}
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
|
||||
const old_payload = block.size - header_size;
|
||||
if (size <= old_payload) return p; // shrink/same: keep the block
|
||||
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
|
||||
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
|
||||
rawFree(@ptrCast(p));
|
||||
return @ptrCast(np);
|
||||
}
|
||||
|
||||
// --- std.mem.Allocator interface (same free list) --------------------------
|
||||
|
||||
pub fn allocator() std.mem.Allocator {
|
||||
return .{ .ptr = undefined, .vtable = &vtable };
|
||||
}
|
||||
|
||||
const vtable = std.mem.Allocator.VTable{
|
||||
.alloc = allocImpl,
|
||||
.resize = resizeImpl,
|
||||
.remap = remapImpl,
|
||||
.free = freeImpl,
|
||||
};
|
||||
|
||||
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
|
||||
return rawAlloc(len);
|
||||
}
|
||||
|
||||
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
|
||||
// In-place iff the new payload still fits the current block.
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
|
||||
return new_len + header_size <= block.size;
|
||||
}
|
||||
|
||||
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||
rawFree(memory.ptr);
|
||||
}
|
||||
@@ -0,0 +1,220 @@
|
||||
//! User-space input helpers: the client and publisher sides of the input service, so a
|
||||
//! program listening for input events — or a driver broadcasting them — doesn't hand-roll
|
||||
//! the IPC. Layered over `ipc` (endpoints, capability passing, `send`) and the shared
|
||||
//! `input-protocol` wire format, the way `device.zig` layers over the raw `device_*` calls.
|
||||
//! See system/services/input/input.zig.
|
||||
//!
|
||||
//! The service carries several device classes (keyboard, mouse, joystick/gamepad). A
|
||||
//! **source** publishes its class with the matching method:
|
||||
//! var source = input.connectSource() orelse return;
|
||||
//! _ = source.publishKeyboardEvent(.{ .kind = ..., .keycode = ..., ... });
|
||||
//! _ = source.publishMouseEvent(.{ ... });
|
||||
//! _ = source.publishJoystickEvent(.{ ... });
|
||||
//!
|
||||
//! A **subscriber** either takes one class with a typed helper —
|
||||
//! var keys = input.subscribeKeyboard() orelse return;
|
||||
//! while (true) { const key = keys.next() orelse continue; ... }
|
||||
//! — or takes several at once and inspects the tagged envelope:
|
||||
//! var listener = input.subscribeAll() orelse return;
|
||||
//! while (true) {
|
||||
//! const event = listener.next() orelse continue;
|
||||
//! if (event.asKeyboard()) |k| { ... } else if (event.asMouse()) |m| { ... }
|
||||
//! }
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const input_protocol = @import("input-protocol");
|
||||
|
||||
pub const DeviceKind = input_protocol.DeviceKind;
|
||||
pub const InputEvent = input_protocol.InputEvent;
|
||||
pub const KeyEvent = input_protocol.KeyEvent;
|
||||
pub const MouseEvent = input_protocol.MouseEvent;
|
||||
pub const JoystickEvent = input_protocol.JoystickEvent;
|
||||
pub const EventKind = input_protocol.EventKind;
|
||||
pub const MouseEventKind = input_protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = input_protocol.JoystickEventKind;
|
||||
pub const Keycode = input_protocol.Keycode;
|
||||
|
||||
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||
/// input.device_mouse)`.
|
||||
pub const device_keyboard = input_protocol.device_keyboard;
|
||||
pub const device_mouse = input_protocol.device_mouse;
|
||||
pub const device_joystick = input_protocol.device_joystick;
|
||||
pub const device_all = input_protocol.device_all;
|
||||
|
||||
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||
/// failing. Returns the service endpoint handle, or null if it never appears.
|
||||
fn lookupService() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.input)) |handle| return handle;
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- subscribing ------------------------------------------------------------
|
||||
|
||||
/// A subscription to the input service: our own endpoint, which the service pushes events
|
||||
/// to. `next` returns each event as a tagged `InputEvent`; use `asKeyboard`/`asMouse`/
|
||||
/// `asJoystick` to decode. Created with `subscribe`/`subscribeAll`; for a single device
|
||||
/// class prefer the typed helpers (`subscribeKeyboard`, ...), which return decoded events.
|
||||
pub const Subscriber = struct {
|
||||
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||
/// was handed to the service as a capability at subscribe time).
|
||||
endpoint: ipc.Handle,
|
||||
receive: [input_protocol.event_size]u8 = undefined,
|
||||
|
||||
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||
/// empty reply this issues is a harmless no-op. Returns null for any non-event wake-up
|
||||
/// (there should be none), so callers can loop.
|
||||
pub fn next(self: *Subscriber) ?InputEvent {
|
||||
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||
if (!got.isMessage() or got.len < input_protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..input_protocol.event_size]);
|
||||
}
|
||||
};
|
||||
|
||||
/// Subscribe to the input classes named in `device_mask` (an OR of `device_*`, or
|
||||
/// `device_all`). Creates an endpoint for the service to push to and hands it over as a
|
||||
/// capability. Returns a `Subscriber` to loop `next` on, or null on failure.
|
||||
pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||
const service = lookupService() orelse return null;
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
|
||||
var request = input_protocol.Request{ .operation = @intFromEnum(input_protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [input_protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < input_protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(input_protocol.Reply, reply[0..input_protocol.reply_size]).status != 0) return null;
|
||||
return .{ .endpoint = endpoint };
|
||||
}
|
||||
|
||||
/// Subscribe to every input class (keyboard, mouse, joystick) on one stream.
|
||||
pub fn subscribeAll() ?Subscriber {
|
||||
return subscribe(device_all);
|
||||
}
|
||||
|
||||
/// A subscriber filtered to keyboard events, whose `next` returns a decoded `KeyEvent`.
|
||||
pub const KeyboardSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *KeyboardSubscriber) ?KeyEvent {
|
||||
return (self.inner.next() orelse return null).asKeyboard();
|
||||
}
|
||||
};
|
||||
|
||||
/// A subscriber filtered to mouse events, whose `next` returns a decoded `MouseEvent`.
|
||||
pub const MouseSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *MouseSubscriber) ?MouseEvent {
|
||||
return (self.inner.next() orelse return null).asMouse();
|
||||
}
|
||||
};
|
||||
|
||||
/// A subscriber filtered to joystick/gamepad events, whose `next` returns a decoded
|
||||
/// `JoystickEvent`.
|
||||
pub const JoystickSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *JoystickSubscriber) ?JoystickEvent {
|
||||
return (self.inner.next() orelse return null).asJoystick();
|
||||
}
|
||||
};
|
||||
|
||||
/// Subscribe to keyboard events only; `next` returns decoded `KeyEvent`s.
|
||||
pub fn subscribeKeyboard() ?KeyboardSubscriber {
|
||||
return .{ .inner = subscribe(device_keyboard) orelse return null };
|
||||
}
|
||||
|
||||
/// Subscribe to mouse events only; `next` returns decoded `MouseEvent`s.
|
||||
pub fn subscribeMouse() ?MouseSubscriber {
|
||||
return .{ .inner = subscribe(device_mouse) orelse return null };
|
||||
}
|
||||
|
||||
/// Subscribe to joystick/gamepad events only; `next` returns decoded `JoystickEvent`s.
|
||||
pub fn subscribeJoystick() ?JoystickSubscriber {
|
||||
return .{ .inner = subscribe(device_joystick) orelse return null };
|
||||
}
|
||||
|
||||
// --- publishing -------------------------------------------------------------
|
||||
|
||||
/// A connection to the input service for a source (a keyboard/mouse/joystick driver) that
|
||||
/// publishes events. Each `publish*Event` is a short synchronous call the service answers
|
||||
/// at once; its own fan-out to subscribers is asynchronous, so publishing never blocks on
|
||||
/// a slow subscriber.
|
||||
pub const Publisher = struct {
|
||||
service: ipc.Handle,
|
||||
|
||||
fn publish(self: Publisher, event: InputEvent) bool {
|
||||
var request = input_protocol.Request{ .operation = @intFromEnum(input_protocol.Operation.publish), .event = event };
|
||||
var reply: [input_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (len < input_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(input_protocol.Reply, reply[0..input_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||
pub fn publishKeyboardEvent(self: Publisher, event: KeyEvent) bool {
|
||||
return self.publish(InputEvent.fromKeyboard(event));
|
||||
}
|
||||
/// Broadcast a mouse event to every subscriber that took mouse events.
|
||||
pub fn publishMouseEvent(self: Publisher, event: MouseEvent) bool {
|
||||
return self.publish(InputEvent.fromMouse(event));
|
||||
}
|
||||
/// Broadcast a joystick/gamepad event to every subscriber that took joystick events.
|
||||
pub fn publishJoystickEvent(self: Publisher, event: JoystickEvent) bool {
|
||||
return self.publish(InputEvent.fromJoystick(event));
|
||||
}
|
||||
};
|
||||
|
||||
/// Connect to the input service as an event source, waiting for it to come up. Returns a
|
||||
/// `Publisher`, or null if the service never registered.
|
||||
pub fn connectSource() ?Publisher {
|
||||
return .{ .service = lookupService() orelse return null };
|
||||
}
|
||||
|
||||
// --- synthetic scaffolding --------------------------------------------------
|
||||
|
||||
/// Synthetic key events, shared by the demo source and the keyboard driver's placeholder
|
||||
/// stream while real scancode decoding is still a follow-up. `step` rolls through A..E,
|
||||
/// emitting for each key a `key_down`, then a `key_press` carrying the character, then a
|
||||
/// `key_up`. Scaffolding, not wire protocol — hence it lives with the helpers.
|
||||
pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||
const Key = struct { code: Keycode, character: u32 };
|
||||
const keys = [_]Key{
|
||||
.{ .code = .a, .character = 'A' },
|
||||
.{ .code = .b, .character = 'B' },
|
||||
.{ .code = .c, .character = 'C' },
|
||||
.{ .code = .d, .character = 'D' },
|
||||
.{ .code = .e, .character = 'E' },
|
||||
};
|
||||
const key = keys[(step / 3) % keys.len];
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(EventKind.key_down), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||
1 => .{ .kind = @intFromEnum(EventKind.key_press), .keycode = @intFromEnum(key.code), .character = key.character, .modifiers = 0 },
|
||||
else => .{ .kind = @intFromEnum(EventKind.key_up), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
/// Synthetic mouse events (placeholder until real PS/2 packet decoding). `step` alternates
|
||||
/// a small diagonal motion with a left-button click.
|
||||
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = input_protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = input_protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = input_protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
/// Synthetic joystick/gamepad events (placeholder until a real controller driver). `step`
|
||||
/// sweeps axis 0 and toggles button 0.
|
||||
pub fn syntheticJoystickEvent(step: usize) JoystickEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(JoystickEventKind.axis), .control = 0, .value = 16384, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(JoystickEventKind.button_down), .control = 0, .value = 0, .buttons = 1 },
|
||||
else => .{ .kind = @intFromEnum(JoystickEventKind.button_up), .control = 0, .value = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,194 @@
|
||||
//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client
|
||||
//! `call`s an endpoint (send + block for reply); the VFS server and drivers are
|
||||
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||
//! added with the first server binary.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// A small-int handle into the calling process's handle table.
|
||||
pub const Handle = usize;
|
||||
|
||||
/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver)
|
||||
/// layer their own wire format on top of the bytes a call carries.
|
||||
pub const Message = extern struct {
|
||||
tag: u64 = 0,
|
||||
a: u64 = 0,
|
||||
b: u64 = 0,
|
||||
c: u64 = 0,
|
||||
};
|
||||
|
||||
/// Whether a system_call return value is a wrapped -errno (lands in the top page).
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Create a new endpoint owned by this process; returns its handle.
|
||||
pub fn createIpcEndpoint() ?Handle {
|
||||
const r = sc.systemCall0(.create_ipc_endpoint);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
pub const CallError = error{Failed};
|
||||
|
||||
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
||||
/// an endpoint the server sent back (e.g. a per-device channel), or null.
|
||||
pub const Reply = struct {
|
||||
len: usize,
|
||||
cap: ?Handle,
|
||||
};
|
||||
|
||||
/// Send `message` to endpoint `h` and block until the server replies into `reply`,
|
||||
/// optionally handing the server a capability (`send_cap`) and receiving one back.
|
||||
/// This is the class-driver "open" primitive: call a bus with `send_cap = null`, get a
|
||||
/// private per-device endpoint back in `.cap`. Two return values (reply length in rax,
|
||||
/// received handle in r8) need a hand-written stub — r8 is read-write (in: reply
|
||||
/// capacity, arg #4; out: the received handle).
|
||||
pub fn callCap(h: Handle, message: []const u8, reply: []u8, send_cap: ?Handle) CallError!Reply {
|
||||
var rax: usize = undefined;
|
||||
var r8: usize = reply.len; // in: reply capacity (arg #4); out: received capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[r8] "+{r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_call)),
|
||||
[a0] "{rdi}" (h),
|
||||
[a1] "{rsi}" (@intFromPtr(message.ptr)),
|
||||
[a2] "{rdx}" (message.len),
|
||||
[a3] "{r10}" (@intFromPtr(reply.ptr)),
|
||||
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return error.Failed;
|
||||
return .{ .len = rax, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
|
||||
/// Send `message` to endpoint `h` and block until the server replies into `reply`.
|
||||
/// Returns the reply length. The common case: no capability passed either way.
|
||||
pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||
return (try callCap(h, message, reply, null)).len;
|
||||
}
|
||||
|
||||
/// Post `message` to endpoint `h`'s asynchronous queue and return immediately — no
|
||||
/// rendezvous, no reply, no blocking. The receiver picks it up through `replyWait` as a
|
||||
/// buffered message (`Received.isMessage`). Unlike `call`, this **cannot hang on a dead
|
||||
/// or slow peer**, which is why a broadcaster (the input service) delivers events this
|
||||
/// way. The payload must fit an endpoint slot (64 bytes); a full queue drops the oldest
|
||||
/// message. Returns false on failure (bad handle, oversized payload, bad buffer).
|
||||
pub fn send(h: Handle, message: []const u8) bool {
|
||||
return !failed(sc.systemCall3(.ipc_send, h, @intFromPtr(message.ptr), message.len));
|
||||
}
|
||||
|
||||
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||
/// GSI. See `isNotification`.
|
||||
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **signal** — the
|
||||
/// lifecycle vocabulary of docs/process-lifecycle.md, delivered to the endpoint
|
||||
/// nominated with `process.bindSignals`. Decode with `process.signalsFrom`.
|
||||
pub const notify_signal_bit: u64 = abi.notify_signal_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **one-shot timer**
|
||||
/// landing (`system.timerOnce`).
|
||||
pub const notify_timer_bit: u64 = abi.notify_timer_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **child-exit
|
||||
/// notice** — a process this one spawned (with an exit endpoint) has ended —
|
||||
/// rather than a device interrupt. The low bits carry the child's process id.
|
||||
pub const notify_exit_bit: u64 = abi.notify_exit_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the wake-up is a **buffered message** — a payload
|
||||
/// posted with `send` (`ipc_send`) — rather than a bare device interrupt or child-exit
|
||||
/// notice. The payload is in the `replyWait` receive buffer (`Received.len` bytes); the
|
||||
/// low bits of the badge carry the sender's task id. See `Received.isMessage`.
|
||||
pub const notify_message_bit: u64 = abi.notify_message_bit;
|
||||
|
||||
/// The result of a `replyWait`: the request length, the sender's badge (a task id, or
|
||||
/// an IRQ notification if the high bit is set), and any capability the request carried.
|
||||
pub const Received = struct {
|
||||
len: usize,
|
||||
badge: u64,
|
||||
cap: ?Handle,
|
||||
|
||||
/// True if this wake-up was an asynchronous notification (a device interrupt
|
||||
/// or a child-exit notice), not a client request. An event loop branches on
|
||||
/// this; there is no reply owed on the notification path.
|
||||
pub fn isNotification(self: Received) bool {
|
||||
return self.badge & notify_badge_bit != 0;
|
||||
}
|
||||
|
||||
/// True if this wake-up tells of a supervised child's end — the notification
|
||||
/// requested by passing an exit endpoint to `system.spawnSupervised`.
|
||||
pub fn isChildExit(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_exit_bit != 0;
|
||||
}
|
||||
|
||||
/// True if this wake-up is a **buffered message** posted with `send` (`ipc_send`):
|
||||
/// there is a payload in the receive buffer (`self.len` bytes) and no reply is owed.
|
||||
/// The subscriber side of a broadcast branches on this.
|
||||
pub fn isMessage(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_message_bit != 0;
|
||||
}
|
||||
|
||||
/// The task id of whoever posted a buffered message, meaningful only when
|
||||
/// Whether this arrival is a signal notification — decode the set with
|
||||
/// `process.signalsFrom(badge)`.
|
||||
pub fn isSignal(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_signal_bit != 0;
|
||||
}
|
||||
|
||||
/// Whether this arrival is a one-shot timer landing (`system.timerOnce`).
|
||||
pub fn isTimer(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_timer_bit != 0;
|
||||
}
|
||||
|
||||
/// `isMessage`. (The badge's low bits, with the three high marker bits masked off.)
|
||||
pub fn senderTaskId(self: Received) u32 {
|
||||
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit | notify_message_bit));
|
||||
}
|
||||
|
||||
/// The interrupt source (a GSI), meaningful only when `isNotification` and
|
||||
/// not `isChildExit`.
|
||||
pub fn source(self: Received) u64 {
|
||||
return self.badge & ~notify_badge_bit;
|
||||
}
|
||||
|
||||
/// The ended child's process id, meaningful only when `isChildExit`.
|
||||
pub fn childProcessId(self: Received) u32 {
|
||||
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit));
|
||||
}
|
||||
};
|
||||
|
||||
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
||||
/// optionally handing it `send_cap`), then block until the next request arrives in
|
||||
/// `receive`. Returns its length, the sender badge, and any capability the request
|
||||
/// carried (in `.cap`). Three return values — length in rax, badge in rdx, received
|
||||
/// handle in r8 — so it needs a hand-written stub: rdx is read-write (in: reply length,
|
||||
/// arg #3; out: badge) and r8 is read-write (in: receive capacity, arg #4; out: handle).
|
||||
pub fn replyWait(h: Handle, reply: []const u8, receive: []u8, send_cap: ?Handle) Received {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = reply.len; // in: reply_len (arg #3); out: badge
|
||||
var r8: usize = receive.len; // in: receive capacity (arg #4); out: received capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
[r8] "+{r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_reply_wait)),
|
||||
[a0] "{rdi}" (h),
|
||||
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
return .{ .len = rax, .badge = rdx, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
//! The per-process logger: std.log wired to the tagged kernel log ring.
|
||||
//!
|
||||
//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped
|
||||
//! logger); this backend formats the line into a fixed buffer and emits ONE
|
||||
//! `debug_write` record carrying the level. The kernel stamps the record with
|
||||
//! the sender's pid and task name (its binary path) — the process does NOT put
|
||||
//! its own name in the payload; attribution is the kernel's, structural and
|
||||
//! unforgeable. Serial shows the kernel-rendered `<path>: message` line, and
|
||||
//! the logger service demultiplexes the ring into one file per process.
|
||||
//!
|
||||
//! Installed for every user binary by the root shim (library/runtime/root.zig)
|
||||
//! via `std_options`; a program can override by declaring its own
|
||||
//! `pub const std_options`.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
fn levelOf(comptime level: std.log.Level) system.KlogLevel {
|
||||
return switch (level) {
|
||||
.err => .err,
|
||||
.warn => .warn,
|
||||
.info => .info,
|
||||
.debug => .debug,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn logFn(
|
||||
comptime level: std.log.Level,
|
||||
comptime scope: @EnumLiteral(),
|
||||
comptime format: []const u8,
|
||||
args: anytype,
|
||||
) void {
|
||||
// One record = one line = at most klog_maximum_message bytes of payload.
|
||||
// On overflow keep what fits and end with "~" so the record is still a
|
||||
// whole line (the kernel would split an embedded rest anyway).
|
||||
var buffer: [256]u8 = undefined;
|
||||
const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") ";
|
||||
const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: {
|
||||
buffer[buffer.len - 1] = '~';
|
||||
break :truncated buffer[0..];
|
||||
};
|
||||
_ = system.writeRecord(levelOf(level), line);
|
||||
}
|
||||
|
||||
/// The std.Options the root shim installs unless the program overrides it.
|
||||
/// Debug level: filtering is the log *reader's* job here — the ring is cheap,
|
||||
/// serial is a dev convenience, and the logger service keeps everything.
|
||||
pub const default_options: std.Options = .{
|
||||
.log_level = .debug,
|
||||
.logFn = logFn,
|
||||
};
|
||||
@@ -0,0 +1,137 @@
|
||||
//! Process-level runtime types: what a user program receives at entry (`Init`,
|
||||
//! the argv contract) and the process end of the lifecycle
|
||||
//! (docs/process-lifecycle.md) — today the exit reason a supervisor reads to
|
||||
//! decide restart; signals and the stop sequence land here with M17.4. Mirrors
|
||||
//! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds
|
||||
//! no data on freestanding targets, so the type is danos's own.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
|
||||
/// Everything a program receives at entry. Passed to
|
||||
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||
/// `pub fn main() void`. An `environment` field is added here once the kernel
|
||||
/// passes a non-empty envp (today it is always empty — see docs/sysv.md).
|
||||
pub const Init = struct {
|
||||
arguments: Arguments,
|
||||
};
|
||||
|
||||
/// The process arguments (argc/argv), parsed from the kernel-built System V
|
||||
/// entry block. The bytes live in the entry block at the top of the stack page,
|
||||
/// NUL-terminated, valid for the process's lifetime.
|
||||
pub const Arguments = struct {
|
||||
/// argc — at least 1: argument 0 is the path or name this binary was
|
||||
/// spawned as.
|
||||
count: usize,
|
||||
/// The argv pointers in the entry block (NULL-terminated after `count`
|
||||
/// entries).
|
||||
vector: [*]const [*:0]const u8,
|
||||
|
||||
/// Argument `index` (0 = the program's own path/name), or null if out of
|
||||
/// range.
|
||||
pub fn get(arguments: Arguments, index: usize) ?[:0]const u8 {
|
||||
if (index >= arguments.count) return null;
|
||||
return std.mem.span(arguments.vector[index]);
|
||||
}
|
||||
|
||||
pub fn iterate(arguments: Arguments) Iterator {
|
||||
return .{ .arguments = arguments };
|
||||
}
|
||||
|
||||
pub const Iterator = struct {
|
||||
arguments: Arguments,
|
||||
index: usize = 0,
|
||||
|
||||
pub fn next(iterator: *Iterator) ?[:0]const u8 {
|
||||
const argument = iterator.arguments.get(iterator.index) orelse return null;
|
||||
iterator.index += 1;
|
||||
return argument;
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
/// How a process ended — what a supervisor's restart policy reads: a clean exit
|
||||
/// meant to stop, a fault wants a restart with backoff, killed means the
|
||||
/// supervisor did it itself (docs/process-lifecycle.md).
|
||||
pub const ExitReason = abi.ExitReason;
|
||||
|
||||
/// How dead child `id` ended. Ask after the exit notification arrives — the
|
||||
/// kernel records the reason before it posts the notification, so this never
|
||||
/// races it. Returns null for an id that never lived, is still alive, was
|
||||
/// evicted from the kernel's bounded record, or is not this process's child
|
||||
/// (the same authority gate as `kill`).
|
||||
pub fn exitReason(id: u32) ?ExitReason {
|
||||
const r = sc.systemCall1(.process_exit_reason, id);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @enumFromInt(r);
|
||||
}
|
||||
|
||||
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||
/// names, message delivery. A signal is a one-way coalescing statement — never a
|
||||
/// question (liveness is the zero-length ping call) and never kill (that is
|
||||
/// `system.kill`, unhandleable by definition).
|
||||
pub const Signal = abi.Signal;
|
||||
|
||||
/// The coalesced set of signals one notification delivered: two pending
|
||||
/// terminates arrive as one. Decode a received badge with `signalsFrom`.
|
||||
pub const SignalSet = struct {
|
||||
pending: u32,
|
||||
|
||||
pub fn has(set: SignalSet, signal: Signal) bool {
|
||||
return set.pending & (@as(u32, 1) << @intFromEnum(signal)) != 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// Nominate `endpoint` as this process's signal endpoint. Signals posted while
|
||||
/// unbound have pended; they are delivered immediately on bind, coalesced.
|
||||
pub fn bindSignals(endpoint: usize) bool {
|
||||
return sc.systemCall1(.signal_bind, endpoint) == 0;
|
||||
}
|
||||
|
||||
/// Decode a received badge into the signals it delivered, or null if it is not
|
||||
/// a signal notification.
|
||||
pub fn signalsFrom(badge: u64) ?SignalSet {
|
||||
if (badge & abi.notify_badge_bit == 0 or badge & abi.notify_signal_bit == 0) return null;
|
||||
return .{ .pending = @truncate(badge & ~(abi.notify_badge_bit | abi.notify_signal_bit)) };
|
||||
}
|
||||
|
||||
/// Post `signal` to child `id` (or to yourself). Supervisor-gated, like kill;
|
||||
/// non-blocking, always — a statement, not a conversation.
|
||||
pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||
return sc.systemCall2(.process_signal, id, @intFromEnum(signal)) == 0;
|
||||
}
|
||||
|
||||
/// The standard stop sequence (docs/process-lifecycle.md): terminate, wait up to
|
||||
/// `deadline_ms` for the exit notification on `exit_endpoint` (the endpoint the
|
||||
/// child was spawned with), then kill. Any *other* notifications arriving on
|
||||
/// that endpoint while stopping are consumed and dropped — a supervisor with
|
||||
/// concurrent traffic implements the same sequence inside its own event loop
|
||||
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||
_ = sendSignal(id, .terminate);
|
||||
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||
var receive: [8]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||
}
|
||||
_ = system.kill(id);
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
}
|
||||
}
|
||||
|
||||
/// Subscribe `endpoint` to published exit events: every process death posts an
|
||||
/// asynchronous notification with the same badge encoding as a supervisor's exit
|
||||
/// notice (decode with `ipc.Received.isChildExit`/`childProcessId`). For stateful
|
||||
/// services: release what the dead client held — file handles, subscriptions —
|
||||
/// because a service must never depend on clients cleaning up after themselves
|
||||
/// (docs/process-lifecycle.md). Ungated, like `system.processes`.
|
||||
pub fn subscribeExits(endpoint: usize) bool {
|
||||
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
//! The root module every user binary is compiled through (build.zig,
|
||||
//! `addUserBinary`). The program's own file is imported as `program`, and this
|
||||
//! shim contributes the declarations Zig resolves from the compilation root —
|
||||
//! `main` (dispatched by runtime.start) and the panic handler — and pulls in the
|
||||
//! `_start` entry shim. A program therefore only defines `pub fn main`; nothing
|
||||
//! else is required in its source file.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const program = @import("program");
|
||||
|
||||
/// Resolved as `@import("root").main` by runtime.start's comptime dispatch.
|
||||
pub const main = program.main;
|
||||
|
||||
/// The panic handler for every safety check in the image (runtime.start.panic).
|
||||
pub const panic = runtime.panic;
|
||||
|
||||
/// std.log for every user binary goes to the tagged kernel log ring (the kernel
|
||||
/// stamps the sender; see runtime.log). A program overrides by declaring its
|
||||
/// own `pub const std_options`.
|
||||
pub const std_options: @import("std").Options =
|
||||
if (@hasDecl(program, "std_options")) program.std_options else runtime.log.default_options;
|
||||
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||
//! freestanding target), so all user programs share one implementation.
|
||||
//!
|
||||
//! A user binary only defines a `pub fn main() void` or
|
||||
//! `pub fn main(init: runtime.process.Init) void` (arguments arrive via `init`).
|
||||
//! The panic handler and the `_start` entry pull live in the shared compilation
|
||||
//! root, library/runtime/root.zig, which build.zig wires around every program —
|
||||
//! nothing to declare per source file.
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
pub const log = @import("log.zig");
|
||||
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||
pub const time = @import("time.zig");
|
||||
pub const heap = @import("heap.zig");
|
||||
pub const ipc = @import("ipc.zig");
|
||||
pub const start = @import("start.zig");
|
||||
|
||||
/// Client for talking to the device manager (the hello handshake a supervised
|
||||
/// driver owes at startup). See library/runtime/device-manager.zig. The wire
|
||||
/// protocol itself is the library/protocol/device-manager module, imported
|
||||
/// directly by drivers and services that speak it.
|
||||
pub const device_manager = @import("device-manager.zig");
|
||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||
/// service. See library/runtime/input.zig and system/services/input/.
|
||||
pub const input = @import("input.zig");
|
||||
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||
pub const device = @import("device.zig");
|
||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||
pub const dma = @import("dma.zig");
|
||||
|
||||
/// Shared cacheable memory: create a region + capability, pass the capability to another
|
||||
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shared-memory.zig
|
||||
/// and docs/display-v2.md.
|
||||
pub const shared_memory = @import("shared-memory.zig");
|
||||
|
||||
// The USB class-driver client moved to its domain home, library/device/usb (module
|
||||
// "usb"): it is bus-family logic, not core runtime, and re-exporting it here compiled it
|
||||
// into every binary. USB class drivers import it directly with @import("usb").
|
||||
|
||||
/// Block-device client: read/write a block device (a USB stick, via
|
||||
/// usb-storage). See library/runtime/block.zig.
|
||||
pub const block = @import("block.zig");
|
||||
|
||||
/// Display-service client: query the mode, and (from D3) create layers, draw, and
|
||||
/// present frames. See library/runtime/display.zig and system/services/display/.
|
||||
pub const display = @import("display.zig");
|
||||
|
||||
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||
/// layer danos programs use directly, and where the operations that later become
|
||||
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||
pub const fs = @import("fs.zig");
|
||||
|
||||
/// Re-exported so the root shim (root.zig) can install it as the panic handler.
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||
pub const process = @import("process.zig");
|
||||
|
||||
/// Threads: `runtime.Thread`, std.Thread-shaped, over the private thread ABI
|
||||
/// (docs/threading.md). A binary must be built multi-threaded to spawn.
|
||||
pub const Thread = @import("thread.zig").Thread;
|
||||
|
||||
/// The service harness: one replyWait loop folding requests, signals, and
|
||||
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||
pub const service = @import("service.zig");
|
||||
|
||||
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||
pub const allocator = heap.allocator;
|
||||
@@ -0,0 +1,83 @@
|
||||
//! The service harness (docs/process-lifecycle.md): one replyWait loop that
|
||||
//! folds protocol requests, signals, and subscribed notifications into
|
||||
//! callbacks — so the lifecycle contract ("answers ping, exits on terminate")
|
||||
//! is satisfied by construction and a service author writes domain logic only.
|
||||
//! Nothing is asynchronous inside the process: a callback runs at a point the
|
||||
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||
//! messages.
|
||||
//!
|
||||
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||
//! service author to implement — a wedged service simply fails to answer, which
|
||||
//! is the diagnosis (see docs/ipc.md).
|
||||
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const process = @import("process.zig");
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Called once with the service's endpoint before the loop starts — the
|
||||
/// place to subscribe to exit events, bind IRQs, or announce readiness.
|
||||
/// Return false to abort startup (the process exits).
|
||||
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||
/// One protocol request from `sender` (a task id): write the reply into
|
||||
/// `reply`, return its length. `capability` is the handle the request
|
||||
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||
/// endpoint). The zero-length ping never reaches this.
|
||||
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||
on_notification: ?*const fn (badge: u64) void = null,
|
||||
/// The reload signal. Default: ignored.
|
||||
on_reload: ?*const fn () void = null,
|
||||
/// The terminate signal, called before the loop returns. The clean exit is
|
||||
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||
/// arrives with no warning; this is for graceful extras only).
|
||||
on_terminate: ?*const fn () void = null,
|
||||
/// Publish the endpoint under a well-known service id at startup.
|
||||
service: ?abi.ServiceId = null,
|
||||
};
|
||||
|
||||
/// Run the service: create and (optionally) register the endpoint, bind signals
|
||||
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
||||
/// loop returns and main's return is the clean exit the supervisor reads as
|
||||
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
||||
/// (a service passes its protocol's message maximum).
|
||||
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||
if (callbacks.service) |id| {
|
||||
if (!ipc.register(id, endpoint)) return;
|
||||
}
|
||||
_ = process.bindSignals(endpoint);
|
||||
if (callbacks.init) |initialise| {
|
||||
if (!initialise(endpoint)) return;
|
||||
}
|
||||
|
||||
var reply_buffer: [maximum_message]u8 = undefined;
|
||||
var reply_len: usize = 0;
|
||||
var receive: [maximum_message]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||
if (got.isNotification()) {
|
||||
reply_len = 0; // nothing owed for a notification
|
||||
if (process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.reload)) {
|
||||
if (callbacks.on_reload) |onReload| onReload();
|
||||
}
|
||||
if (signals.has(.terminate)) {
|
||||
if (callbacks.on_terminate) |onTerminate| onTerminate();
|
||||
return; // the loop's return IS the clean exit
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||
continue;
|
||||
}
|
||||
if (got.len == 0) {
|
||||
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||
continue;
|
||||
}
|
||||
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
//! User-space shared memory: `shared_memory_create` / `shared_memory_map`. A process creates a shareable,
|
||||
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||
//! `shared_memory_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||
//! generalization of capability passing from endpoints to memory objects.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||
}
|
||||
|
||||
/// A shared region: the `ptr` the CPU touches, and the `handle` (a capability) to hand to
|
||||
/// another process as an `ipc_call` send_cap.
|
||||
pub const Region = struct {
|
||||
ptr: [*]u8,
|
||||
handle: ipc.Handle,
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||
/// Returns the region or null on failure. Two return values — virtual_address in rax, handle in rdx —
|
||||
/// so this is a hand-written stub like `dma.alloc`.
|
||||
pub fn create(len: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: the capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shared_memory_create)),
|
||||
[a0] "{rdi}" (len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .ptr = @ptrFromInt(rax), .handle = rdx, .len = len };
|
||||
}
|
||||
|
||||
/// Map the shared region named by a capability `handle` this process received (via an
|
||||
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||
/// Returns the pointer, or null on failure.
|
||||
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
const r = sc.systemCall1(.shared_memory_map, handle);
|
||||
if (failed(r)) return null;
|
||||
return @ptrFromInt(r);
|
||||
}
|
||||
|
||||
/// The guest-physical base of the shared region named by `handle` (which this process must
|
||||
/// hold a capability for). The region's frames are contiguous, so this single address plus
|
||||
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||
/// `attach_backing`. Returns null on failure.
|
||||
pub fn physical(handle: ipc.Handle) ?usize {
|
||||
const r = sc.systemCall1(.shared_memory_physical, handle);
|
||||
if (failed(r)) return null;
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||
//! `entry = _start` in build.zig); the shared compilation root, root.zig, forces
|
||||
//! this file to be analysed with `comptime { _ = &runtime.start._start; }`, so
|
||||
//! the whole runtime is linked in.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
const process = @import("process.zig");
|
||||
|
||||
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||
/// auxiliary vector, then the strings (see system/kernel/process.zig,
|
||||
/// `buildEntryStack`). Capture that address in rdi — the first SysV argument —
|
||||
/// before `call` disturbs the stack; the call's pushed return address also puts
|
||||
/// rsp ≡ 8 (mod 16), satisfying the ABI before any Zig frame runs. The `ud2` is a
|
||||
/// safety net if `rt_start` ever returns.
|
||||
pub export fn _start() callconv(.naked) noreturn {
|
||||
asm volatile (
|
||||
\\mov %%rsp, %%rdi
|
||||
\\call rt_start
|
||||
\\ud2
|
||||
);
|
||||
}
|
||||
|
||||
/// The first Zig frame, entered with `stack` pointing at the kernel-built entry
|
||||
/// block. Build the `process.Init` from it and dispatch to the program's `main`,
|
||||
/// whose signature is inspected at comptime. The heap is lazy (first alloc grows
|
||||
/// it), so there is no other runtime init to order here.
|
||||
export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||
const init: process.Init = .{ .arguments = .{
|
||||
.count = stack[0],
|
||||
.vector = @ptrCast(stack + 1),
|
||||
} };
|
||||
system.exit(callMain(init));
|
||||
}
|
||||
|
||||
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||
/// zero parameters or one `process.Init`; returns void, noreturn, u8, !void, or !u8.
|
||||
fn callMain(init: process.Init) u8 {
|
||||
const root = @import("root"); // root.zig, re-exporting the program's main
|
||||
const main_information = @typeInfo(@TypeOf(root.main)).@"fn";
|
||||
|
||||
const call_arguments = switch (main_information.params.len) {
|
||||
0 => .{},
|
||||
1 => arguments: {
|
||||
const Parameter = main_information.params[0].type orelse
|
||||
@compileError("main's parameter must be runtime.process.Init (not anytype)");
|
||||
if (Parameter != process.Init)
|
||||
@compileError("main's parameter must be runtime.process.Init, found " ++ @typeName(Parameter));
|
||||
break :arguments .{init};
|
||||
},
|
||||
else => @compileError("main takes no parameters or a single runtime.process.Init"),
|
||||
};
|
||||
|
||||
const ReturnType = main_information.return_type.?;
|
||||
switch (@typeInfo(ReturnType)) {
|
||||
.noreturn => @call(.auto, root.main, call_arguments),
|
||||
.void => {
|
||||
@call(.auto, root.main, call_arguments);
|
||||
return 0;
|
||||
},
|
||||
.int => {
|
||||
if (ReturnType != u8)
|
||||
@compileError("main's integer return type must be u8, found " ++ @typeName(ReturnType));
|
||||
return @call(.auto, root.main, call_arguments);
|
||||
},
|
||||
.error_union => {
|
||||
const payload = @call(.auto, root.main, call_arguments) catch |err| {
|
||||
var buffer: [128]u8 = undefined;
|
||||
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
|
||||
_ = system.write(line);
|
||||
return 1; // distinct from panic's 127
|
||||
};
|
||||
if (@TypeOf(payload) == void) return 0;
|
||||
if (@TypeOf(payload) == u8) return payload;
|
||||
@compileError("main's error-union payload must be void or u8, found " ++ @typeName(@TypeOf(payload)));
|
||||
},
|
||||
else => @compileError("main must return void, noreturn, u8, !void, or !u8, found " ++ @typeName(ReturnType)),
|
||||
}
|
||||
}
|
||||
|
||||
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||
system.exit(127);
|
||||
}
|
||||
}.panic);
|
||||
@@ -0,0 +1,67 @@
|
||||
//! Raw `system_call` instruction wrappers for user space — one per arity.
|
||||
//!
|
||||
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
||||
//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and
|
||||
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
||||
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||
//! instruction, so the kernel reads the 4th argument from r10.
|
||||
|
||||
const abi = @import("abi");
|
||||
const SystemCall = abi.SystemCall;
|
||||
|
||||
pub inline fn systemCall0(n: SystemCall) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall1(n: SystemCall, a0: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
[a3] "{r10}" (a3),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
[a3] "{r10}" (a3),
|
||||
[a4] "{r8}" (a4),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
@@ -0,0 +1,266 @@
|
||||
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single
|
||||
//! source of truth shared with the kernel dispatcher.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||
pub const PROT_READ: usize = abi.prot_read;
|
||||
pub const PROT_WRITE: usize = abi.prot_write;
|
||||
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||
|
||||
/// One `processes` entry — re-exported from the shared ABI so a user program can
|
||||
/// declare its snapshot buffer without importing `abi` itself.
|
||||
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// The tagged-log level of a record — re-exported so runtime.log and the logger
|
||||
/// service don't import `abi` themselves.
|
||||
pub const KlogLevel = abi.KlogLevel;
|
||||
pub const KlogStatus = abi.KlogStatus;
|
||||
pub const KlogRecordHeader = abi.KlogRecordHeader;
|
||||
pub const klog_record_header_size = abi.klog_record_header_size;
|
||||
pub const klog_record_alignment = abi.klog_record_alignment;
|
||||
pub const klog_record_magic = abi.klog_record_magic;
|
||||
pub const klog_flag_truncated = abi.klog_flag_truncated;
|
||||
pub const klog_maximum_message = abi.klog_maximum_message;
|
||||
pub const maximum_process_name = abi.maximum_process_name;
|
||||
pub const FileAttributes = abi.FileAttributes;
|
||||
pub const DirectoryEntryHeader = abi.DirectoryEntryHeader;
|
||||
pub const file_kind_regular = abi.file_kind_regular;
|
||||
pub const file_kind_directory = abi.file_kind_directory;
|
||||
|
||||
/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary
|
||||
/// output goes through std.log -> writeRecord). The kernel stamps the record
|
||||
/// with this process's id and name. Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return writeRecord(.raw, message);
|
||||
}
|
||||
|
||||
/// Emit one leveled record into the tagged kernel log ring. The kernel stamps
|
||||
/// pid/name/sequence/timestamp; the payload should be a single line (embedded
|
||||
/// newlines split into further records).
|
||||
pub fn writeRecord(level: KlogLevel, message: []const u8) usize {
|
||||
return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level));
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds.
|
||||
pub fn sleep(ms: usize) void {
|
||||
_ = sc.systemCall1(.sleep, ms);
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
|
||||
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
|
||||
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
|
||||
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
|
||||
/// and restart backoff are built from.
|
||||
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
||||
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
||||
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
||||
///
|
||||
/// const deadline = clock() + timeout_ns;
|
||||
/// while (clock() < deadline) { ... }
|
||||
pub fn clock() u64 {
|
||||
return @intCast(sc.systemCall0(.clock));
|
||||
}
|
||||
|
||||
/// Wall-clock time in Unix epoch seconds (UTC) — the real date/time, from the RTC.
|
||||
/// Unlike `clock` (monotonic since boot), this tracks calendar time, so it is what a
|
||||
/// filesystem stamps as a file's modification time. Formatting it into a calendar
|
||||
/// date/timezone is user-space policy layered on top.
|
||||
pub fn wallClock() u64 {
|
||||
return @intCast(sc.systemCall0(.wall_clock));
|
||||
}
|
||||
|
||||
/// Copy bytes out of the tagged kernel log ring — framed records of everything
|
||||
/// every process (and the kernel) has emitted — starting at stream offset
|
||||
/// `offset`, into `out`. Returns the byte count (0 = caught up), or null when
|
||||
/// `offset` fell behind the ring's tail (those records were overwritten) or
|
||||
/// lies past its head; re-sync via `klogStatus`. A reader parses
|
||||
/// [KlogRecordHeader][name][message] frames (8-byte aligned) from the bytes.
|
||||
pub fn klogRead(offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// The log ring's live cursors (oldest retained offset, end of stream, next
|
||||
/// sequence number) plus the wall-clock time of boot — how a log reader starts,
|
||||
/// detects loss, and names a per-boot log directory.
|
||||
pub fn klogStatus() ?KlogStatus {
|
||||
var status: KlogStatus = undefined;
|
||||
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null;
|
||||
return status;
|
||||
}
|
||||
|
||||
/// Where fs_resolve routed a path: served by the kernel (a permanent node
|
||||
/// token for fs_node) or by a userspace filesystem backend (an endpoint handle
|
||||
/// plus the rewritten mount-relative path, returned in the caller's buffer).
|
||||
pub const FsRoute = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: usize, path_len: usize },
|
||||
};
|
||||
|
||||
/// Route `path` through the kernel VFS. For a backend route the rewritten
|
||||
/// mount-relative path lands in `out` (behind a kernel-written length prefix,
|
||||
/// already stripped here: out[0..path_len] is the path).
|
||||
pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)),
|
||||
[a0] "{rdi}" (@intFromPtr(path.ptr)),
|
||||
[a1] "{rsi}" (path.len),
|
||||
[a3] "{r10}" (@intFromPtr(out.ptr)),
|
||||
[a4] "{r8}" (out.len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (@as(isize, @bitCast(rax)) < 0) return null;
|
||||
if (rax == abi.fs_route_kernel) return .{ .kernel = rdx };
|
||||
if (rax != abi.fs_route_backend) return null;
|
||||
const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8);
|
||||
if (path_len + 2 > out.len) return null;
|
||||
std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]);
|
||||
return .{ .backend = .{ .handle = rdx, .path_len = path_len } };
|
||||
}
|
||||
|
||||
/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read).
|
||||
pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// A kernel-served node's metadata (fs_node status).
|
||||
pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes {
|
||||
var attributes: abi.FileAttributes = undefined;
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attributes), @sizeOf(abi.FileAttributes));
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return attributes;
|
||||
}
|
||||
|
||||
/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills
|
||||
/// `out` with [DirectoryEntryHeader][name]; returns total bytes (0 = end).
|
||||
pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// Mount a userspace filesystem's endpoint at `prefix`, with an optional
|
||||
/// backend-side `rewrite` prefix ("" = none). Possession of the endpoint
|
||||
/// handle is the capability.
|
||||
pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool {
|
||||
return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0;
|
||||
}
|
||||
|
||||
pub fn fsUnmount(prefix: []const u8) bool {
|
||||
return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0;
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
pub fn exit(code: usize) noreturn {
|
||||
_ = sc.systemCall1(.exit, code);
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||
/// process, returning the child's process id (or null on failure). The child's
|
||||
/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process
|
||||
/// allowed to `kill` it. This is how a supervisor (the device manager) launches a
|
||||
/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the
|
||||
/// POSIX layer later).
|
||||
pub fn spawn(name: []const u8) ?u32 {
|
||||
return spawnSupervised(name, &.{}, null);
|
||||
}
|
||||
|
||||
/// Like `spawn`, but hands the child command-line arguments: they arrive as
|
||||
/// argv[1..] on its System V entry stack (argv[0] is still `name`).
|
||||
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||
return spawnSupervised(name, arguments, null);
|
||||
}
|
||||
|
||||
/// The full spawn: command-line arguments for the child, and an optional endpoint
|
||||
/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends
|
||||
/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via
|
||||
/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in
|
||||
/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can
|
||||
/// supervise many children. Arguments are marshalled to the kernel as one
|
||||
/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the
|
||||
/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or
|
||||
/// null on failure.
|
||||
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||
var blob: [256]u8 = undefined;
|
||||
var len: usize = 0;
|
||||
for (arguments, 0..) |argument, i| {
|
||||
if (i != 0) {
|
||||
if (len >= blob.len) return null;
|
||||
blob[len] = 0;
|
||||
len += 1;
|
||||
}
|
||||
if (len + argument.len > blob.len) return null;
|
||||
@memcpy(blob[len..][0..argument.len], argument);
|
||||
len += argument.len;
|
||||
}
|
||||
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @intCast(r);
|
||||
}
|
||||
|
||||
/// Snapshot the process table into `out` (up to its length) and return the total
|
||||
/// number of live processes — which may exceed `out.len`; call again with a larger
|
||||
/// buffer for the full listing. Kernel tasks are included, with an empty name.
|
||||
/// The primitive `ps` is built on.
|
||||
pub fn processes(out: []abi.ProcessDescriptor) usize {
|
||||
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||
}
|
||||
|
||||
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||
pub fn isProcessRunning(name: []const u8) bool {
|
||||
var table: [32]ProcessDescriptor = undefined;
|
||||
const total = processes(&table);
|
||||
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// End process `id`. Only its supervisor — the process that spawned it — may;
|
||||
/// anyone else gets false, as does a stale or unknown id (ids are never reused).
|
||||
/// Delivery is prompt but asynchronous, like a signal: a target caught running on
|
||||
/// another core dies at its next system call or timer tick. True means the kill
|
||||
/// is accepted and irrevocable; the exit notification (if an endpoint was given
|
||||
/// at spawn) confirms completion.
|
||||
pub fn kill(id: u32) bool {
|
||||
return sc.systemCall1(.process_kill, id) == 0;
|
||||
}
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||
/// memory and return the base virtual address. On failure returns a value in the
|
||||
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||
pub fn mmap(len: usize, prot: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, prot);
|
||||
}
|
||||
|
||||
/// Release a range previously handed out by `mmap`.
|
||||
pub fn munmap(base: usize, len: usize) usize {
|
||||
return sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
|
||||
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||
/// -errno, which lands in the top page — no real grant base is ever that high).
|
||||
pub inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
@@ -0,0 +1,493 @@
|
||||
//! `runtime.Thread` — threads for danos, shaped like Zig's `std.Thread` but built on
|
||||
//! danos's private thread ABI (docs/threading.md). Several tasks share one address
|
||||
//! space; `spawn` starts one, the kernel delivers the closure pointer in the new
|
||||
//! thread's rdi, a plain Zig trampoline runs the user function and calls `thread_exit`,
|
||||
//! and `join` blocks on the thread's exit notification. See docs/threading.md for why
|
||||
//! this mirrors `std.Thread`'s API rather than being the literal type.
|
||||
//!
|
||||
//! The closure (the function's captured args) lives at the **top of the thread's own
|
||||
//! stack**, not the heap — each thread's stack is private, so there is no shared-heap
|
||||
//! concurrency in the spawn/join machinery (the runtime heap is not yet thread-safe).
|
||||
//! A binary must be built multi-threaded (`addThreadedUserBinary`) before it may spawn.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const system = @import("system.zig");
|
||||
|
||||
/// True in a real danos binary; false when this module is compiled for host unit tests.
|
||||
/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state
|
||||
/// machines can be exercised on the host against `std.Thread.Futex` (docs/threading-plan.md
|
||||
/// M11), while the danos build uses the futex syscalls.
|
||||
const on_danos = builtin.os.tag == .freestanding;
|
||||
|
||||
/// A thread stack, if the caller does not override it. 64 KiB of mmap'd, zeroed pages.
|
||||
pub const default_stack_size: usize = 64 * 1024;
|
||||
|
||||
/// Bytes reserved at the top of each thread's stack for its per-thread TLS block (the
|
||||
/// self-pointer plus scratch slots reachable via `%fs`). docs/threading-plan.md M10.
|
||||
const tls_block_size: usize = 64;
|
||||
|
||||
pub const Thread = struct {
|
||||
/// The kernel task id of the spawned thread — what `join` waits on.
|
||||
tid: u32,
|
||||
/// The mmap'd stack, reclaimed by `join` (or at process exit after `detach`).
|
||||
stack_base: usize,
|
||||
stack_size: usize,
|
||||
|
||||
pub const Id = u32;
|
||||
|
||||
pub const SpawnConfig = struct {
|
||||
/// Bytes of stack, rounded up to whole pages by the kernel's mmap.
|
||||
stack_size: usize = default_stack_size,
|
||||
};
|
||||
|
||||
pub const SpawnError = error{
|
||||
/// The kernel refused the thread, the stack mmap failed, or no endpoint was free.
|
||||
SystemResources,
|
||||
};
|
||||
|
||||
/// Start `function(args...)` on a new thread sharing this address space. Mirrors
|
||||
/// `std.Thread.spawn`. The thread's return value is discarded (as in `std.Thread`);
|
||||
/// return data through shared state.
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread {
|
||||
const Args = @TypeOf(args);
|
||||
const Closure = struct {
|
||||
tls_base: usize,
|
||||
args: Args,
|
||||
/// Entered directly by the kernel with `self` in rdi (C ABI). Establishes this
|
||||
/// thread's TLS pointer, runs the user function, then ends the thread.
|
||||
fn entry(self_addr: usize) callconv(.c) noreturn {
|
||||
const self: *@This() = @ptrFromInt(self_addr);
|
||||
setThreadPointer(self.tls_base); // per-thread thread pointer before any user code
|
||||
@call(.auto, function, self.args);
|
||||
exitThread();
|
||||
}
|
||||
};
|
||||
|
||||
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(base)) return error.SystemResources;
|
||||
|
||||
// Top of the thread's own stack, downward: the closure, then a small per-thread TLS
|
||||
// block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is
|
||||
// scratch for user TLS), then the stack proper (rsp starts below the TLS block, so
|
||||
// the growing stack never overwrites either).
|
||||
var closure_addr = (base + config.stack_size) - @sizeOf(Closure);
|
||||
closure_addr &= ~@as(usize, @alignOf(Closure) - 1); // align the closure down
|
||||
|
||||
const tls_base = (closure_addr - tls_block_size) & ~@as(usize, 15);
|
||||
const tls: [*]usize = @ptrFromInt(tls_base);
|
||||
tls[0] = tls_base; // self-pointer (fs:0), as the x86_64 TLS ABI expects
|
||||
|
||||
const closure: *Closure = @ptrFromInt(closure_addr);
|
||||
closure.* = .{ .tls_base = tls_base, .args = args };
|
||||
|
||||
var stack_top = tls_base & ~@as(usize, 15); // 16-align below the TLS block
|
||||
stack_top -= 8; // ...then rsp % 16 == 8 at the C entry
|
||||
|
||||
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr);
|
||||
if (threadSpawnFailed(tid)) {
|
||||
_ = system.munmap(base, config.stack_size);
|
||||
return error.SystemResources;
|
||||
}
|
||||
return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size };
|
||||
}
|
||||
|
||||
/// Block until this thread finishes, then reclaim its stack. Mirrors
|
||||
/// `std.Thread.join`. The exit endpoint is private to this thread, so the first
|
||||
/// child-exit notification on it is this thread's.
|
||||
pub fn join(self: Thread) void {
|
||||
_ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited
|
||||
_ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||
}
|
||||
|
||||
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
|
||||
/// reclaimed at process exit (docs/threading-plan.md M3 — kernel-reaper stack reclaim
|
||||
/// for detached threads is a later refinement). Mirrors `std.Thread.detach`.
|
||||
pub fn detach(self: Thread) void {
|
||||
_ = self;
|
||||
}
|
||||
|
||||
/// The calling thread's id (its kernel task id). Mirrors `std.Thread.getCurrentId`.
|
||||
pub fn getCurrentId() Id {
|
||||
return @intCast(sc.systemCall0(.thread_self));
|
||||
}
|
||||
|
||||
/// The dense 0-based index of the core the calling thread is running on. A danos
|
||||
/// extension beyond `std.Thread`, used to observe genuine cross-core parallelism.
|
||||
pub fn currentCore() Id {
|
||||
return @intCast(sc.systemCall0(.current_core));
|
||||
}
|
||||
|
||||
/// Ask the kernel to end the calling thread. A WORKER never returns from this;
|
||||
/// the process's MAIN thread gets the kernel's refusal (-EPERM — the group ends
|
||||
/// only through exit, a fault, or process_kill, docs/shared-fate-plan.md) and
|
||||
/// the call returns. Exists for exactly that refusal path; workers end through
|
||||
/// the spawn trampoline, and a process ends through `system.exit`.
|
||||
pub fn tryExitCurrent() void {
|
||||
_ = sc.systemCall0(.thread_exit);
|
||||
}
|
||||
|
||||
/// `std.Thread.Futex`-shaped block/wake on a `u32` atomic — the primitive the
|
||||
/// blocking `Mutex`/`Condition`/`Semaphore` are built on. Waiters park in the
|
||||
/// kernel (no busy-wait), so an idle core still halts (docs/halting.md).
|
||||
pub const Futex = struct {
|
||||
/// Block while `ptr.* == expect`. Returns when woken by `wake`, or promptly if
|
||||
/// the value already differs (safe against spurious returns, as in std): the
|
||||
/// caller re-checks its condition in a loop.
|
||||
pub fn wait(ptr: *const std.atomic.Value(u32), expect: u32) void {
|
||||
if (comptime on_danos) {
|
||||
_ = futexWait(@intFromPtr(ptr), expect, 0);
|
||||
} else {
|
||||
// Host unit-test mock: spin+yield until the value changes (`wake` is a
|
||||
// no-op — the callers re-check their condition in a loop anyway). Correct,
|
||||
// if busy; fine for the state-machine tests.
|
||||
while (ptr.load(.acquire) == expect) std.Thread.yield() catch {};
|
||||
}
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first.
|
||||
pub fn timedWait(ptr: *const std.atomic.Value(u32), expect: u32, timeout_ns: u64) error{Timeout}!void {
|
||||
if (comptime on_danos) {
|
||||
if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout;
|
||||
} else {
|
||||
var spins: u64 = 0;
|
||||
const limit = timeout_ns / 1000 + 1;
|
||||
while (ptr.load(.acquire) == expect) : (spins += 1) {
|
||||
if (spins >= limit) return error.Timeout;
|
||||
std.Thread.yield() catch {};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Wake up to `max_waiters` threads blocked on `ptr`.
|
||||
pub fn wake(ptr: *const std.atomic.Value(u32), max_waiters: u32) void {
|
||||
if (comptime on_danos) {
|
||||
_ = futexWake(@intFromPtr(ptr), max_waiters);
|
||||
} else {
|
||||
// host mock: spin-waiters re-check their condition, so no wake is needed.
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// A mutual-exclusion lock, `std.Thread.Mutex`-shaped. The classic three-state
|
||||
/// futex mutex (unlocked / locked / contended): the fast path is a single CAS, and
|
||||
/// only a contended lock ever enters the kernel.
|
||||
pub const Mutex = struct {
|
||||
state: std.atomic.Value(u32) = std.atomic.Value(u32).init(unlocked),
|
||||
|
||||
const unlocked: u32 = 0;
|
||||
const locked: u32 = 1;
|
||||
const contended: u32 = 2;
|
||||
|
||||
/// Try to take the lock without blocking; returns whether it was acquired.
|
||||
pub fn tryLock(m: *Mutex) bool {
|
||||
return m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) == null;
|
||||
}
|
||||
|
||||
/// Acquire the lock, blocking in the kernel while it is contended.
|
||||
pub fn lock(m: *Mutex) void {
|
||||
if (m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) != null) m.lockSlow();
|
||||
}
|
||||
|
||||
fn lockSlow(m: *Mutex) void {
|
||||
@branchHint(.cold);
|
||||
// Mark the lock contended and take it as soon as it falls unlocked; park on
|
||||
// the futex while it stays contended. Marking contended may cause a spurious
|
||||
// wake on unlock (harmless), never a missed one.
|
||||
while (m.state.swap(contended, .acquire) != unlocked) {
|
||||
Futex.wait(&m.state, contended);
|
||||
}
|
||||
}
|
||||
|
||||
/// Release the lock; wake one waiter if the lock was contended.
|
||||
pub fn unlock(m: *Mutex) void {
|
||||
if (m.state.swap(unlocked, .release) == contended) Futex.wake(&m.state, 1);
|
||||
}
|
||||
};
|
||||
|
||||
/// A condition variable, `std.Thread.Condition`-shaped. Spurious wakeups are
|
||||
/// allowed — always wait in a predicate loop with the mutex held. Built on a futex
|
||||
/// sequence counter: a waiter samples the seq, drops the mutex, and parks until the
|
||||
/// seq changes (a signal that races the unlock bumps the seq, so it is not missed).
|
||||
pub const Condition = struct {
|
||||
seq: std.atomic.Value(u32) = std.atomic.Value(u32).init(0),
|
||||
|
||||
/// Atomically release `mutex` and block until signalled, then re-acquire it.
|
||||
pub fn wait(c: *Condition, mutex: *Mutex) void {
|
||||
const seq = c.seq.load(.acquire);
|
||||
mutex.unlock();
|
||||
Futex.wait(&c.seq, seq);
|
||||
mutex.lock();
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first. The
|
||||
/// mutex is re-acquired either way.
|
||||
pub fn timedWait(c: *Condition, mutex: *Mutex, timeout_ns: u64) error{Timeout}!void {
|
||||
const seq = c.seq.load(.acquire);
|
||||
mutex.unlock();
|
||||
const timed_out = if (Futex.timedWait(&c.seq, seq, timeout_ns)) |_| false else |_| true;
|
||||
mutex.lock();
|
||||
if (timed_out) return error.Timeout;
|
||||
}
|
||||
|
||||
/// Wake one waiter.
|
||||
pub fn signal(c: *Condition) void {
|
||||
_ = c.seq.fetchAdd(1, .release);
|
||||
Futex.wake(&c.seq, 1);
|
||||
}
|
||||
|
||||
/// Wake all waiters.
|
||||
pub fn broadcast(c: *Condition) void {
|
||||
_ = c.seq.fetchAdd(1, .release);
|
||||
Futex.wake(&c.seq, std.math.maxInt(u32));
|
||||
}
|
||||
};
|
||||
|
||||
/// A counting semaphore, `std.Thread.Semaphore`-shaped: a permit count guarded by a
|
||||
/// `Mutex` + `Condition`.
|
||||
pub const Semaphore = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
permits: usize = 0,
|
||||
|
||||
/// Take a permit, blocking until one is available.
|
||||
pub fn wait(s: *Semaphore) void {
|
||||
s.mutex.lock();
|
||||
defer s.mutex.unlock();
|
||||
while (s.permits == 0) s.cond.wait(&s.mutex);
|
||||
s.permits -= 1;
|
||||
}
|
||||
|
||||
/// Return a permit and wake a waiter.
|
||||
pub fn post(s: *Semaphore) void {
|
||||
s.mutex.lock();
|
||||
defer s.mutex.unlock();
|
||||
s.permits += 1;
|
||||
s.cond.signal();
|
||||
}
|
||||
};
|
||||
|
||||
/// A reader/writer lock, `std.Thread.RwLock`-shaped: many concurrent readers OR one
|
||||
/// exclusive writer. Reader-preferring (a steady stream of readers can delay a writer),
|
||||
/// built on `Mutex` + `Condition` over a signed state: `>0` = that many readers hold
|
||||
/// it, `-1` = a writer holds it, `0` = free.
|
||||
pub const RwLock = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
state: i64 = 0,
|
||||
|
||||
/// Acquire shared (read) access, blocking while a writer holds the lock.
|
||||
pub fn lockShared(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
while (rw.state < 0) rw.cond.wait(&rw.mutex);
|
||||
rw.state += 1;
|
||||
}
|
||||
|
||||
/// Try to acquire shared access without blocking.
|
||||
pub fn tryLockShared(rw: *RwLock) bool {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
if (rw.state < 0) return false;
|
||||
rw.state += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release shared access; wake a waiting writer once the last reader leaves.
|
||||
pub fn unlockShared(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
rw.state -= 1;
|
||||
if (rw.state == 0) rw.cond.broadcast();
|
||||
}
|
||||
|
||||
/// Acquire exclusive (write) access, blocking until no readers or writer remain.
|
||||
pub fn lock(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
while (rw.state != 0) rw.cond.wait(&rw.mutex);
|
||||
rw.state = -1;
|
||||
}
|
||||
|
||||
/// Try to acquire exclusive access without blocking.
|
||||
pub fn tryLock(rw: *RwLock) bool {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
if (rw.state != 0) return false;
|
||||
rw.state = -1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Release exclusive access; wake all waiters (they re-check their condition).
|
||||
pub fn unlock(rw: *RwLock) void {
|
||||
rw.mutex.lock();
|
||||
defer rw.mutex.unlock();
|
||||
rw.state = 0;
|
||||
rw.cond.broadcast();
|
||||
}
|
||||
};
|
||||
|
||||
/// A `std.Thread.WaitGroup`-shaped counter: `start` before spawning work, `finish` as
|
||||
/// each unit completes, `wait` blocks until the count returns to zero.
|
||||
pub const WaitGroup = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
counter: usize = 0,
|
||||
|
||||
/// Register one pending unit of work.
|
||||
pub fn start(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
wg.counter += 1;
|
||||
}
|
||||
|
||||
/// Mark one unit done; wake waiters if that was the last.
|
||||
pub fn finish(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
wg.counter -= 1;
|
||||
if (wg.counter == 0) wg.cond.broadcast();
|
||||
}
|
||||
|
||||
/// Block until every started unit has finished.
|
||||
pub fn wait(wg: *WaitGroup) void {
|
||||
wg.mutex.lock();
|
||||
defer wg.mutex.unlock();
|
||||
while (wg.counter != 0) wg.cond.wait(&wg.mutex);
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
/// thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid, or a wrapped error.
|
||||
fn threadSpawn(entry: usize, stack_top: usize, arg: usize) usize {
|
||||
const exit_endpoint: usize = @intCast(abi.no_cap); // join uses thread_join, not an endpoint
|
||||
return sc.systemCall4(.thread_spawn, entry, stack_top, arg, exit_endpoint);
|
||||
}
|
||||
|
||||
/// The kernel returns a real (small) task id on success and a wrapped `-1` on failure;
|
||||
/// no valid task id ever exceeds a u32.
|
||||
inline fn threadSpawnFailed(ret: usize) bool {
|
||||
return ret > std.math.maxInt(u32);
|
||||
}
|
||||
|
||||
/// End the calling thread. Never returns.
|
||||
fn exitThread() noreturn {
|
||||
_ = sc.systemCall0(.thread_exit);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Set the calling thread's FS base (its user TLS thread pointer).
|
||||
fn setThreadPointer(addr: usize) void {
|
||||
_ = sc.systemCall1(.set_thread_pointer, addr);
|
||||
}
|
||||
|
||||
/// futex_wait(addr, expect, timeout_ns) -> status (abi.futex_*).
|
||||
fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize {
|
||||
return sc.systemCall3(.futex_wait, addr, expect, timeout_ns);
|
||||
}
|
||||
|
||||
/// futex_wake(addr, count) -> number woken.
|
||||
fn futexWake(addr: usize, count: u32) usize {
|
||||
return sc.systemCall2(.futex_wake, addr, count);
|
||||
}
|
||||
|
||||
// --- host unit tests (docs/threading-plan.md M11) ---------------------------
|
||||
//
|
||||
// These run under `zig build test` on the host: the `Futex` seam above uses
|
||||
// `std.Thread.Futex` off-danos, so the lock/condvar state machines can be exercised by
|
||||
// real host threads. They are never compiled into a danos binary (test blocks only build
|
||||
// under test), so their `std.Thread` use is fine even though `std.Thread` is unavailable
|
||||
// on the freestanding target.
|
||||
|
||||
test "Mutex serialises concurrent increments across host threads" {
|
||||
var m: Thread.Mutex = .{};
|
||||
var counter: u64 = 0;
|
||||
const workers = 8;
|
||||
const per = 20_000;
|
||||
const Ctx = struct {
|
||||
m: *Thread.Mutex,
|
||||
c: *u64,
|
||||
fn run(ctx: @This()) void {
|
||||
var i: usize = 0;
|
||||
while (i < per) : (i += 1) {
|
||||
ctx.m.lock();
|
||||
ctx.c.* += 1;
|
||||
ctx.m.unlock();
|
||||
}
|
||||
}
|
||||
};
|
||||
var handles: [workers]std.Thread = undefined;
|
||||
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .m = &m, .c = &counter }});
|
||||
for (handles) |h| h.join();
|
||||
try std.testing.expectEqual(@as(u64, workers * per), counter);
|
||||
}
|
||||
|
||||
test "RwLock never lets a reader observe a half-written pair" {
|
||||
var rw: Thread.RwLock = .{};
|
||||
var a: u64 = 0;
|
||||
var b: u64 = 0; // invariant while a lock is held: a == b
|
||||
var stop = std.atomic.Value(bool).init(false);
|
||||
var ok = std.atomic.Value(bool).init(true);
|
||||
|
||||
const Writer = struct {
|
||||
rw: *Thread.RwLock,
|
||||
a: *u64,
|
||||
b: *u64,
|
||||
stop: *std.atomic.Value(bool),
|
||||
fn run(w: @This()) void {
|
||||
var v: u64 = 1;
|
||||
while (!w.stop.load(.acquire)) : (v +%= 1) {
|
||||
w.rw.lock();
|
||||
w.a.* = v; // update both halves under the exclusive lock...
|
||||
w.b.* = v;
|
||||
w.rw.unlock();
|
||||
}
|
||||
}
|
||||
};
|
||||
const Reader = struct {
|
||||
rw: *Thread.RwLock,
|
||||
a: *u64,
|
||||
b: *u64,
|
||||
ok: *std.atomic.Value(bool),
|
||||
fn run(r: @This()) void {
|
||||
var i: usize = 0;
|
||||
while (i < 200_000) : (i += 1) {
|
||||
r.rw.lockShared();
|
||||
if (r.a.* != r.b.*) r.ok.store(false, .release); // ...so a reader must never see them differ
|
||||
r.rw.unlockShared();
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
var writers: [2]std.Thread = undefined;
|
||||
for (&writers) |*w| w.* = try std.Thread.spawn(.{}, Writer.run, .{Writer{ .rw = &rw, .a = &a, .b = &b, .stop = &stop }});
|
||||
var readers: [4]std.Thread = undefined;
|
||||
for (&readers) |*rd| rd.* = try std.Thread.spawn(.{}, Reader.run, .{Reader{ .rw = &rw, .a = &a, .b = &b, .ok = &ok }});
|
||||
for (readers) |rd| rd.join();
|
||||
stop.store(true, .release);
|
||||
for (writers) |w| w.join();
|
||||
try std.testing.expect(ok.load(.acquire));
|
||||
}
|
||||
|
||||
test "WaitGroup blocks until every started unit finishes" {
|
||||
var wg: Thread.WaitGroup = .{};
|
||||
var done = std.atomic.Value(u32).init(0);
|
||||
const n = 6;
|
||||
const Ctx = struct {
|
||||
wg: *Thread.WaitGroup,
|
||||
done: *std.atomic.Value(u32),
|
||||
fn run(c: @This()) void {
|
||||
_ = c.done.fetchAdd(1, .monotonic);
|
||||
c.wg.finish();
|
||||
}
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < n) : (i += 1) wg.start();
|
||||
var handles: [n]std.Thread = undefined;
|
||||
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .wg = &wg, .done = &done }});
|
||||
wg.wait(); // must not return until all n finished
|
||||
try std.testing.expectEqual(@as(u32, n), done.load(.acquire));
|
||||
for (handles) |h| h.join();
|
||||
}
|
||||
@@ -0,0 +1,169 @@
|
||||
//! The danos time interface — monotonic time, delays, and deadlines for user space.
|
||||
//!
|
||||
//! There is no time *service*: the kernel already owns the scheduling timer and
|
||||
//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a
|
||||
//! scale), never an IPC round trip (docs/timers.md explains why). This module is a
|
||||
//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig`
|
||||
//! — an ergonomic `Instant`/`Duration` front door, not new mechanism.
|
||||
//!
|
||||
//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or
|
||||
//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed
|
||||
//! CLOCK_REALTIME) layered on top later.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
const nanos_per_micro: u64 = 1_000;
|
||||
const nanos_per_milli: u64 = 1_000_000;
|
||||
const nanos_per_second: u64 = 1_000_000_000;
|
||||
|
||||
/// A span of time, held as nanoseconds. Constructors name their unit; accessors
|
||||
/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the
|
||||
/// kernel's millisecond granularity and rounding down could return early.
|
||||
pub const Duration = struct {
|
||||
ns: u64,
|
||||
|
||||
pub fn fromNanos(n: u64) Duration {
|
||||
return .{ .ns = n };
|
||||
}
|
||||
pub fn fromMicros(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_micro };
|
||||
}
|
||||
pub fn fromMillis(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_milli };
|
||||
}
|
||||
pub fn fromSeconds(n: u64) Duration {
|
||||
return .{ .ns = n *| nanos_per_second };
|
||||
}
|
||||
|
||||
pub fn asNanos(d: Duration) u64 {
|
||||
return d.ns;
|
||||
}
|
||||
pub fn asMicros(d: Duration) u64 {
|
||||
return d.ns / nanos_per_micro;
|
||||
}
|
||||
pub fn asMillis(d: Duration) u64 {
|
||||
return d.ns / nanos_per_milli;
|
||||
}
|
||||
pub fn asSeconds(d: Duration) u64 {
|
||||
return d.ns / nanos_per_second;
|
||||
}
|
||||
|
||||
/// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel.
|
||||
/// A non-zero sub-millisecond duration becomes 1 ms rather than 0.
|
||||
pub fn ceilMillis(d: Duration) u64 {
|
||||
return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli;
|
||||
}
|
||||
|
||||
pub fn plus(a: Duration, b: Duration) Duration {
|
||||
return .{ .ns = a.ns +| b.ns };
|
||||
}
|
||||
};
|
||||
|
||||
/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract
|
||||
/// instants to measure elapsed time; it never runs backward, so `since` is safe to
|
||||
/// saturate at zero rather than wrap.
|
||||
pub const Instant = struct {
|
||||
ns: u64,
|
||||
|
||||
/// The span from `earlier` to `self`, saturating at zero if `earlier` is later
|
||||
/// (which the monotonic clock should never produce, but callers may pass any pair).
|
||||
pub fn since(self: Instant, earlier: Instant) Duration {
|
||||
return .{ .ns = self.ns -| earlier.ns };
|
||||
}
|
||||
|
||||
/// How long since this instant, sampled now.
|
||||
pub fn elapsed(self: Instant) Duration {
|
||||
return now().since(self);
|
||||
}
|
||||
|
||||
/// This instant advanced by `d` (a deadline, `d` from here).
|
||||
pub fn plus(self: Instant, d: Duration) Instant {
|
||||
return .{ .ns = self.ns +| d.ns };
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock has reached this instant (used as a deadline).
|
||||
pub fn reached(deadline: Instant) bool {
|
||||
return now().ns >= deadline.ns;
|
||||
}
|
||||
};
|
||||
|
||||
/// The current monotonic time.
|
||||
pub fn now() Instant {
|
||||
return .{ .ns = system.clock() };
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||
/// want a plain integer instead of an `Instant`.
|
||||
pub fn monotonicNanos() u64 {
|
||||
return system.clock();
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||
/// "unavailable" instead of assuming the clock advances.
|
||||
pub fn available() bool {
|
||||
return system.clock() != 0;
|
||||
}
|
||||
|
||||
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||
/// `spin`.
|
||||
pub fn sleep(d: Duration) void {
|
||||
system.sleep(d.ceilMillis());
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||
pub fn sleepMillis(ms: u64) void {
|
||||
system.sleep(ms);
|
||||
}
|
||||
|
||||
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||
/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot.
|
||||
/// Prefer `sleep` for anything at or above a millisecond.
|
||||
pub fn spin(d: Duration) void {
|
||||
const deadline = now().plus(d);
|
||||
while (!deadline.reached()) {}
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||
pub fn after(endpoint: usize, d: Duration) bool {
|
||||
return system.timerOnce(endpoint, d.ceilMillis());
|
||||
}
|
||||
|
||||
test "Duration unit conversions round toward zero" {
|
||||
try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos());
|
||||
try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos());
|
||||
try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis());
|
||||
try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros());
|
||||
}
|
||||
|
||||
test "ceilMillis rounds up, and never turns a nonzero span into zero" {
|
||||
try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis());
|
||||
try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis());
|
||||
}
|
||||
|
||||
test "Instant arithmetic: since saturates, plus/reached form deadlines" {
|
||||
const t0 = Instant{ .ns = 1_000 };
|
||||
const t1 = Instant{ .ns = 4_000 };
|
||||
try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos());
|
||||
// earlier-than-self can't happen on a monotonic clock; saturate rather than wrap.
|
||||
try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos());
|
||||
const deadline = t0.plus(Duration.fromNanos(2_500));
|
||||
try std.testing.expectEqual(@as(u64, 3_500), deadline.ns);
|
||||
}
|
||||
|
||||
test "saturating arithmetic does not overflow at the u64 ceiling" {
|
||||
const big = Duration.fromSeconds(std.math.maxInt(u64));
|
||||
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos());
|
||||
const late = Instant{ .ns = std.math.maxInt(u64) };
|
||||
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns);
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
/* Shared link layout for every user binary (init, servers, drivers).
|
||||
*
|
||||
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
|
||||
* inside the kernel's user region). Same discipline as the kernel's script:
|
||||
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
|
||||
* user-ELF loader can map each segment with exact W^X permissions. Note the
|
||||
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
|
||||
* base, so the entry point comes from e_entry, not the base address.
|
||||
*/
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
/* The `.large` code model (needed for the >4 GiB image base) emits code and
|
||||
* data into .ltext/.lrodata/.ldata/.lbss; fold those into the matching
|
||||
* permission segment alongside the normal names. */
|
||||
.text ALIGN(4K) : {
|
||||
*(.text .text.*)
|
||||
*(.ltext .ltext.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : {
|
||||
*(.rodata .rodata.*)
|
||||
*(.lrodata .lrodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : {
|
||||
*(.data .data.*)
|
||||
*(.ldata .ldata.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space; the loader zeroes the
|
||||
* filesz..memsz gap. */
|
||||
.bss ALIGN(4K) : {
|
||||
*(.bss .bss.*)
|
||||
*(.lbss .lbss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user