Re-organize the source tree as a monorepo mirroring the FHS
The source layout now mirrors the runtime filesystem hierarchy
(docs/danos-file-system-hierarchy-FSH.md): what lives under system/ in the
source is what a running danos represents under /system. Each service and
driver is a sub-project directory that is its own Zig module — cross-project
references go by module name, never by a path into another project's files.
Moves (all git mv, history preserved):
- src/ -> system/ (danos internals; the self-representation)
root.zig -> danos.zig (the kernel<->user contract module)
kernel/arch/ -> kernel/architecture/ (arch -> architecture)
device/ -> devices/ (what /system/devices reflects)
boot/ -> /boot (the loaders, top level)
- sbin/ -> split by role:
init, vfs -> system/services/<name>/<name>.zig
hpetd, busd -> system/drivers/<name>/<name>.zig
vfs-test -> system/services/vfs/vfs-test.zig (inside the vfs project)
- lib/ -> library/runtime/ (room for other libraries beside runtime)
The VFS wire protocol becomes its own module, system/services/vfs/protocol.zig
("vfs-protocol"): the vfs sub-project exposes its interface, and the runtime's
file layer imports it by name. First instance of the "protocol module" pattern
(docs/driver-model.md); usb/block will expose theirs the same way.
Also: fix a naming-standard violation in the protocol — Op -> Operation (and
req -> request, _pad -> _padding). Docs updated: /system/services added to the
FHS doc, a repository-layout section added to the docs index, and stale source
paths swept across comments and docs.
Runtime boot paths are unchanged (the bootloader still loads /sbin/init);
aligning the runtime filesystem to the FHS is a separate follow-up. Suite 35/35
plus host tests green.
This commit is contained in:
@@ -0,0 +1,67 @@
|
||||
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||
//! mapping registers or routing an IRQ.
|
||||
|
||||
const danos = @import("danos");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
pub const DeviceDescriptor = danos.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = danos.ResourceDescriptor;
|
||||
pub const DeviceClass = danos.DeviceClass;
|
||||
pub const ResourceKind = danos.ResourceKind;
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||
}
|
||||
|
||||
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||
pub fn claim(id: u64) bool {
|
||||
return !failed(sc.systemCall1(.device_claim, id));
|
||||
}
|
||||
|
||||
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||
/// `device_id` into this address space; returns the register base virtual address.
|
||||
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||
pub const no_parent = danos.no_parent;
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||
/// that class of device can `claim` it — that is how a bus hands off a device.
|
||||
///
|
||||
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
||||
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
||||
/// anything else, because a device descriptor is a licence to map physical memory and
|
||||
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||
/// through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
||||
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
||||
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
||||
}
|
||||
|
||||
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
||||
/// status register holds its line asserted) — the kernel left the line masked
|
||||
/// precisely because it could not do that for you. Skip it and the interrupt never
|
||||
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
||||
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
||||
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
|
||||
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
|
||||
//! and Zig `std` containers share one heap.
|
||||
//!
|
||||
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||
//! that changes on this side of the system_call boundary is where memory comes from
|
||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||
//! itself, and the kernel picks the base address.
|
||||
//!
|
||||
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
||||
//! lock and larger alignments come when user programs gain threads.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const system = @import("system.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
/// A block header, at the start of every block; while free it also links the
|
||||
/// free list via `next`.
|
||||
const Block = extern struct {
|
||||
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||
next: ?*Block, // free-list link (only meaningful while free)
|
||||
};
|
||||
|
||||
const header_size = @sizeOf(Block); // 16
|
||||
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||
const chunk = 64 * 1024;
|
||||
|
||||
var free_list: ?*Block = null;
|
||||
|
||||
fn alignUp(value: usize, alignment: usize) usize {
|
||||
return (value + alignment - 1) & ~(alignment - 1);
|
||||
}
|
||||
|
||||
fn payloadOf(block: *Block) [*]u8 {
|
||||
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||
}
|
||||
|
||||
/// Ask the kernel for more pages and add them as a free block. Because each
|
||||
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||
const ret = system.mmap(bytes, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(ret)) return false;
|
||||
|
||||
const block: *Block = @ptrFromInt(ret);
|
||||
block.size = bytes;
|
||||
insertFree(block); // coalesces if this grant is adjacent to a prior one
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Insert a block into the address-ordered free list, coalescing with the
|
||||
/// physically adjacent free blocks on either side.
|
||||
fn insertFree(block: *Block) void {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |c| : (current = c.next) {
|
||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||
previous = c;
|
||||
}
|
||||
|
||||
block.next = current;
|
||||
if (previous) |p| p.next = block else free_list = block;
|
||||
|
||||
// Merge forward into `current` if they're contiguous.
|
||||
if (current) |c| {
|
||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||
block.size += c.size;
|
||||
block.next = c.next;
|
||||
}
|
||||
}
|
||||
// Merge `previous` forward into `block` if they're contiguous.
|
||||
if (previous) |p| {
|
||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||
p.size += block.size;
|
||||
p.next = block.next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < 2) : (attempts += 1) {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |block| : ({
|
||||
previous = block;
|
||||
current = block.next;
|
||||
}) {
|
||||
if (block.size < need) continue;
|
||||
|
||||
if (block.size >= need + minimum_block) {
|
||||
// Split: carve `need` off the front, leave the rest free.
|
||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||
rest.size = block.size - need;
|
||||
rest.next = block.next;
|
||||
if (previous) |p| p.next = rest else free_list = rest;
|
||||
block.size = need;
|
||||
} else {
|
||||
// Take the whole block.
|
||||
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||
}
|
||||
return payloadOf(block);
|
||||
}
|
||||
|
||||
// Nothing fit: grow and try once more.
|
||||
if (!grow(need)) return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
// --- C ABI: the global implicit heap ---------------------------------------
|
||||
// `extern "C"` symbols so future C code links the same malloc/free directly.
|
||||
|
||||
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||
if (size == 0) return null;
|
||||
const p = rawAlloc(size) orelse return null;
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||
const p = ptr orelse return;
|
||||
rawFree(@ptrCast(p));
|
||||
}
|
||||
|
||||
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
|
||||
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
|
||||
if (total == 0) return null;
|
||||
const p = rawAlloc(total) orelse return null;
|
||||
@memset(p[0..total], 0);
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
|
||||
const p = ptr orelse return malloc(size);
|
||||
if (size == 0) {
|
||||
rawFree(@ptrCast(p));
|
||||
return null;
|
||||
}
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
|
||||
const old_payload = block.size - header_size;
|
||||
if (size <= old_payload) return p; // shrink/same: keep the block
|
||||
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
|
||||
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
|
||||
rawFree(@ptrCast(p));
|
||||
return @ptrCast(np);
|
||||
}
|
||||
|
||||
// --- std.mem.Allocator interface (same free list) --------------------------
|
||||
|
||||
pub fn allocator() std.mem.Allocator {
|
||||
return .{ .ptr = undefined, .vtable = &vtable };
|
||||
}
|
||||
|
||||
const vtable = std.mem.Allocator.VTable{
|
||||
.alloc = allocImpl,
|
||||
.resize = resizeImpl,
|
||||
.remap = remapImpl,
|
||||
.free = freeImpl,
|
||||
};
|
||||
|
||||
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
|
||||
return rawAlloc(len);
|
||||
}
|
||||
|
||||
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
|
||||
// In-place iff the new payload still fits the current block.
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
|
||||
return new_len + header_size <= block.size;
|
||||
}
|
||||
|
||||
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||
rawFree(memory.ptr);
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client
|
||||
//! `call`s an endpoint (send + block for reply); the VFS server and drivers are
|
||||
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||
//! added with the first server binary.
|
||||
|
||||
const danos = @import("danos");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// A small-int handle into the calling process's handle table.
|
||||
pub const Handle = usize;
|
||||
|
||||
/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver)
|
||||
/// layer their own wire format on top of the bytes a call carries.
|
||||
pub const Message = extern struct {
|
||||
tag: u64 = 0,
|
||||
a: u64 = 0,
|
||||
b: u64 = 0,
|
||||
c: u64 = 0,
|
||||
};
|
||||
|
||||
/// Whether a system_call return value is a wrapped -errno (lands in the top page).
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Create a new endpoint owned by this process; returns its handle.
|
||||
pub fn createEndpoint() ?Handle {
|
||||
const r = sc.systemCall0(.create_endpoint);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||
pub fn register(id: danos.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: danos.ServiceId) ?Handle {
|
||||
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
pub const CallError = error{Failed};
|
||||
|
||||
/// Send `message` to endpoint `h` and block until the server replies into `reply`.
|
||||
/// Returns the reply length.
|
||||
pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||
const r = sc.systemCall5(.ipc_call, h, @intFromPtr(message.ptr), message.len, @intFromPtr(reply.ptr), reply.len);
|
||||
return if (failed(r)) error.Failed else r;
|
||||
}
|
||||
|
||||
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||
/// GSI. See `isNotification`.
|
||||
pub const notify_badge_bit: u64 = danos.notify_badge_bit;
|
||||
|
||||
/// The result of a `replyWait`: the request length and the sender's badge (a
|
||||
/// task id, or an IRQ notification if the high bit is set).
|
||||
pub const Received = struct {
|
||||
len: usize,
|
||||
badge: u64,
|
||||
|
||||
/// True if this wake-up was a device interrupt, not a client request. A driver's
|
||||
/// event loop branches on this; there is no reply owed on the notification path.
|
||||
pub fn isNotification(self: Received) bool {
|
||||
return self.badge & notify_badge_bit != 0;
|
||||
}
|
||||
|
||||
/// The interrupt source (a GSI), meaningful only when `isNotification`.
|
||||
pub fn source(self: Received) u64 {
|
||||
return self.badge & ~notify_badge_bit;
|
||||
}
|
||||
};
|
||||
|
||||
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if
|
||||
/// any), then block until the next request arrives in `receive`. Returns its length
|
||||
/// and the sender badge. This system_call returns two values — the length in rax and
|
||||
/// the badge in rdx — so it needs a hand-written stub: rdx is a read-write
|
||||
/// operand (input = reply length, arg #3; output = badge).
|
||||
pub fn replyWait(h: Handle, reply: []const u8, receive: []u8) Received {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = reply.len; // in: reply_len (arg #3 -> rdx); out: badge
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(danos.SystemCall.ipc_reply_wait)),
|
||||
[a0] "{rdi}" (h),
|
||||
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||
[a4] "{r8}" (receive.len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
return .{ .len = rax, .badge = rdx };
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||
//! freestanding target), so all user programs share one implementation.
|
||||
//!
|
||||
//! A user binary needs three lines:
|
||||
//! const runtime = @import("runtime");
|
||||
//! pub const panic = runtime.panic;
|
||||
//! comptime { _ = &runtime.start._start; } // pull the entry shim in
|
||||
//! and a `pub fn main() void`.
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
pub const heap = @import("heap.zig");
|
||||
pub const ipc = @import("ipc.zig");
|
||||
pub const start = @import("start.zig");
|
||||
/// The VFS wire protocol (shared with the VFS server).
|
||||
pub const vfs_protocol = @import("vfs-protocol");
|
||||
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||
pub const unistd = @import("unistd.zig");
|
||||
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||
pub const stdio = @import("stdio.zig");
|
||||
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||
pub const device = @import("device.zig");
|
||||
|
||||
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||
pub const allocator = heap.allocator;
|
||||
@@ -0,0 +1,32 @@
|
||||
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
||||
//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
/// The kernel enters at `_start` with rsp 16-aligned, but a SystemV function expects
|
||||
/// rsp ≡ 8 (mod 16) on entry (as if reached by `call`). The `call` below pushes
|
||||
/// the 8-byte return address, satisfying the ABI before any Zig frame runs; the
|
||||
/// `ud2` is a safety net if `rt_start` ever returns.
|
||||
pub export fn _start() callconv(.naked) noreturn {
|
||||
asm volatile (
|
||||
\\call rt_start
|
||||
\\ud2
|
||||
);
|
||||
}
|
||||
|
||||
/// The first Zig frame. The heap is lazy (first alloc grows it), so there is no
|
||||
/// runtime init to order here — just hand control to the program's `main`.
|
||||
export fn rt_start() callconv(.c) noreturn {
|
||||
const root = @import("root"); // the user binary's root source file
|
||||
root.main();
|
||||
system.exit(0);
|
||||
}
|
||||
|
||||
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||
system.exit(127);
|
||||
}
|
||||
}.panic);
|
||||
@@ -0,0 +1,115 @@
|
||||
//! A small C stdio layer over the POSIX-style file API (unistd.zig). Unbuffered
|
||||
//! for now — each fread/fwrite is one VFS round trip; an internal buffer (fewer
|
||||
//! IPC calls) is a later optimisation. Both a Zig-callable API and `extern "C"`
|
||||
//! symbols are provided, so Zig and future C programs share it.
|
||||
|
||||
const std = @import("std");
|
||||
const unistd = @import("unistd.zig");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
pub const SEEK_SET = unistd.SEEK_SET;
|
||||
pub const SEEK_CURRENT = unistd.SEEK_CURRENT;
|
||||
pub const SEEK_END = unistd.SEEK_END;
|
||||
|
||||
/// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the
|
||||
/// heap; `fclose` frees it.
|
||||
pub const FILE = extern struct {
|
||||
fd: i32,
|
||||
eof: c_int = 0,
|
||||
err: c_int = 0,
|
||||
};
|
||||
|
||||
fn flagsFor(mode: []const u8) u32 {
|
||||
if (mode.len == 0) return 0;
|
||||
return switch (mode[0]) {
|
||||
'w', 'a' => unistd.O_CREAT,
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Open `path` in `mode` ("r"/"w"/"a", '+' ignored for now). Returns null on error.
|
||||
pub fn fopen(path: []const u8, mode: []const u8) ?*FILE {
|
||||
const fd = unistd.open(path, flagsFor(mode));
|
||||
if (fd < 0) return null;
|
||||
const f = heap.allocator().create(FILE) catch {
|
||||
unistd.close(fd);
|
||||
return null;
|
||||
};
|
||||
f.* = .{ .fd = fd };
|
||||
if (mode.len > 0 and mode[0] == 'a') _ = unistd.lseek(fd, 0, unistd.SEEK_END);
|
||||
return f;
|
||||
}
|
||||
|
||||
pub fn fclose(f: *FILE) c_int {
|
||||
unistd.close(f.fd);
|
||||
heap.allocator().destroy(f);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Read `size*nmemb` bytes; returns the number of whole items read.
|
||||
pub fn fread(buffer: []u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||
const total = size * nmemb;
|
||||
if (total == 0) return 0;
|
||||
const n = unistd.read(f.fd, buffer[0..@min(buffer.len, total)]);
|
||||
if (n <= 0) {
|
||||
f.eof = 1;
|
||||
return 0;
|
||||
}
|
||||
return @as(usize, @intCast(n)) / size;
|
||||
}
|
||||
|
||||
/// Write `size*nmemb` bytes; returns the number of whole items written.
|
||||
pub fn fwrite(data: []const u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||
const total = @min(data.len, size * nmemb);
|
||||
if (total == 0) return 0;
|
||||
const n = unistd.write(f.fd, data[0..total]);
|
||||
if (n <= 0) {
|
||||
f.err = 1;
|
||||
return 0;
|
||||
}
|
||||
return @as(usize, @intCast(n)) / size;
|
||||
}
|
||||
|
||||
pub fn fseek(f: *FILE, off: i64, whence: u32) c_int {
|
||||
f.eof = 0;
|
||||
return if (unistd.lseek(f.fd, off, whence) < 0) -1 else 0;
|
||||
}
|
||||
|
||||
pub fn ftell(f: *FILE) i64 {
|
||||
return unistd.lseek(f.fd, 0, unistd.SEEK_CURRENT);
|
||||
}
|
||||
|
||||
pub fn rewind(f: *FILE) void {
|
||||
_ = fseek(f, 0, SEEK_SET);
|
||||
}
|
||||
|
||||
pub fn feof(f: *FILE) c_int {
|
||||
return f.eof;
|
||||
}
|
||||
|
||||
pub fn ferror(f: *FILE) c_int {
|
||||
return f.err;
|
||||
}
|
||||
|
||||
pub fn fputs(s: []const u8, f: *FILE) c_int {
|
||||
return if (unistd.write(f.fd, s) < 0) -1 else 0;
|
||||
}
|
||||
|
||||
pub fn fputc(c: u8, f: *FILE) c_int {
|
||||
const b = [_]u8{c};
|
||||
return if (unistd.write(f.fd, &b) == 1) c else -1;
|
||||
}
|
||||
|
||||
pub fn fgetc(f: *FILE) c_int {
|
||||
var b: [1]u8 = undefined;
|
||||
const n = unistd.read(f.fd, &b);
|
||||
if (n <= 0) {
|
||||
f.eof = 1;
|
||||
return -1; // EOF
|
||||
}
|
||||
return b[0];
|
||||
}
|
||||
|
||||
// Real `extern "C"` symbols (fopen/fread/fseek/...) — with a C-string signature
|
||||
// distinct from the Zig slice API above — land with the first C program, wired
|
||||
// via @export so they don't collide with these Zig names.
|
||||
@@ -0,0 +1,52 @@
|
||||
//! Raw `system_call` instruction wrappers for user space — one per arity.
|
||||
//!
|
||||
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
||||
//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and
|
||||
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
||||
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||
//! instruction, so the kernel reads the 4th argument from r10.
|
||||
|
||||
const danos = @import("danos");
|
||||
const SystemCall = danos.SystemCall;
|
||||
|
||||
pub inline fn systemCall0(n: SystemCall) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall1(n: SystemCall, a0: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3), [a4] "{r8}" (a4),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||
//! stubs, one per kernel call. Numbers come from `danos.SystemCall`, the single
|
||||
//! source of truth shared with the kernel dispatcher.
|
||||
|
||||
const danos = @import("danos");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||
pub const PROT_READ: usize = danos.prot_read;
|
||||
pub const PROT_WRITE: usize = danos.prot_write;
|
||||
pub const PROT_EXEC: usize = danos.prot_exec;
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds.
|
||||
pub fn sleep(ms: usize) void {
|
||||
_ = sc.systemCall1(.sleep, ms);
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
pub fn exit(code: usize) noreturn {
|
||||
_ = sc.systemCall1(.exit, code);
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||
/// memory and return the base virtual address. On failure returns a value in the
|
||||
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||
pub fn mmap(len: usize, prot: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, prot);
|
||||
}
|
||||
|
||||
/// Release a range previously handed out by `mmap`.
|
||||
pub fn munmap(base: usize, len: usize) usize {
|
||||
return sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
|
||||
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||
/// -errno, which lands in the top page — no real grant base is ever that high).
|
||||
pub inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
@@ -0,0 +1,149 @@
|
||||
//! POSIX-style file API for user programs — the low level under C stdio. Files
|
||||
//! are named objects served by the user-space VFS server (system/services/vfs/vfs.zig); each
|
||||
//! call marshals a request, IPC_Calls the VFS, and unmarshals the reply. The
|
||||
//! kernel knows nothing of files or fds — the fd table lives here, per process.
|
||||
|
||||
const std = @import("std");
|
||||
const protocol = @import("vfs-protocol");
|
||||
const ipc = @import("ipc.zig");
|
||||
const danos = @import("danos");
|
||||
|
||||
pub const O_CREAT = protocol.O_CREAT;
|
||||
pub const SEEK_SET: u32 = 0;
|
||||
pub const SEEK_CURRENT: u32 = 1;
|
||||
pub const SEEK_END: u32 = 2;
|
||||
|
||||
// Resolve (and cache) the VFS server endpoint, looked up by well-known id.
|
||||
var vfs_handle: usize = 0;
|
||||
var vfs_resolved = false;
|
||||
fn vfs() ?usize {
|
||||
if (!vfs_resolved) {
|
||||
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||
vfs_resolved = true;
|
||||
}
|
||||
return vfs_handle;
|
||||
}
|
||||
|
||||
const maximum_fds = 32;
|
||||
const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 };
|
||||
var fds = [_]Fd{.{}} ** maximum_fds;
|
||||
|
||||
fn allocFd() ?usize {
|
||||
for (&fds, 0..) |*f, i| {
|
||||
if (!f.used) {
|
||||
f.* = .{ .used = true };
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||
|
||||
/// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||
/// [Reply header][receive payload]. The receive payload is written into `out`.
|
||||
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
const h = vfs() orelse return null;
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||
const rpl = @min(n - protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
/// Open (or create, with O_CREAT) `path`; returns an fd or -1.
|
||||
pub fn open(path: []const u8, flags: u32) i32 {
|
||||
const fd = allocFd() orelse return -1;
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
|
||||
const r = transact(request, path, &.{}) orelse {
|
||||
fds[fd].used = false;
|
||||
return -1;
|
||||
};
|
||||
if (r.reply.status != 0) {
|
||||
fds[fd].used = false;
|
||||
return -1;
|
||||
}
|
||||
fds[fd] = .{ .used = true, .node = r.reply.node, .offset = 0 };
|
||||
return @intCast(fd);
|
||||
}
|
||||
|
||||
fn fdPtr(fd: i32) ?*Fd {
|
||||
if (fd < 0 or fd >= maximum_fds) return null;
|
||||
const f = &fds[@intCast(fd)];
|
||||
return if (f.used) f else null;
|
||||
}
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count or -1.
|
||||
pub fn read(fd: i32, buffer: []u8) isize {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, &.{}, buffer) orelse return -1;
|
||||
if (r.reply.status != 0) return -1;
|
||||
f.offset += r.reply.len;
|
||||
return @intCast(r.reply.len);
|
||||
}
|
||||
|
||||
/// Write `data` at the current offset; returns the count or -1.
|
||||
pub fn write(fd: i32, data: []const u8) isize {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, data[0..want], &.{}) orelse return -1;
|
||||
if (r.reply.status != 0) return -1;
|
||||
f.offset += r.reply.len;
|
||||
return @intCast(r.reply.len);
|
||||
}
|
||||
|
||||
/// Reposition the fd's offset. Returns the new offset or -1. (SEEK_END needs the
|
||||
/// file size, which `stat` provides; handled by fetching it here.)
|
||||
pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const base: i64 = switch (whence) {
|
||||
SEEK_SET => 0,
|
||||
SEEK_CURRENT => @intCast(f.offset),
|
||||
SEEK_END => blk: {
|
||||
const request = protocol.Request{ .operation = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1;
|
||||
const st = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]);
|
||||
break :blk @intCast(st.size);
|
||||
},
|
||||
else => return -1,
|
||||
};
|
||||
const pos = base + off;
|
||||
if (pos < 0) return -1;
|
||||
f.offset = @intCast(pos);
|
||||
return pos;
|
||||
}
|
||||
|
||||
/// Stat `path`. Returns 0 or -1.
|
||||
pub fn stat(path: []const u8, out: *protocol.Stat) i32 {
|
||||
// Open, stat by node, close — simple and enough for now.
|
||||
const fd = open(path, 0);
|
||||
if (fd < 0) return -1;
|
||||
defer close(fd);
|
||||
const f = fdPtr(fd).?;
|
||||
const request = protocol.Request{ .operation = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1;
|
||||
out.* = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Close an fd (best effort — tells the VFS to release the open file).
|
||||
pub fn close(fd: i32) void {
|
||||
const f = fdPtr(fd) orelse return;
|
||||
const request = protocol.Request{ .operation = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(request, &.{}, &.{});
|
||||
f.used = false;
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
/* Shared link layout for every user binary (init, servers, drivers).
|
||||
*
|
||||
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
|
||||
* inside the kernel's user region). Same discipline as the kernel's script:
|
||||
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
|
||||
* user-ELF loader can map each segment with exact W^X permissions. Note the
|
||||
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
|
||||
* base, so the entry point comes from e_entry, not the base address.
|
||||
*/
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
/* The `.large` code model (needed for the >4 GiB image base) emits code and
|
||||
* data into .ltext/.lrodata/.ldata/.lbss; fold those into the matching
|
||||
* permission segment alongside the normal names. */
|
||||
.text ALIGN(4K) : {
|
||||
*(.text .text.*)
|
||||
*(.ltext .ltext.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : {
|
||||
*(.rodata .rodata.*)
|
||||
*(.lrodata .lrodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : {
|
||||
*(.data .data.*)
|
||||
*(.ldata .ldata.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space; the loader zeroes the
|
||||
* filesz..memsz gap. */
|
||||
.bss ALIGN(4K) : {
|
||||
*(.bss .bss.*)
|
||||
*(.lbss .lbss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user