M7: synchronous IPC — endpoints, handle table, IPC_Call/ReplyWait
The microkernel message backbone the VFS server and drivers will ride on. - src/kernel/ipc_sync.zig: Endpoint (sender FIFO threaded via Task.next + a recv WaitQueue for servers + a small notification ring). call() (client sends, wakes a server, blocks) and replyWait() (server replies to the held caller, then receives the next). Reply routing keys on Task.ipc_client — synchronous IPC owes one reply at a time. Payloads copy frame-to-frame through the physmap (copyAcross on arch.translate, added in M5); an unmapped page fails the copy instead of #PF-ing. MSG_MAX 256. - Bootstrap naming: an integer name registry (danos.ServiceId, vfs=1) with create_endpoint / ipc_register / ipc_lookup — any process finds a server without threading a handle through spawn. - notifyFromIsr(): ISR-safe async wake (badge with the high bit set), the hook M10's IRQ-as-message needs. Unused/untested until then. - Task gains handles[16] (opaque *Endpoint, to avoid a sched<->ipc import cycle) + ipc_client/send/reply/status fields; sched gains blockCurrentLocked/readyLocked; arch gains setSyscallResult2 (rdx badge). - Syscalls 6..10 wired in process.zig; exit() now drops the caller's endpoint refs. lib/ipc.zig: user-side createEndpoint/register/lookup/call (server-side replyWait lands with the first server in M9). - New `ipc-call` test: two kernel tasks ping-pong 100 calls, every reply request+1. Suite 29/29.
This commit is contained in:
+46
-7
@@ -1,10 +1,16 @@
|
||||
//! User-space IPC helpers. The kernel's synchronous IPC syscalls
|
||||
//! (create_endpoint, ipc_register/lookup, ipc_call, ipc_reply_wait) arrive in a
|
||||
//! later milestone; this reserves the module boundary now so drivers and servers
|
||||
//! can be written against `rt.ipc` without restructuring once they light up.
|
||||
//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client
|
||||
//! `call`s an endpoint (send + block for reply); the VFS server and drivers are
|
||||
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||
//! added with the first server binary.
|
||||
|
||||
/// A fixed-size, register-friendly message payload. The wire format for the VFS
|
||||
/// and driver protocols is layered on top of this by the servers themselves.
|
||||
const danos = @import("danos");
|
||||
const sc = @import("syscall.zig");
|
||||
|
||||
/// A small-int handle into the calling process's handle table.
|
||||
pub const Handle = usize;
|
||||
|
||||
/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver)
|
||||
/// layer their own wire format on top of the bytes a call carries.
|
||||
pub const Message = extern struct {
|
||||
tag: u64 = 0,
|
||||
a: u64 = 0,
|
||||
@@ -12,4 +18,37 @@ pub const Message = extern struct {
|
||||
c: u64 = 0,
|
||||
};
|
||||
|
||||
// call() / replyWait() / createEndpoint() land with the kernel IPC syscalls.
|
||||
/// Whether a syscall return value is a wrapped -errno (lands in the top page).
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Create a new endpoint owned by this process; returns its handle.
|
||||
pub fn createEndpoint() ?Handle {
|
||||
const r = sc.syscall0(.create_endpoint);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||
pub fn register(id: danos.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.syscall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: danos.ServiceId) ?Handle {
|
||||
const r = sc.syscall1(.ipc_lookup, @intFromEnum(id));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
pub const CallError = error{Failed};
|
||||
|
||||
/// Send `msg` to endpoint `h` and block until the server replies into `reply`.
|
||||
/// Returns the reply length.
|
||||
pub fn call(h: Handle, msg: []const u8, reply: []u8) CallError!usize {
|
||||
const r = sc.syscall5(.ipc_call, h, @intFromPtr(msg.ptr), msg.len, @intFromPtr(reply.ptr), reply.len);
|
||||
return if (failed(r)) error.Failed else r;
|
||||
}
|
||||
|
||||
// replyWait() (server side — returns message length + sender badge) lands with
|
||||
// the first server binary (M9), where it can be exercised end to end.
|
||||
|
||||
@@ -79,6 +79,14 @@ pub fn setSyscallResult(state: *CpuState, value: u64) void {
|
||||
state.rax = value;
|
||||
}
|
||||
|
||||
/// Write a *second* syscall return value (rdx here — restored by both the
|
||||
/// syscall/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by
|
||||
/// sysretq). Used by IPC_ReplyWait to hand back the sender's badge alongside the
|
||||
/// message length in rax.
|
||||
pub fn setSyscallResult2(state: *CpuState, value: u64) void {
|
||||
state.rdx = value;
|
||||
}
|
||||
|
||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
||||
/// so it can be the very first thing called.
|
||||
pub fn serialInit() void {
|
||||
|
||||
@@ -0,0 +1,277 @@
|
||||
//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a
|
||||
//! rendezvous point; a client `call`s it (send a message, block for a reply) and
|
||||
//! a server `replyWait`s on it (reply to the last client, then block for the next
|
||||
//! request). This is the substrate the user-space VFS server and device drivers
|
||||
//! are reached through — `open`/`read`/`write` become user-space wrappers that
|
||||
//! marshal a request into a `call`.
|
||||
//!
|
||||
//! Design (see docs/syscall.md, the plan):
|
||||
//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame
|
||||
//! through the physmap (`copyAcross`), which is mapped in every address space's
|
||||
//! shared kernel half — so the kernel reads/writes either process's user memory
|
||||
//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing.
|
||||
//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply
|
||||
//! to exactly one client at a time; that caller is held in `Task.ipc_client`.
|
||||
//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without
|
||||
//! becoming runnable*, which a WaitQueue can't express, so callers queue on the
|
||||
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
||||
//! waiting for work use a normal WaitQueue.
|
||||
//!
|
||||
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
|
||||
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
|
||||
//! copy is a later security-track item, matching the existing debug_write gap.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
const sched = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const Task = sched.Task;
|
||||
|
||||
/// Largest message a single call/reply may carry. Bumping it is trivial; kept
|
||||
/// small because the copy runs under the big kernel lock.
|
||||
pub const MSG_MAX: usize = 256;
|
||||
|
||||
pub const max_handles = sched.ipc_max_handles;
|
||||
pub const max_services = 8;
|
||||
|
||||
/// Errno-style failures, returned as `-value` in the syscall result register.
|
||||
pub const EBADF: i64 = 1; // bad handle
|
||||
pub const E2BIG: i64 = 2; // message exceeds MSG_MAX
|
||||
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
||||
pub const ENOENT: i64 = 4; // no such registered service
|
||||
pub const ENOSPC: i64 = 5; // handle table or registry full
|
||||
pub const ENOMEM: i64 = 6; // out of memory
|
||||
|
||||
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
|
||||
/// message from a client — there is no reply owed. The low bits carry the source
|
||||
/// (a GSI for IRQs). Used by `notifyFromIsr`/M10; the message path uses a plain
|
||||
/// task-id badge with this bit clear.
|
||||
pub const notify_badge_bit: u64 = 1 << 63;
|
||||
|
||||
/// End of the user (low) canonical half — user buffers must lie below it.
|
||||
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
|
||||
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||
pub const Endpoint = struct {
|
||||
refcount: u32 = 1,
|
||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||
sender_head: ?*Task = null,
|
||||
sender_tail: ?*Task = null,
|
||||
// Servers blocked in `replyWait` awaiting a request.
|
||||
recv_wq: sched.WaitQueue = .{},
|
||||
// Pending asynchronous notifications (badges), a small coalescing ring.
|
||||
notify_buf: [8]u64 = undefined,
|
||||
notify_head: u8 = 0,
|
||||
notify_tail: u8 = 0,
|
||||
};
|
||||
|
||||
pub fn createEndpoint() ?*Endpoint {
|
||||
const ep = heap.allocator().create(Endpoint) catch return null;
|
||||
ep.* = .{};
|
||||
return ep;
|
||||
}
|
||||
|
||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||
pub fn dropRef(ep: *Endpoint) void {
|
||||
if (ep.refcount > 1) {
|
||||
ep.refcount -= 1;
|
||||
} else {
|
||||
heap.allocator().destroy(ep);
|
||||
}
|
||||
}
|
||||
|
||||
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
||||
|
||||
fn enqueueSender(ep: *Endpoint, t: *Task) void {
|
||||
t.next = null;
|
||||
if (ep.sender_tail) |tail| tail.next = t else ep.sender_head = t;
|
||||
ep.sender_tail = t;
|
||||
}
|
||||
|
||||
fn dequeueSender(ep: *Endpoint) ?*Task {
|
||||
const t = ep.sender_head orelse return null;
|
||||
ep.sender_head = t.next;
|
||||
if (ep.sender_head == null) ep.sender_tail = null;
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
// --- cross-address-space copy ----------------------------------------------
|
||||
|
||||
/// Copy `len` bytes from `src_va` in address space `src_as` to `dst_va` in
|
||||
/// `dst_as`, walking each side's page tables through the physmap (no CR3 switch).
|
||||
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
||||
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
|
||||
/// unmapped or out of range. Handles page-straddling buffers.
|
||||
fn copyAcross(src_as: u64, src_va: u64, dst_as: u64, dst_va: u64, len: usize) bool {
|
||||
const src_root = if (src_as != 0) src_as else arch.kernelPageTable();
|
||||
const dst_root = if (dst_as != 0) dst_as else arch.kernelPageTable();
|
||||
if (src_as != 0 and (src_va >= user_half_end or src_va + len > user_half_end)) return false;
|
||||
if (dst_as != 0 and (dst_va >= user_half_end or dst_va + len > user_half_end)) return false;
|
||||
|
||||
var off: usize = 0;
|
||||
while (off < len) {
|
||||
const s = arch.translate(src_root, src_va + off) orelse return false;
|
||||
const d = arch.translate(dst_root, dst_va + off) orelse return false;
|
||||
const s_left = page_size - ((src_va + off) & (page_size - 1));
|
||||
const d_left = page_size - ((dst_va + off) & (page_size - 1));
|
||||
const n = @min(@min(s_left, d_left), len - off);
|
||||
const src: [*]const u8 = @ptrFromInt(danos.physToVirt(s));
|
||||
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(d));
|
||||
@memcpy(dst[0..n], src[0..n]);
|
||||
off += n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- the two IPC operations -------------------------------------------------
|
||||
|
||||
/// Client side of IPC_Call: send `[msg_ptr, msg_len)` to `ep` and block until a
|
||||
/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a
|
||||
/// negative errno. Runs as the current task.
|
||||
pub fn call(ep: *Endpoint, msg_ptr: u64, msg_len: u64, reply_ptr: u64, reply_cap: u64) i64 {
|
||||
if (msg_len > MSG_MAX or reply_cap > MSG_MAX) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const me = sched.cur();
|
||||
me.ipc_send_ptr = msg_ptr;
|
||||
me.ipc_send_len = msg_len;
|
||||
me.ipc_reply_ptr = reply_ptr;
|
||||
me.ipc_reply_cap = reply_cap;
|
||||
me.ipc_status = 0;
|
||||
|
||||
enqueueSender(ep, me); // join the FIFO, then...
|
||||
sched.wakeLocked(&ep.recv_wq); // ...wake a waiting server (no-op if none)
|
||||
sched.blockCurrentLocked(); // block until the reply readies us again
|
||||
|
||||
return me.ipc_status; // reply length or -errno, written by the replier
|
||||
}
|
||||
|
||||
/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client
|
||||
/// we currently owe (if any), then receive the next request into
|
||||
/// `[recv_ptr, recv_cap)`, blocking until one arrives. Writes the sender's badge
|
||||
/// to `out_badge` and returns the request length, or a negative errno. A pending
|
||||
/// notification is delivered ahead of client requests (length 0, badge with
|
||||
/// `notify_badge_bit` set, no reply owed).
|
||||
pub fn replyWait(ep: *Endpoint, reply_ptr: u64, reply_len: u64, recv_ptr: u64, recv_cap: u64, out_badge: *u64) i64 {
|
||||
if (reply_len > MSG_MAX or recv_cap > MSG_MAX) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const me = sched.cur();
|
||||
|
||||
// (1) Reply to the client we're still holding, if any.
|
||||
if (me.ipc_client) |client| {
|
||||
me.ipc_client = null;
|
||||
const n = @min(reply_len, client.ipc_reply_cap);
|
||||
if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
|
||||
client.ipc_status = @intCast(n);
|
||||
} else {
|
||||
client.ipc_status = -EFAULT;
|
||||
}
|
||||
sched.readyLocked(client); // its `call` now returns
|
||||
}
|
||||
|
||||
// (2) Receive the next request (or notification), blocking until one is ready.
|
||||
while (true) {
|
||||
if (popNotify(ep)) |badge| {
|
||||
out_badge.* = badge | notify_badge_bit;
|
||||
return 0; // notification: no payload, no reply owed
|
||||
}
|
||||
if (dequeueSender(ep)) |caller| {
|
||||
const n = @min(caller.ipc_send_len, recv_cap);
|
||||
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, recv_ptr, n)) {
|
||||
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
||||
sched.readyLocked(caller);
|
||||
continue;
|
||||
}
|
||||
me.ipc_client = caller; // remember who to reply to
|
||||
out_badge.* = caller.id;
|
||||
return @intCast(n);
|
||||
}
|
||||
sched.waitLocked(&ep.recv_wq); // nothing yet — sleep until woken, then retry
|
||||
}
|
||||
}
|
||||
|
||||
// --- asynchronous notification (for IRQ-as-message, M10) --------------------
|
||||
|
||||
fn popNotify(ep: *Endpoint) ?u64 {
|
||||
if (ep.notify_head == ep.notify_tail) return null;
|
||||
const badge = ep.notify_buf[ep.notify_head % ep.notify_buf.len];
|
||||
ep.notify_head +%= 1;
|
||||
return badge;
|
||||
}
|
||||
|
||||
/// Post an asynchronous notification carrying `badge` to `ep` and wake a waiting
|
||||
/// receiver. ISR-safe: takes the big kernel lock and releases it without touching
|
||||
/// the interrupt flag (the ISR's iretq restores it), exactly like the timer tick.
|
||||
/// A full ring drops the notification (the driver re-reads device state anyway).
|
||||
pub fn notifyFromIsr(ep: *Endpoint, badge: u64) void {
|
||||
_ = sync.enter();
|
||||
if (ep.notify_tail -% ep.notify_head < ep.notify_buf.len) {
|
||||
ep.notify_buf[ep.notify_tail % ep.notify_buf.len] = badge;
|
||||
ep.notify_tail +%= 1;
|
||||
}
|
||||
sched.wakeLocked(&ep.recv_wq);
|
||||
sync.leaveIsr();
|
||||
}
|
||||
|
||||
// --- per-process handle table + name registry -------------------------------
|
||||
|
||||
/// Install `ep` in task `t`'s handle table; returns the small-int handle or
|
||||
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
||||
pub fn installHandle(t: *Task, ep: *Endpoint) i64 {
|
||||
for (&t.handles, 0..) |*slot, i| {
|
||||
if (slot.* == null) {
|
||||
slot.* = @ptrCast(ep);
|
||||
return @intCast(i);
|
||||
}
|
||||
}
|
||||
return -ENOSPC;
|
||||
}
|
||||
|
||||
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
if (h >= t.handles.len) return null;
|
||||
const slot = t.handles[@intCast(h)] orelse return null;
|
||||
return @ptrCast(@alignCast(slot));
|
||||
}
|
||||
|
||||
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
||||
/// exit path so a dead server's endpoints don't linger referenced.
|
||||
pub fn closeHandles(t: *Task) void {
|
||||
for (&t.handles) |*slot| {
|
||||
if (slot.*) |p| {
|
||||
dropRef(@ptrCast(@alignCast(p)));
|
||||
slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var registry: [max_services]?*Endpoint = .{null} ** max_services;
|
||||
|
||||
/// Publish `ep` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||
pub fn register(id: u32, ep: *Endpoint) i64 {
|
||||
if (id >= max_services) return -ENOENT;
|
||||
if (registry[id]) |old| dropRef(old);
|
||||
ep.refcount += 1;
|
||||
registry[id] = ep;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, taking a reference for the caller to
|
||||
/// install in its handle table. Null if nothing is registered there.
|
||||
pub fn lookup(id: u32) ?*Endpoint {
|
||||
if (id >= max_services) return null;
|
||||
const ep = registry[id] orelse return null;
|
||||
ep.refcount += 1;
|
||||
return ep;
|
||||
}
|
||||
+70
-3
@@ -26,6 +26,7 @@ const arch = @import("arch");
|
||||
const pmm = @import("pmm.zig");
|
||||
const sched = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const ipc = @import("ipc_sync.zig");
|
||||
const log = @import("log.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
@@ -89,9 +90,13 @@ fn syscall(state: *arch.CpuState) void {
|
||||
switch (@as(Syscall, @enumFromInt(arch.syscallNumber(state)))) {
|
||||
.exit => {
|
||||
exit_code = arch.syscallArg(state, 0);
|
||||
// A scheduled process frees its address space and reschedules; a
|
||||
// borrowed test thread unwinds back to the kernel that entered it.
|
||||
if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit();
|
||||
// A scheduled process drops its endpoint references, frees its address
|
||||
// space, and reschedules; a borrowed test thread unwinds back to the
|
||||
// kernel that entered it.
|
||||
if (sched.currentIsUserProcess()) {
|
||||
ipc.closeHandles(sched.cur());
|
||||
sched.exitUser();
|
||||
} else arch.userExit();
|
||||
},
|
||||
.yield => {
|
||||
sched.yield();
|
||||
@@ -104,10 +109,72 @@ fn syscall(state: *arch.CpuState) void {
|
||||
.debug_write => sysDebugWrite(state),
|
||||
.mmap => sysMmap(state),
|
||||
.munmap => sysMunmap(state),
|
||||
.create_endpoint => sysCreateEndpoint(state),
|
||||
.ipc_register => sysIpcRegister(state),
|
||||
.ipc_lookup => sysIpcLookup(state),
|
||||
.ipc_call => sysIpcCall(state),
|
||||
.ipc_reply_wait => sysIpcReplyWait(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
|
||||
/// Return `-errno` in the syscall result register.
|
||||
fn failErr(state: *arch.CpuState, errno: i64) void {
|
||||
arch.setSyscallResult(state, @bitCast(-errno));
|
||||
}
|
||||
|
||||
/// create_endpoint() -> handle: allocate an endpoint and install it in the
|
||||
/// caller's handle table.
|
||||
fn sysCreateEndpoint(state: *arch.CpuState) void {
|
||||
const ep = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM);
|
||||
const h = ipc.installHandle(sched.cur(), ep);
|
||||
if (h < 0) {
|
||||
ipc.dropRef(ep);
|
||||
return failErr(state, ipc.ENOSPC);
|
||||
}
|
||||
arch.setSyscallResult(state, @intCast(h));
|
||||
}
|
||||
|
||||
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
||||
/// well-known id so other processes can find it.
|
||||
fn sysIpcRegister(state: *arch.CpuState) void {
|
||||
const id: u32 = @truncate(arch.syscallArg(state, 0));
|
||||
const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||
arch.setSyscallResult(state, @bitCast(ipc.register(id, ep)));
|
||||
}
|
||||
|
||||
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
||||
/// handle to it in the caller.
|
||||
fn sysIpcLookup(state: *arch.CpuState) void {
|
||||
const id: u32 = @truncate(arch.syscallArg(state, 0));
|
||||
const ep = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
||||
const h = ipc.installHandle(sched.cur(), ep);
|
||||
if (h < 0) {
|
||||
ipc.dropRef(ep);
|
||||
return failErr(state, ipc.ENOSPC);
|
||||
}
|
||||
arch.setSyscallResult(state, @intCast(h));
|
||||
}
|
||||
|
||||
/// ipc_call(handle, msg_ptr, msg_len, reply_ptr, reply_cap) -> reply_len.
|
||||
/// Blocks until the server replies; the trap frame lives on this task's kernel
|
||||
/// stack, so it survives the block and receives the result on resume.
|
||||
fn sysIpcCall(state: *arch.CpuState) void {
|
||||
const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const r = ipc.call(ep, arch.syscallArg(state, 1), arch.syscallArg(state, 2), arch.syscallArg(state, 3), arch.syscallArg(state, 4));
|
||||
arch.setSyscallResult(state, @bitCast(r));
|
||||
}
|
||||
|
||||
/// ipc_reply_wait(handle, reply_ptr, reply_len, recv_ptr, recv_cap) -> recv_len,
|
||||
/// with the sender's badge in the secondary result register (rdx).
|
||||
fn sysIpcReplyWait(state: *arch.CpuState) void {
|
||||
const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
var badge: u64 = 0;
|
||||
const r = ipc.replyWait(ep, arch.syscallArg(state, 1), arch.syscallArg(state, 2), arch.syscallArg(state, 3), arch.syscallArg(state, 4), &badge);
|
||||
arch.setSyscallResult(state, @bitCast(r));
|
||||
arch.setSyscallResult2(state, badge);
|
||||
}
|
||||
|
||||
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
|
||||
/// A bring-up diagnostic — real output goes through the VFS/console later.
|
||||
///
|
||||
|
||||
@@ -50,9 +50,26 @@ pub const Task = struct {
|
||||
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
||||
// as the user heap grows; user task only.
|
||||
heap_next: u64 = 0,
|
||||
next: ?*Task = null, // ready-queue link
|
||||
// --- synchronous IPC (ipc_sync.zig) ---
|
||||
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
||||
// opaque here so the scheduler and IPC modules don't import each other.
|
||||
handles: [ipc_max_handles]?*anyopaque = .{null} ** ipc_max_handles,
|
||||
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||
ipc_client: ?*Task = null,
|
||||
ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS)
|
||||
ipc_send_len: u64 = 0,
|
||||
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
|
||||
ipc_reply_cap: u64 = 0,
|
||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||
};
|
||||
|
||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
pub const ipc_max_handles = 16;
|
||||
|
||||
var tasks = [_]Task{.{}} ** max_tasks;
|
||||
var next_id: u32 = 1;
|
||||
|
||||
@@ -399,6 +416,23 @@ pub fn wakeLocked(wq: *WaitQueue) void {
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Block the current task and switch away, without putting it on any wait queue —
|
||||
/// the caller has already linked it wherever it belongs (e.g. an endpoint's sender
|
||||
/// FIFO). Precondition: the big kernel lock is held; still held on return (when the
|
||||
/// task is made ready again). The IPC layer's counterpart to `waitLocked`.
|
||||
pub fn blockCurrentLocked() void {
|
||||
cur().state = .blocked;
|
||||
schedule();
|
||||
}
|
||||
|
||||
/// Make a specific (currently blocked) task ready to run again. Precondition: the
|
||||
/// big kernel lock is held. Used by the IPC layer to wake a specific caller/server
|
||||
/// rather than "some waiter on a queue".
|
||||
pub fn readyLocked(t: *Task) void {
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Block on `wq` (a self-contained critical section).
|
||||
pub fn wait(wq: *WaitQueue) void {
|
||||
const flags = sync.enter();
|
||||
|
||||
@@ -17,6 +17,7 @@ const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const sched = @import("scheduler.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const ipcsync = @import("ipc_sync.zig");
|
||||
const process = @import("process.zig");
|
||||
|
||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
||||
@@ -73,6 +74,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
||||
eventTest();
|
||||
} else if (eql(case, "ipc")) {
|
||||
ipcTest();
|
||||
} else if (eql(case, "ipc-call")) {
|
||||
ipcCallTest();
|
||||
} else if (eql(case, "smp")) {
|
||||
smpTest();
|
||||
} else if (eql(case, "affinity")) {
|
||||
@@ -774,6 +777,65 @@ fn userMemTest() void {
|
||||
result();
|
||||
}
|
||||
|
||||
// --- synchronous IPC --------------------------------------------------------
|
||||
|
||||
var ipc_ep: *ipcsync.Endpoint = undefined;
|
||||
var ipc_replies_ok: bool = false;
|
||||
var ipc_done: bool = false;
|
||||
|
||||
/// Echo-increment server: reply to each request with request+1, forever.
|
||||
fn ipcServer() void {
|
||||
var reply_buf: [8]u8 = undefined;
|
||||
var reply_len: u64 = 0;
|
||||
var badge: u64 = 0;
|
||||
while (true) {
|
||||
var recv: [8]u8 = undefined;
|
||||
const n = ipcsync.replyWait(ipc_ep, @intFromPtr(&reply_buf), reply_len, @intFromPtr(&recv), recv.len, &badge);
|
||||
if (n < 0) sched.exit();
|
||||
const v = std.mem.readInt(u64, recv[0..8], .little);
|
||||
std.mem.writeInt(u64, reply_buf[0..8], v + 1, .little);
|
||||
reply_len = 8;
|
||||
}
|
||||
}
|
||||
|
||||
/// Client: make 100 synchronous calls, checking every reply is request+1.
|
||||
fn ipcClient() void {
|
||||
var ok = true;
|
||||
var i: u64 = 0;
|
||||
while (i < 100) : (i += 1) {
|
||||
var msg: [8]u8 = undefined;
|
||||
std.mem.writeInt(u64, msg[0..8], i, .little);
|
||||
var reply: [8]u8 = undefined;
|
||||
const n = ipcsync.call(ipc_ep, @intFromPtr(&msg), 8, @intFromPtr(&reply), reply.len);
|
||||
if (n != 8 or std.mem.readInt(u64, reply[0..8], .little) != i + 1) ok = false;
|
||||
}
|
||||
ipc_replies_ok = ok;
|
||||
ipc_done = true;
|
||||
sched.exit();
|
||||
}
|
||||
|
||||
/// Synchronous IPC: a client and a server (two kernel tasks) ping-pong 100 calls
|
||||
/// through one Endpoint. Each round exercises the full rendezvous — the client
|
||||
/// blocks in `call`, the server blocks in `replyWait`, the reply is routed back to
|
||||
/// the exact caller, and `copyAcross` moves the bytes — so 100 correct replies
|
||||
/// prove the block/wake and reply-routing paths. (Kernel tasks, so no user ELF.)
|
||||
fn ipcCallTest() void {
|
||||
log("DANOS-TEST-BEGIN: ipc-call\n", .{});
|
||||
ipc_ep = ipcsync.createEndpoint().?;
|
||||
ipc_replies_ok = false;
|
||||
ipc_done = false;
|
||||
sched.spawn(ipcServer, 5); // above this task, so the workers run
|
||||
sched.spawn(ipcClient, 5);
|
||||
|
||||
const done: *volatile bool = &ipc_done;
|
||||
var spins: u64 = 0;
|
||||
while (!done.* and spins < 100_000_000) : (spins += 1) sched.yield();
|
||||
|
||||
check("client completed 100 synchronous calls", ipc_done);
|
||||
check("every reply was request+1 (rendezvous + reply routing intact)", ipc_replies_ok);
|
||||
result();
|
||||
}
|
||||
|
||||
var proc_worker_run: bool = true;
|
||||
var proc_worker_ran: bool = false;
|
||||
|
||||
|
||||
@@ -70,6 +70,19 @@ pub const Syscall = enum(u64) {
|
||||
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||
create_endpoint = 6, // create_endpoint() -> handle: a new IPC endpoint
|
||||
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||
ipc_call = 9, // ipc_call(h, msg, len, reply, cap) -> reply_len: send + block for reply
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, recv, cap) -> recv_len (+badge in rdx)
|
||||
_,
|
||||
};
|
||||
|
||||
/// Well-known IPC service ids for the bootstrap name registry (create_endpoint +
|
||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||
pub const ServiceId = enum(u32) {
|
||||
vfs = 1,
|
||||
_,
|
||||
};
|
||||
|
||||
|
||||
@@ -116,6 +116,11 @@ CASES = [
|
||||
{"name": "ipc",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Synchronous IPC: a client and server ping-pong 100 calls through an
|
||||
# endpoint (rendezvous, reply routing, cross-address-space copy).
|
||||
{"name": "ipc-call",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Parallelism: needs more than one core, so this case boots with -smp 4.
|
||||
{"name": "smp",
|
||||
"smp": 4,
|
||||
|
||||
Reference in New Issue
Block a user