From eabb81684adcea4c90176f0c6f3d53d9940aa86d Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Thu, 9 Jul 2026 07:20:57 +0100 Subject: [PATCH] =?UTF-8?q?M7:=20synchronous=20IPC=20=E2=80=94=20endpoints?= =?UTF-8?q?,=20handle=20table,=20IPC=5FCall/ReplyWait?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The microkernel message backbone the VFS server and drivers will ride on. - src/kernel/ipc_sync.zig: Endpoint (sender FIFO threaded via Task.next + a recv WaitQueue for servers + a small notification ring). call() (client sends, wakes a server, blocks) and replyWait() (server replies to the held caller, then receives the next). Reply routing keys on Task.ipc_client — synchronous IPC owes one reply at a time. Payloads copy frame-to-frame through the physmap (copyAcross on arch.translate, added in M5); an unmapped page fails the copy instead of #PF-ing. MSG_MAX 256. - Bootstrap naming: an integer name registry (danos.ServiceId, vfs=1) with create_endpoint / ipc_register / ipc_lookup — any process finds a server without threading a handle through spawn. - notifyFromIsr(): ISR-safe async wake (badge with the high bit set), the hook M10's IRQ-as-message needs. Unused/untested until then. - Task gains handles[16] (opaque *Endpoint, to avoid a sched<->ipc import cycle) + ipc_client/send/reply/status fields; sched gains blockCurrentLocked/readyLocked; arch gains setSyscallResult2 (rdx badge). - Syscalls 6..10 wired in process.zig; exit() now drops the caller's endpoint refs. lib/ipc.zig: user-side createEndpoint/register/lookup/call (server-side replyWait lands with the first server in M9). - New `ipc-call` test: two kernel tasks ping-pong 100 calls, every reply request+1. Suite 29/29. --- lib/ipc.zig | 53 ++++++- src/kernel/arch/x86_64/cpu.zig | 8 + src/kernel/ipc_sync.zig | 277 +++++++++++++++++++++++++++++++++ src/kernel/process.zig | 73 ++++++++- src/kernel/scheduler.zig | 36 ++++- src/kernel/tests.zig | 62 ++++++++ src/root.zig | 13 ++ test/qemu_test.py | 5 + 8 files changed, 516 insertions(+), 11 deletions(-) create mode 100644 src/kernel/ipc_sync.zig diff --git a/lib/ipc.zig b/lib/ipc.zig index 436ddbb..0262cbd 100644 --- a/lib/ipc.zig +++ b/lib/ipc.zig @@ -1,10 +1,16 @@ -//! User-space IPC helpers. The kernel's synchronous IPC syscalls -//! (create_endpoint, ipc_register/lookup, ipc_call, ipc_reply_wait) arrive in a -//! later milestone; this reserves the module boundary now so drivers and servers -//! can be written against `rt.ipc` without restructuring once they light up. +//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client +//! `call`s an endpoint (send + block for reply); the VFS server and drivers are +//! reached this way. The server side (`replyWait`, which returns two values) is +//! added with the first server binary. -/// A fixed-size, register-friendly message payload. The wire format for the VFS -/// and driver protocols is layered on top of this by the servers themselves. +const danos = @import("danos"); +const sc = @import("syscall.zig"); + +/// A small-int handle into the calling process's handle table. +pub const Handle = usize; + +/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver) +/// layer their own wire format on top of the bytes a call carries. pub const Message = extern struct { tag: u64 = 0, a: u64 = 0, @@ -12,4 +18,37 @@ pub const Message = extern struct { c: u64 = 0, }; -// call() / replyWait() / createEndpoint() land with the kernel IPC syscalls. +/// Whether a syscall return value is a wrapped -errno (lands in the top page). +inline fn failed(r: usize) bool { + return r > ~@as(usize, 0) - 4095; +} + +/// Create a new endpoint owned by this process; returns its handle. +pub fn createEndpoint() ?Handle { + const r = sc.syscall0(.create_endpoint); + return if (failed(r)) null else r; +} + +/// Publish endpoint `h` under a well-known service id so other processes find it. +pub fn register(id: danos.ServiceId, h: Handle) bool { + return !failed(sc.syscall2(.ipc_register, @intFromEnum(id), h)); +} + +/// Find the endpoint published under `id`, installing a handle to it in this +/// process. +pub fn lookup(id: danos.ServiceId) ?Handle { + const r = sc.syscall1(.ipc_lookup, @intFromEnum(id)); + return if (failed(r)) null else r; +} + +pub const CallError = error{Failed}; + +/// Send `msg` to endpoint `h` and block until the server replies into `reply`. +/// Returns the reply length. +pub fn call(h: Handle, msg: []const u8, reply: []u8) CallError!usize { + const r = sc.syscall5(.ipc_call, h, @intFromPtr(msg.ptr), msg.len, @intFromPtr(reply.ptr), reply.len); + return if (failed(r)) error.Failed else r; +} + +// replyWait() (server side — returns message length + sender badge) lands with +// the first server binary (M9), where it can be exercised end to end. diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index 6d52006..aeca0e3 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -79,6 +79,14 @@ pub fn setSyscallResult(state: *CpuState, value: u64) void { state.rax = value; } +/// Write a *second* syscall return value (rdx here — restored by both the +/// syscall/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by +/// sysretq). Used by IPC_ReplyWait to hand back the sender's badge alongside the +/// message length in rax. +pub fn setSyscallResult2(state: *CpuState, value: u64) void { + state.rdx = value; +} + /// Bring up the serial port (the kernel's machine-readable log). No dependencies, /// so it can be the very first thing called. pub fn serialInit() void { diff --git a/src/kernel/ipc_sync.zig b/src/kernel/ipc_sync.zig new file mode 100644 index 0000000..320c5be --- /dev/null +++ b/src/kernel/ipc_sync.zig @@ -0,0 +1,277 @@ +//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a +//! rendezvous point; a client `call`s it (send a message, block for a reply) and +//! a server `replyWait`s on it (reply to the last client, then block for the next +//! request). This is the substrate the user-space VFS server and device drivers +//! are reached through — `open`/`read`/`write` become user-space wrappers that +//! marshal a request into a `call`. +//! +//! Design (see docs/syscall.md, the plan): +//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame +//! through the physmap (`copyAcross`), which is mapped in every address space's +//! shared kernel half — so the kernel reads/writes either process's user memory +//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing. +//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply +//! to exactly one client at a time; that caller is held in `Task.ipc_client`. +//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without +//! becoming runnable*, which a WaitQueue can't express, so callers queue on the +//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers +//! waiting for work use a normal WaitQueue. +//! +//! Trust model (bring-up): copies honour only page presence and a user-half bound, +//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked +//! copy is a later security-track item, matching the existing debug_write gap. + +const std = @import("std"); +const danos = @import("danos"); +const arch = @import("arch"); +const sched = @import("scheduler.zig"); +const sync = @import("sync.zig"); +const heap = @import("heap.zig"); + +const page_size = danos.page_size; +const Task = sched.Task; + +/// Largest message a single call/reply may carry. Bumping it is trivial; kept +/// small because the copy runs under the big kernel lock. +pub const MSG_MAX: usize = 256; + +pub const max_handles = sched.ipc_max_handles; +pub const max_services = 8; + +/// Errno-style failures, returned as `-value` in the syscall result register. +pub const EBADF: i64 = 1; // bad handle +pub const E2BIG: i64 = 2; // message exceeds MSG_MAX +pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half +pub const ENOENT: i64 = 4; // no such registered service +pub const ENOSPC: i64 = 5; // handle table or registry full +pub const ENOMEM: i64 = 6; // out of memory + +/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a +/// message from a client — there is no reply owed. The low bits carry the source +/// (a GSI for IRQs). Used by `notifyFromIsr`/M10; the message path uses a plain +/// task-id badge with this bit clear. +pub const notify_badge_bit: u64 = 1 << 63; + +/// End of the user (low) canonical half — user buffers must lie below it. +const user_half_end: u64 = 0x0000_8000_0000_0000; + +/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle +/// (per process) and/or by a registry slot, counted by `refcount`. +pub const Endpoint = struct { + refcount: u32 = 1, + // Callers blocked in `call`, awaiting receive, in FIFO order (threaded via + // Task.next; each such task is .blocked and in no scheduler queue). + sender_head: ?*Task = null, + sender_tail: ?*Task = null, + // Servers blocked in `replyWait` awaiting a request. + recv_wq: sched.WaitQueue = .{}, + // Pending asynchronous notifications (badges), a small coalescing ring. + notify_buf: [8]u64 = undefined, + notify_head: u8 = 0, + notify_tail: u8 = 0, +}; + +pub fn createEndpoint() ?*Endpoint { + const ep = heap.allocator().create(Endpoint) catch return null; + ep.* = .{}; + return ep; +} + +/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked +/// today like other kernel objects — but the refcount bookkeeping lands now.) +pub fn dropRef(ep: *Endpoint) void { + if (ep.refcount > 1) { + ep.refcount -= 1; + } else { + heap.allocator().destroy(ep); + } +} + +// --- sender FIFO (endpoint-local, via Task.next) ---------------------------- + +fn enqueueSender(ep: *Endpoint, t: *Task) void { + t.next = null; + if (ep.sender_tail) |tail| tail.next = t else ep.sender_head = t; + ep.sender_tail = t; +} + +fn dequeueSender(ep: *Endpoint) ?*Task { + const t = ep.sender_head orelse return null; + ep.sender_head = t.next; + if (ep.sender_head == null) ep.sender_tail = null; + t.next = null; + return t; +} + +// --- cross-address-space copy ---------------------------------------------- + +/// Copy `len` bytes from `src_va` in address space `src_as` to `dst_va` in +/// `dst_as`, walking each side's page tables through the physmap (no CR3 switch). +/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User +/// buffers must lie in the low half. Returns false — never #PFs — if any page is +/// unmapped or out of range. Handles page-straddling buffers. +fn copyAcross(src_as: u64, src_va: u64, dst_as: u64, dst_va: u64, len: usize) bool { + const src_root = if (src_as != 0) src_as else arch.kernelPageTable(); + const dst_root = if (dst_as != 0) dst_as else arch.kernelPageTable(); + if (src_as != 0 and (src_va >= user_half_end or src_va + len > user_half_end)) return false; + if (dst_as != 0 and (dst_va >= user_half_end or dst_va + len > user_half_end)) return false; + + var off: usize = 0; + while (off < len) { + const s = arch.translate(src_root, src_va + off) orelse return false; + const d = arch.translate(dst_root, dst_va + off) orelse return false; + const s_left = page_size - ((src_va + off) & (page_size - 1)); + const d_left = page_size - ((dst_va + off) & (page_size - 1)); + const n = @min(@min(s_left, d_left), len - off); + const src: [*]const u8 = @ptrFromInt(danos.physToVirt(s)); + const dst: [*]u8 = @ptrFromInt(danos.physToVirt(d)); + @memcpy(dst[0..n], src[0..n]); + off += n; + } + return true; +} + +// --- the two IPC operations ------------------------------------------------- + +/// Client side of IPC_Call: send `[msg_ptr, msg_len)` to `ep` and block until a +/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a +/// negative errno. Runs as the current task. +pub fn call(ep: *Endpoint, msg_ptr: u64, msg_len: u64, reply_ptr: u64, reply_cap: u64) i64 { + if (msg_len > MSG_MAX or reply_cap > MSG_MAX) return -E2BIG; + const flags = sync.enter(); + defer sync.leave(flags); + + const me = sched.cur(); + me.ipc_send_ptr = msg_ptr; + me.ipc_send_len = msg_len; + me.ipc_reply_ptr = reply_ptr; + me.ipc_reply_cap = reply_cap; + me.ipc_status = 0; + + enqueueSender(ep, me); // join the FIFO, then... + sched.wakeLocked(&ep.recv_wq); // ...wake a waiting server (no-op if none) + sched.blockCurrentLocked(); // block until the reply readies us again + + return me.ipc_status; // reply length or -errno, written by the replier +} + +/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client +/// we currently owe (if any), then receive the next request into +/// `[recv_ptr, recv_cap)`, blocking until one arrives. Writes the sender's badge +/// to `out_badge` and returns the request length, or a negative errno. A pending +/// notification is delivered ahead of client requests (length 0, badge with +/// `notify_badge_bit` set, no reply owed). +pub fn replyWait(ep: *Endpoint, reply_ptr: u64, reply_len: u64, recv_ptr: u64, recv_cap: u64, out_badge: *u64) i64 { + if (reply_len > MSG_MAX or recv_cap > MSG_MAX) return -E2BIG; + const flags = sync.enter(); + defer sync.leave(flags); + + const me = sched.cur(); + + // (1) Reply to the client we're still holding, if any. + if (me.ipc_client) |client| { + me.ipc_client = null; + const n = @min(reply_len, client.ipc_reply_cap); + if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) { + client.ipc_status = @intCast(n); + } else { + client.ipc_status = -EFAULT; + } + sched.readyLocked(client); // its `call` now returns + } + + // (2) Receive the next request (or notification), blocking until one is ready. + while (true) { + if (popNotify(ep)) |badge| { + out_badge.* = badge | notify_badge_bit; + return 0; // notification: no payload, no reply owed + } + if (dequeueSender(ep)) |caller| { + const n = @min(caller.ipc_send_len, recv_cap); + if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, recv_ptr, n)) { + caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving + sched.readyLocked(caller); + continue; + } + me.ipc_client = caller; // remember who to reply to + out_badge.* = caller.id; + return @intCast(n); + } + sched.waitLocked(&ep.recv_wq); // nothing yet — sleep until woken, then retry + } +} + +// --- asynchronous notification (for IRQ-as-message, M10) -------------------- + +fn popNotify(ep: *Endpoint) ?u64 { + if (ep.notify_head == ep.notify_tail) return null; + const badge = ep.notify_buf[ep.notify_head % ep.notify_buf.len]; + ep.notify_head +%= 1; + return badge; +} + +/// Post an asynchronous notification carrying `badge` to `ep` and wake a waiting +/// receiver. ISR-safe: takes the big kernel lock and releases it without touching +/// the interrupt flag (the ISR's iretq restores it), exactly like the timer tick. +/// A full ring drops the notification (the driver re-reads device state anyway). +pub fn notifyFromIsr(ep: *Endpoint, badge: u64) void { + _ = sync.enter(); + if (ep.notify_tail -% ep.notify_head < ep.notify_buf.len) { + ep.notify_buf[ep.notify_tail % ep.notify_buf.len] = badge; + ep.notify_tail +%= 1; + } + sched.wakeLocked(&ep.recv_wq); + sync.leaveIsr(); +} + +// --- per-process handle table + name registry ------------------------------- + +/// Install `ep` in task `t`'s handle table; returns the small-int handle or +/// -ENOSPC. The caller has already taken/holds the reference the slot represents. +pub fn installHandle(t: *Task, ep: *Endpoint) i64 { + for (&t.handles, 0..) |*slot, i| { + if (slot.* == null) { + slot.* = @ptrCast(ep); + return @intCast(i); + } + } + return -ENOSPC; +} + +/// Resolve a handle to its endpoint, or null if out of range / unused. +pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint { + if (h >= t.handles.len) return null; + const slot = t.handles[@intCast(h)] orelse return null; + return @ptrCast(@alignCast(slot)); +} + +/// Drop every endpoint reference an exiting task holds. Called from the scheduler +/// exit path so a dead server's endpoints don't linger referenced. +pub fn closeHandles(t: *Task) void { + for (&t.handles) |*slot| { + if (slot.*) |p| { + dropRef(@ptrCast(@alignCast(p))); + slot.* = null; + } + } +} + +var registry: [max_services]?*Endpoint = .{null} ** max_services; + +/// Publish `ep` under well-known `id` (takes a reference). Returns 0 or -errno. +pub fn register(id: u32, ep: *Endpoint) i64 { + if (id >= max_services) return -ENOENT; + if (registry[id]) |old| dropRef(old); + ep.refcount += 1; + registry[id] = ep; + return 0; +} + +/// Find the endpoint published under `id`, taking a reference for the caller to +/// install in its handle table. Null if nothing is registered there. +pub fn lookup(id: u32) ?*Endpoint { + if (id >= max_services) return null; + const ep = registry[id] orelse return null; + ep.refcount += 1; + return ep; +} diff --git a/src/kernel/process.zig b/src/kernel/process.zig index 140c4f5..e5888e2 100644 --- a/src/kernel/process.zig +++ b/src/kernel/process.zig @@ -26,6 +26,7 @@ const arch = @import("arch"); const pmm = @import("pmm.zig"); const sched = @import("scheduler.zig"); const sync = @import("sync.zig"); +const ipc = @import("ipc_sync.zig"); const log = @import("log.zig"); const page_size = danos.page_size; @@ -89,9 +90,13 @@ fn syscall(state: *arch.CpuState) void { switch (@as(Syscall, @enumFromInt(arch.syscallNumber(state)))) { .exit => { exit_code = arch.syscallArg(state, 0); - // A scheduled process frees its address space and reschedules; a - // borrowed test thread unwinds back to the kernel that entered it. - if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit(); + // A scheduled process drops its endpoint references, frees its address + // space, and reschedules; a borrowed test thread unwinds back to the + // kernel that entered it. + if (sched.currentIsUserProcess()) { + ipc.closeHandles(sched.cur()); + sched.exitUser(); + } else arch.userExit(); }, .yield => { sched.yield(); @@ -104,10 +109,72 @@ fn syscall(state: *arch.CpuState) void { .debug_write => sysDebugWrite(state), .mmap => sysMmap(state), .munmap => sysMunmap(state), + .create_endpoint => sysCreateEndpoint(state), + .ipc_register => sysIpcRegister(state), + .ipc_lookup => sysIpcLookup(state), + .ipc_call => sysIpcCall(state), + .ipc_reply_wait => sysIpcReplyWait(state), _ => fail(state), } } +/// Return `-errno` in the syscall result register. +fn failErr(state: *arch.CpuState, errno: i64) void { + arch.setSyscallResult(state, @bitCast(-errno)); +} + +/// create_endpoint() -> handle: allocate an endpoint and install it in the +/// caller's handle table. +fn sysCreateEndpoint(state: *arch.CpuState) void { + const ep = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM); + const h = ipc.installHandle(sched.cur(), ep); + if (h < 0) { + ipc.dropRef(ep); + return failErr(state, ipc.ENOSPC); + } + arch.setSyscallResult(state, @intCast(h)); +} + +/// ipc_register(service_id, handle): publish the caller's endpoint under a +/// well-known id so other processes can find it. +fn sysIpcRegister(state: *arch.CpuState) void { + const id: u32 = @truncate(arch.syscallArg(state, 0)); + const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 1)) orelse return failErr(state, ipc.EBADF); + arch.setSyscallResult(state, @bitCast(ipc.register(id, ep))); +} + +/// ipc_lookup(service_id) -> handle: find a published endpoint and install a +/// handle to it in the caller. +fn sysIpcLookup(state: *arch.CpuState) void { + const id: u32 = @truncate(arch.syscallArg(state, 0)); + const ep = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT); + const h = ipc.installHandle(sched.cur(), ep); + if (h < 0) { + ipc.dropRef(ep); + return failErr(state, ipc.ENOSPC); + } + arch.setSyscallResult(state, @intCast(h)); +} + +/// ipc_call(handle, msg_ptr, msg_len, reply_ptr, reply_cap) -> reply_len. +/// Blocks until the server replies; the trap frame lives on this task's kernel +/// stack, so it survives the block and receives the result on resume. +fn sysIpcCall(state: *arch.CpuState) void { + const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 0)) orelse return failErr(state, ipc.EBADF); + const r = ipc.call(ep, arch.syscallArg(state, 1), arch.syscallArg(state, 2), arch.syscallArg(state, 3), arch.syscallArg(state, 4)); + arch.setSyscallResult(state, @bitCast(r)); +} + +/// ipc_reply_wait(handle, reply_ptr, reply_len, recv_ptr, recv_cap) -> recv_len, +/// with the sender's badge in the secondary result register (rdx). +fn sysIpcReplyWait(state: *arch.CpuState) void { + const ep = ipc.resolveHandle(sched.cur(), arch.syscallArg(state, 0)) orelse return failErr(state, ipc.EBADF); + var badge: u64 = 0; + const r = ipc.replyWait(ep, arch.syscallArg(state, 1), arch.syscallArg(state, 2), arch.syscallArg(state, 3), arch.syscallArg(state, 4), &badge); + arch.setSyscallResult(state, @bitCast(r)); + arch.setSyscallResult2(state, badge); +} + /// debug_write(ptr, len): copy bytes from user memory into the kernel log. /// A bring-up diagnostic — real output goes through the VFS/console later. /// diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 15fc47d..f219310 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -50,9 +50,26 @@ pub const Task = struct { // process.zig lazily seeds it to the arena base on the first mmap). Bumped up // as the user heap grows; user task only. heap_next: u64 = 0, - next: ?*Task = null, // ready-queue link + // --- synchronous IPC (ipc_sync.zig) --- + // Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept + // opaque here so the scheduler and IPC modules don't import each other. + handles: [ipc_max_handles]?*anyopaque = .{null} ** ipc_max_handles, + // A server holds the caller it currently owes a reply to (set by ReplyWait's + // receive, cleared when it replies). A client, while blocked in Call, records + // its message + reply buffers here and its result lands in `ipc_status`. + ipc_client: ?*Task = null, + ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS) + ipc_send_len: u64 = 0, + ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr) + ipc_reply_cap: u64 = 0, + ipc_status: i64 = 0, // client: reply length / -errno, written by the replier + next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link) }; +/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because +/// it dimensions a field of `Task`; ipc_sync.zig re-exports it. +pub const ipc_max_handles = 16; + var tasks = [_]Task{.{}} ** max_tasks; var next_id: u32 = 1; @@ -399,6 +416,23 @@ pub fn wakeLocked(wq: *WaitQueue) void { enqueue(t); } +/// Block the current task and switch away, without putting it on any wait queue — +/// the caller has already linked it wherever it belongs (e.g. an endpoint's sender +/// FIFO). Precondition: the big kernel lock is held; still held on return (when the +/// task is made ready again). The IPC layer's counterpart to `waitLocked`. +pub fn blockCurrentLocked() void { + cur().state = .blocked; + schedule(); +} + +/// Make a specific (currently blocked) task ready to run again. Precondition: the +/// big kernel lock is held. Used by the IPC layer to wake a specific caller/server +/// rather than "some waiter on a queue". +pub fn readyLocked(t: *Task) void { + t.state = .ready; + enqueue(t); +} + /// Block on `wq` (a self-contained critical section). pub fn wait(wq: *WaitQueue) void { const flags = sync.enter(); diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index e5836f6..7435bb1 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -17,6 +17,7 @@ const pmm = @import("pmm.zig"); const heap = @import("heap.zig"); const sched = @import("scheduler.zig"); const ipc = @import("ipc.zig"); +const ipcsync = @import("ipc_sync.zig"); const process = @import("process.zig"); /// Formatted write straight to serial, independent of the framebuffer console. @@ -73,6 +74,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { eventTest(); } else if (eql(case, "ipc")) { ipcTest(); + } else if (eql(case, "ipc-call")) { + ipcCallTest(); } else if (eql(case, "smp")) { smpTest(); } else if (eql(case, "affinity")) { @@ -774,6 +777,65 @@ fn userMemTest() void { result(); } +// --- synchronous IPC -------------------------------------------------------- + +var ipc_ep: *ipcsync.Endpoint = undefined; +var ipc_replies_ok: bool = false; +var ipc_done: bool = false; + +/// Echo-increment server: reply to each request with request+1, forever. +fn ipcServer() void { + var reply_buf: [8]u8 = undefined; + var reply_len: u64 = 0; + var badge: u64 = 0; + while (true) { + var recv: [8]u8 = undefined; + const n = ipcsync.replyWait(ipc_ep, @intFromPtr(&reply_buf), reply_len, @intFromPtr(&recv), recv.len, &badge); + if (n < 0) sched.exit(); + const v = std.mem.readInt(u64, recv[0..8], .little); + std.mem.writeInt(u64, reply_buf[0..8], v + 1, .little); + reply_len = 8; + } +} + +/// Client: make 100 synchronous calls, checking every reply is request+1. +fn ipcClient() void { + var ok = true; + var i: u64 = 0; + while (i < 100) : (i += 1) { + var msg: [8]u8 = undefined; + std.mem.writeInt(u64, msg[0..8], i, .little); + var reply: [8]u8 = undefined; + const n = ipcsync.call(ipc_ep, @intFromPtr(&msg), 8, @intFromPtr(&reply), reply.len); + if (n != 8 or std.mem.readInt(u64, reply[0..8], .little) != i + 1) ok = false; + } + ipc_replies_ok = ok; + ipc_done = true; + sched.exit(); +} + +/// Synchronous IPC: a client and a server (two kernel tasks) ping-pong 100 calls +/// through one Endpoint. Each round exercises the full rendezvous — the client +/// blocks in `call`, the server blocks in `replyWait`, the reply is routed back to +/// the exact caller, and `copyAcross` moves the bytes — so 100 correct replies +/// prove the block/wake and reply-routing paths. (Kernel tasks, so no user ELF.) +fn ipcCallTest() void { + log("DANOS-TEST-BEGIN: ipc-call\n", .{}); + ipc_ep = ipcsync.createEndpoint().?; + ipc_replies_ok = false; + ipc_done = false; + sched.spawn(ipcServer, 5); // above this task, so the workers run + sched.spawn(ipcClient, 5); + + const done: *volatile bool = &ipc_done; + var spins: u64 = 0; + while (!done.* and spins < 100_000_000) : (spins += 1) sched.yield(); + + check("client completed 100 synchronous calls", ipc_done); + check("every reply was request+1 (rendezvous + reply routing intact)", ipc_replies_ok); + result(); +} + var proc_worker_run: bool = true; var proc_worker_ran: bool = false; diff --git a/src/root.zig b/src/root.zig index b328ba4..ff52b4c 100644 --- a/src/root.zig +++ b/src/root.zig @@ -70,6 +70,19 @@ pub const Syscall = enum(u64) { sleep = 3, // sleep(ms): block the caller for ms milliseconds mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages munmap = 5, // munmap(base, len): release pages from a prior mmap + create_endpoint = 6, // create_endpoint() -> handle: a new IPC endpoint + ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id + ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint + ipc_call = 9, // ipc_call(h, msg, len, reply, cap) -> reply_len: send + block for reply + ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, recv, cap) -> recv_len (+badge in rdx) + _, +}; + +/// Well-known IPC service ids for the bootstrap name registry (create_endpoint + +/// ipc_register/ipc_lookup). Small integers, so no string interning is needed +/// during bring-up. The VFS server registers under `vfs`; clients look it up. +pub const ServiceId = enum(u32) { + vfs = 1, _, }; diff --git a/test/qemu_test.py b/test/qemu_test.py index 5eec46f..42c30b6 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -116,6 +116,11 @@ CASES = [ {"name": "ipc", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Synchronous IPC: a client and server ping-pong 100 calls through an + # endpoint (rendezvous, reply routing, cross-address-space copy). + {"name": "ipc-call", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # Parallelism: needs more than one core, so this case boots with -smp 4. {"name": "smp", "smp": 4,