//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a //! rendezvous point; a client `call`s it (send a message, block for a reply) and //! a server `replyWait`s on it (reply to the last client, then block for the next //! request). This is the substrate the user-space VFS server and device drivers //! are reached through — `open`/`read`/`write` become user-space wrappers that //! marshal a request into a `call`. //! //! Design (see docs/syscall.md, the plan): //! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame //! through the physmap (`copyAcross`), which is mapped in every address space's //! shared kernel half — so the kernel reads/writes either process's user memory //! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing. //! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply //! to exactly one client at a time; that caller is held in `Task.ipc_client`. //! - **Sender FIFO on the endpoint.** A blocked caller must be *received without //! becoming runnable*, which a WaitQueue can't express, so callers queue on the //! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers //! waiting for work use a normal WaitQueue. //! //! Trust model (bring-up): copies honour only page presence and a user-half bound, //! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked //! copy is a later security-track item, matching the existing debug_write gap. const std = @import("std"); const boot_handoff = @import("boot-handoff"); const abi = @import("abi"); const architecture = @import("architecture"); const scheduler = @import("scheduler.zig"); const sync = @import("sync.zig"); const heap = @import("heap.zig"); const page_size = abi.page_size; const Task = scheduler.Task; /// Largest message a single call/reply may carry. Bumping it is trivial; kept /// small because the copy runs under the big kernel lock. pub const MESSAGE_MAXIMUM: usize = 256; pub const maximum_handles = scheduler.ipc_maximum_handles; pub const maximum_services = 8; /// Errno-style failures, returned as `-value` in the system_call result register. pub const EBADF: i64 = 1; // bad handle pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half pub const ENOENT: i64 = 4; // no such registered service pub const ENOSPC: i64 = 5; // handle table or registry full pub const ENOMEM: i64 = 6; // out of memory pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed) pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id) pub const EPERM: i64 = 9; // not permitted (process_kill by anyone but the supervisor) /// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a /// message from a client — there is no reply owed. The low bits carry the source /// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in system/kernel/irq.zig; /// the message path uses a plain task-id badge with this bit clear. Defined in the /// shared kernel↔user ABI (system/abi.zig), because ring 3 has to test the same bit. pub const notify_badge_bit: u64 = abi.notify_badge_bit; /// Set (with `notify_badge_bit`) when a `replyWait` wake carries a buffered payload /// posted by `send` (`ipc_send`), rather than a bare IRQ/exit notification. Shared with /// ring 3 through the ABI so the receiver can tell "a message arrived" from "the hardware /// spoke". pub const notify_message_bit: u64 = abi.notify_message_bit; /// Largest payload a single `send` (`ipc_send`) may post. Kept small — the payload rides /// inline in every `Endpoint`, and the async path is for events (a `KeyEvent` is 16 /// bytes), not bulk transfer, which is what `call` and future shared pages are for. pub const POST_MAXIMUM: usize = 64; /// Depth of an endpoint's async payload ring. Absorbs a burst while a receiver is briefly /// busy; a full ring drops the *oldest* message (see `send`). const post_capacity: usize = 16; /// One buffered message: a length-prefixed payload plus the sender's task id (delivered /// in the low bits of the receiver's badge). const PostSlot = struct { length: u16 = 0, sender_id: u64 = 0, bytes: [POST_MAXIMUM]u8 = undefined, }; /// End of the user (low) canonical half — user buffers must lie below it. const user_half_end: u64 = 0x0000_8000_0000_0000; /// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle /// (per process) and/or by a registry slot, counted by `refcount`. pub const Endpoint = struct { refcount: u32 = 1, // Callers blocked in `call`, awaiting receive, in FIFO order (threaded via // Task.next; each such task is .blocked and in no scheduler queue). sender_head: ?*Task = null, sender_tail: ?*Task = null, // Servers blocked in `replyWait` awaiting a request. receive_wait_queue: scheduler.WaitQueue = .{}, // Pending asynchronous notifications (badges), a small coalescing ring. notify_buffer: [8]u64 = undefined, notify_head: u8 = 0, notify_tail: u8 = 0, // Pending buffered messages (payloads posted by `send`), a small FIFO ring. Unlike // notifications — which are a level and coalesce — these are discrete messages, so a // full ring drops the oldest rather than merging. post_buffer: [post_capacity]PostSlot = undefined, post_head: u16 = 0, post_tail: u16 = 0, }; pub fn createIpcEndpoint() ?*Endpoint { const endpoint = heap.allocator().create(Endpoint) catch return null; endpoint.* = .{}; return endpoint; } /// Drop a reference; free the endpoint when the last one goes. (Frames are leaked /// today like other kernel objects — but the refcount bookkeeping lands now.) pub fn dropRef(endpoint: *Endpoint) void { if (endpoint.refcount > 1) { endpoint.refcount -= 1; } else { heap.allocator().destroy(endpoint); } } // --- sender FIFO (endpoint-local, via Task.next) ---------------------------- fn enqueueSender(endpoint: *Endpoint, t: *Task) void { t.ipc_wait_endpoint = @ptrCast(endpoint); // so a kill can unlink a parked caller t.next = null; if (endpoint.sender_tail) |tail| tail.next = t else endpoint.sender_head = t; endpoint.sender_tail = t; } fn dequeueSender(endpoint: *Endpoint) ?*Task { const t = endpoint.sender_head orelse return null; endpoint.sender_head = t.next; if (endpoint.sender_head == null) endpoint.sender_tail = null; t.ipc_wait_endpoint = null; t.next = null; return t; } /// Unlink `t` from the sender FIFO it queues in, if any — the kill path for a /// client parked in `call` that no server has received yet. Without this, a dead /// caller would later be dequeued as a dangling pointer. The endpoint is still /// alive here: `t`'s own handle table holds a reference until closeHandles runs /// (which the kill path does *after* this). Precondition: the big kernel lock is /// held. pub fn abandonSenderLocked(t: *Task) void { const endpoint: *Endpoint = @ptrCast(@alignCast(t.ipc_wait_endpoint orelse return)); t.ipc_wait_endpoint = null; var previous: ?*Task = null; var node = endpoint.sender_head; while (node) |n| : ({ previous = n; node = n.next; }) { if (n != t) continue; if (previous) |p| p.next = t.next else endpoint.sender_head = t.next; if (endpoint.sender_tail == t) endpoint.sender_tail = previous; t.next = null; return; } } // --- cross-address-space copy ---------------------------------------------- /// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in /// `destination_as`, walking each side's page tables through the physmap (no CR3 switch). /// `*_as == 0` means the kernel address space (for kernel-task endpoints). User /// buffers must lie in the low half. Returns false — never #PFs — if any page is /// unmapped or out of range. Handles page-straddling buffers. fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool { const source_root = if (source_as != 0) source_as else architecture.kernelPageTable(); const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable(); if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false; if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false; var off: usize = 0; while (off < len) { const s = architecture.translate(source_root, source_va + off) orelse return false; const d = architecture.translate(destination_root, destination_va + off) orelse return false; const s_left = page_size - ((source_va + off) & (page_size - 1)); const d_left = page_size - ((destination_va + off) & (page_size - 1)); const n = @min(@min(s_left, d_left), len - off); const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(s)); const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(d)); @memcpy(destination[0..n], source[0..n]); off += n; } return true; } /// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel /// buffer `destination`, walking the user page tables through the physmap. Returns false if /// the range escapes the user half or any source page is unmapped — so a bad user /// pointer *fails the system_call* rather than faulting the kernel (danos has no /// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt /// the machine). The correct way to pull a fixed-size struct in from user space, and /// a single fetch: no TOCTOU against a hostile pointer. pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool { if (user_as == 0) return false; // not a user address space if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false; var off: usize = 0; while (off < destination.len) { const s = architecture.translate(user_as, user_va + off) orelse return false; const s_left = page_size - ((user_va + off) & (page_size - 1)); const n = @min(s_left, destination.len - off); const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(s)); @memcpy(destination[off..][0..n], source[0..n]); off += n; } return true; } // --- the two IPC operations ------------------------------------------------- /// Share the capability named by handle `cap` in `from`'s table into `to`'s table, /// bumping the endpoint's refcount (the sender keeps its handle — this is a copy, not /// a move). Returns the handle it landed at in `to` (>= 0), or `-EBADF` if `cap` names /// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when /// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message. fn shareCapability(from: *Task, to: *Task, cap: u64) i64 { const endpoint = resolveHandle(from, cap) orelse return -EBADF; endpoint.refcount += 1; const handle = installHandle(to, endpoint); if (handle < 0) { dropRef(endpoint); // undo the bump; the receiver had no room return -ENOSPC; } return handle; } /// Client side of IPC_Call: send `[message_ptr, message_len)` to `endpoint` and block until a /// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a /// negative errno. `send_cap` (a handle, or `no_cap`) is an endpoint transferred to the /// server with the request; `out_received_cap` receives the handle of an endpoint the /// server sent back in its reply, or `no_cap`. Runs as the current task. pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64, send_cap: u64, out_received_cap: *u64) i64 { if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG; const flags = sync.enter(); defer sync.leave(flags); const me = scheduler.current(); me.ipc_send_ptr = message_ptr; me.ipc_send_len = message_len; me.ipc_reply_ptr = reply_ptr; me.ipc_reply_cap = reply_cap; me.ipc_send_cap = send_cap; me.ipc_received_cap = abi.no_cap; me.ipc_status = 0; enqueueSender(endpoint, me); // join the FIFO, then... scheduler.wakeLocked(&endpoint.receive_wait_queue); // ...wake a waiting server (no-op if none) scheduler.blockCurrentLocked(); // block until the reply readies us again out_received_cap.* = me.ipc_received_cap; // a capability the replier sent back, or no_cap return me.ipc_status; // reply length or -errno, written by the replier } /// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client /// we currently owe (if any), then receive the next request into /// `[receive_ptr, receive_cap)`, blocking until one arrives. Writes the sender's badge /// to `out_badge` and returns the request length, or a negative errno. A pending /// notification is delivered ahead of client requests (length 0, badge with /// `notify_badge_bit` set, no reply owed). pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, send_cap: u64, out_badge: *u64, out_received_cap: *u64) i64 { if (reply_len > MESSAGE_MAXIMUM or receive_cap > MESSAGE_MAXIMUM) return -E2BIG; const flags = sync.enter(); defer sync.leave(flags); const me = scheduler.current(); out_received_cap.* = abi.no_cap; // no capability received unless a request delivers one // (1) Reply to the client we're still holding, if any — carrying `send_cap` to it. if (me.ipc_client) |client| { me.ipc_client = null; const n = @min(reply_len, client.ipc_reply_cap); client.ipc_received_cap = abi.no_cap; if (!copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) { client.ipc_status = -EFAULT; } else if (send_cap != abi.no_cap) { // Transfer the reply's capability into the client. A failure fails the // client's `call` rather than delivering a reply without its promised cap. const shared = shareCapability(me, client, send_cap); if (shared < 0) { client.ipc_status = shared; // -EBADF (bad handle) or -ENOSPC (client table full) } else { client.ipc_received_cap = @intCast(shared); client.ipc_status = @intCast(n); } } else { client.ipc_status = @intCast(n); } scheduler.readyLocked(client); // its `call` now returns } // (2) Receive the next request (or notification / buffered message), blocking until // one is ready. Bare notifications (IRQ/exit) come first — they're latency-sensitive // and carry no payload — then buffered messages, then synchronous client requests. while (true) { if (popNotify(endpoint)) |badge| { out_badge.* = badge | notify_badge_bit; return 0; // notification: no payload, no reply owed, no cap } if (popPost(endpoint)) |slot| { const n = @min(@as(usize, slot.length), receive_cap); // Copy from the kernel-resident ring slot (source aspace 0) into the receiver. if (!copyAcross(0, @intFromPtr(&slot.bytes), me.aspace, receive_ptr, n)) { continue; // bad receive buffer: drop this message, keep serving } out_badge.* = slot.sender_id | notify_badge_bit | notify_message_bit; return @intCast(n); // async message: payload delivered, no reply owed, no cap } if (dequeueSender(endpoint)) |caller| { const n = @min(caller.ipc_send_len, receive_cap); if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) { caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving scheduler.readyLocked(caller); continue; } // Install the capability the caller sent, if any, into my table. A failure // fails the caller's `call` and does not deliver — no half-delivered cap. if (caller.ipc_send_cap != abi.no_cap) { const shared = shareCapability(caller, me, caller.ipc_send_cap); if (shared < 0) { caller.ipc_status = shared; // -EBADF or -ENOSPC scheduler.readyLocked(caller); continue; } out_received_cap.* = @intCast(shared); } me.ipc_client = caller; // remember who to reply to out_badge.* = caller.id; return @intCast(n); } scheduler.waitLocked(&endpoint.receive_wait_queue); // nothing yet — sleep until woken, then retry } } // --- asynchronous notification (for IRQ-as-message, M10) -------------------- fn popNotify(endpoint: *Endpoint) ?u64 { if (endpoint.notify_head == endpoint.notify_tail) return null; const badge = endpoint.notify_buffer[endpoint.notify_head % endpoint.notify_buffer.len]; endpoint.notify_head +%= 1; return badge; } /// Take the oldest buffered message from the post ring, or null if empty. Returns a /// pointer into the endpoint's own storage — valid until the next `send`/`popPost` under /// the same lock region, which is all the copy-out in `replyWait` needs. fn popPost(endpoint: *Endpoint) ?*const PostSlot { if (endpoint.post_head == endpoint.post_tail) return null; const slot = &endpoint.post_buffer[endpoint.post_head % post_capacity]; endpoint.post_head +%= 1; return slot; } /// Client-free side of async IPC (`ipc_send`): copy `[source_va, len)` from address space /// `source_as` into `endpoint`'s post ring and wake a waiting receiver — **without /// blocking the sender** and with no reply owed. `sender_id` rides along, delivered in the /// low bits of the receiver's badge. Returns 0, or a negative errno (`-E2BIG` if the /// payload exceeds `POST_MAXIMUM`, `-EFAULT` if the source buffer is unmapped / out of the /// user half). A full ring drops the *oldest* message (advancing `post_head`), because a /// buffered message is discrete, not a level: keeping the newest keeps input responsive. /// Precondition: the big kernel lock is held. pub fn sendLocked(endpoint: *Endpoint, source_as: u64, source_va: u64, len: u64, sender_id: u64) i64 { if (len > POST_MAXIMUM) return -E2BIG; // Drop the oldest if the ring is full, so this newest message always lands. if (endpoint.post_tail -% endpoint.post_head >= post_capacity) endpoint.post_head +%= 1; const slot = &endpoint.post_buffer[endpoint.post_tail % post_capacity]; if (!copyFromUser(source_as, source_va, slot.bytes[0..@intCast(len)])) return -EFAULT; slot.length = @intCast(len); slot.sender_id = sender_id; endpoint.post_tail +%= 1; scheduler.wakeLocked(&endpoint.receive_wait_queue); return 0; } /// `sendLocked` wrapped in its own critical section, for the `ipc_send` syscall path. pub fn send(endpoint: *Endpoint, source_as: u64, source_va: u64, len: u64, sender_id: u64) i64 { const flags = sync.enter(); defer sync.leave(flags); return sendLocked(endpoint, source_as, source_va, len, sender_id); } /// Post an asynchronous notification carrying `badge` to `endpoint` and wake a waiting /// receiver. Precondition: the big kernel lock is held. /// /// The lock must already cover whatever produced `endpoint` — an ISR that looked the /// endpoint up in a table and *then* took the lock could be racing a process exit /// that unbinds and frees it in between. See irq.dispatch, which holds one lock /// region across the table read and this call. /// /// A full ring drops the notification. That is the correct semantics, not a /// concession: a notification is a *level* ("this device wants attention"), and the /// driver re-reads device state on wake. It is never a count of events. pub fn notifyLocked(endpoint: *Endpoint, badge: u64) void { if (endpoint.notify_tail -% endpoint.notify_head < endpoint.notify_buffer.len) { endpoint.notify_buffer[endpoint.notify_tail % endpoint.notify_buffer.len] = badge; endpoint.notify_tail +%= 1; } scheduler.wakeLocked(&endpoint.receive_wait_queue); } /// `notifyLocked` as a self-contained ISR critical section, for a caller that holds /// `endpoint` by some means other than a table the lock protects. Releases the lock without /// touching the interrupt flag (the ISR's iretq restores it), like the timer tick. pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void { _ = sync.enter(); notifyLocked(endpoint, badge); sync.leaveIsr(); } // --- per-process handle table + name registry ------------------------------- /// Install `endpoint` in task `t`'s handle table; returns the small-int handle or /// -ENOSPC. The caller has already taken/holds the reference the slot represents. pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 { for (&t.handles, 0..) |*slot, i| { if (slot.* == null) { slot.* = @ptrCast(endpoint); return @intCast(i); } } return -ENOSPC; } /// Resolve a handle to its endpoint, or null if out of range / unused. pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint { if (h >= t.handles.len) return null; const slot = t.handles[@intCast(h)] orelse return null; return @ptrCast(@alignCast(slot)); } /// Drop every endpoint reference an exiting task holds. Called from the scheduler /// exit path so a dead server's endpoints don't linger referenced. pub fn closeHandles(t: *Task) void { for (&t.handles) |*slot| { if (slot.*) |p| { dropRef(@ptrCast(@alignCast(p))); slot.* = null; } } } var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services; /// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno. pub fn register(id: u32, endpoint: *Endpoint) i64 { if (id >= maximum_services) return -ENOENT; if (registry[id]) |old| dropRef(old); endpoint.refcount += 1; registry[id] = endpoint; return 0; } /// Find the endpoint published under `id`, taking a reference for the caller to /// install in its handle table. Null if nothing is registered there. pub fn lookup(id: u32) ?*Endpoint { if (id >= maximum_services) return null; const endpoint = registry[id] orelse return null; endpoint.refcount += 1; return endpoint; }