From a581712b095ef18b6f73fae700dfc4b80366a761 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Fri, 10 Jul 2026 19:23:19 +0100 Subject: [PATCH] M13: IPC capability passing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ipc_call and ipc_reply_wait grow a `send_cap` argument (r9) and a `received_cap` return (r8): an endpoint travels alongside a message, installed into the receiver's handle table. The transfer is a share, not a move — the endpoint's refcount is bumped and the sender keeps its handle. If the receiver's table is full the call fails -ENOSPC and the message is NOT delivered (a half-delivered capability is worse than a failed send); a bad handle fails -EBADF. Both directions carry a cap: a client's call hands one to the server (seen in the server's replyWait), and the server's reply hands one back (seen in the client's call return). This is the "open" primitive the driver model was blocked on: a bus driver mints a per-device endpoint and hands it to a class driver, giving it a private channel to one device without the 8-slot global name registry. Kernel: shareCapability in ipc-synchronous.zig at both copy points; new setSystemCallResult3 (r8, saved/restored by the syscall stub); Task gains ipc_send_cap / ipc_received_cap. Runtime: callCap + Reply, replyWait gains send_cap and Received.cap; plain call/replyWait delegate with no_cap. New abi.no_cap. New ipc-cap test (two kernel tasks exercise both directions, each verifying the endpoint it received is the same object shared, refcount bumped to 2). No class driver consumes callCap yet — it lands with the first one. Suite 37/37 plus host tests. --- docs/danos-file-system-hierarchy-FSH.md | 48 ++++++------- docs/driver-model.md | 12 +++- library/runtime/ipc.zig | 61 +++++++++++++---- system/abi.zig | 5 ++ system/drivers/hpet/hpet.zig | 2 +- system/kernel/architecture/x86_64/cpu.zig | 8 +++ system/kernel/ipc-synchronous.zig | 60 +++++++++++++--- system/kernel/process.zig | 8 ++- system/kernel/scheduler.zig | 2 + system/kernel/tests.zig | 83 ++++++++++++++++++++++- system/services/vfs/vfs.zig | 2 +- test/qemu_test.py | 5 ++ 12 files changed, 243 insertions(+), 53 deletions(-) diff --git a/docs/danos-file-system-hierarchy-FSH.md b/docs/danos-file-system-hierarchy-FSH.md index 2e96831..3478036 100644 --- a/docs/danos-file-system-hierarchy-FSH.md +++ b/docs/danos-file-system-hierarchy-FSH.md @@ -4,39 +4,39 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o ## Directory structure -| Path | Description | -|-----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| / | Primary hierarchy root and root directory of the entire file system hierarchy. | -| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). | -| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). | -| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). | -| /etc | Host-specific system-wide configuration files. | -| /home | Users' home directories, containing saved files, personal settings, etc. | -| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. | -| /sbin | Essential system binaries (e.g init) | -| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems | +| Path | Description | +|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| / | Primary hierarchy root and root directory of the entire file system hierarchy. | +| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). | +| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). | +| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). | +| /etc | Host-specific system-wide configuration files. | +| /home | Users' home directories, containing saved files, personal settings, etc. | +| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. | +| /sbin | Essential system binaries (e.g init) | +| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems | | /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. | | /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) | -| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) | +| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) | | /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) | | /system/kernel | the kernel image | -| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. | -| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. | -| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. | +| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. | +| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. | +| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. | ## File types POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry. -| type | symbol | Description | -|-------------------|--------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). | -| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. | -| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. | -| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. | -| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). | -| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). | -| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. | +| type | symbol | Description | +|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). | +| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. | +| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. | +| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. | +| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). | +| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). | +| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. | ## /dev diff --git a/docs/driver-model.md b/docs/driver-model.md index 31b5819..d8137df 100644 --- a/docs/driver-model.md +++ b/docs/driver-model.md @@ -132,6 +132,12 @@ If a class driver needs `mmio`, it has become an HCD and should be one. - **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before EOI; `irq_ack` is the unmask. - **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment. +- **M13** — capability passing. `ipc_call` / `ipc_reply_wait` grew a `send_cap` argument + and a `received_cap` return (r8): an endpoint travels with a message, installed into + the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full + table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus + driver mints a per-device endpoint and hands it to a class driver. The runtime exposes + `callCap` and `replyWait(..., send_cap)`; no class driver consumes it yet. - **`system_spawn`** — a user-space supervisor starts a driver: `system_spawn(name)` loads a binary bundled in the initial-ramdisk as a fresh ring-3 process. This is what turned the device manager from "log the match" into "run the driver": the kernel now @@ -146,7 +152,11 @@ HCDs and class drivers do not work yet. Here is exactly why, and exactly what wo # Proposed ABI -## M13 — capability passing, for class drivers +## M13 — capability passing, for class drivers ✅ done + +*Implemented as described below (see "What exists today"). The signatures landed +verbatim: `send_cap` in r9, `received_cap` returned in r8, `-ENOSPC` on a full receiver +table with no delivery. The rest of this section is the original design note.* **The blocker.** A class driver has to reach *its* device. Today the only way to find an endpoint is the name registry: `ipc_register(service_id, h)` / `ipc_lookup(id)`, diff --git a/library/runtime/ipc.zig b/library/runtime/ipc.zig index e3195ae..b9acac9 100644 --- a/library/runtime/ipc.zig +++ b/library/runtime/ipc.zig @@ -43,11 +43,40 @@ pub fn lookup(id: abi.ServiceId) ?Handle { pub const CallError = error{Failed}; +/// The result of a capability-passing `callCap`: the reply length, and the handle of +/// an endpoint the server sent back (e.g. a per-device channel), or null. +pub const Reply = struct { + len: usize, + cap: ?Handle, +}; + +/// Send `message` to endpoint `h` and block until the server replies into `reply`, +/// optionally handing the server a capability (`send_cap`) and receiving one back. +/// This is the class-driver "open" primitive: call a bus with `send_cap = null`, get a +/// private per-device endpoint back in `.cap`. Two return values (reply length in rax, +/// received handle in r8) need a hand-written stub — r8 is read-write (in: reply +/// capacity, arg #4; out: the received handle). +pub fn callCap(h: Handle, message: []const u8, reply: []u8, send_cap: ?Handle) CallError!Reply { + var rax: usize = undefined; + var r8: usize = reply.len; // in: reply capacity (arg #4); out: received capability handle + asm volatile ("syscall" + : [rax] "={rax}" (rax), + [r8] "+{r8}" (r8), + : [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_call)), + [a0] "{rdi}" (h), + [a1] "{rsi}" (@intFromPtr(message.ptr)), + [a2] "{rdx}" (message.len), + [a3] "{r10}" (@intFromPtr(reply.ptr)), + [a5] "{r9}" (send_cap orelse abi.no_cap), + : .{ .rcx = true, .r11 = true, .memory = true }); + if (failed(rax)) return error.Failed; + return .{ .len = rax, .cap = if (r8 == abi.no_cap) null else r8 }; +} + /// Send `message` to endpoint `h` and block until the server replies into `reply`. -/// Returns the reply length. +/// Returns the reply length. The common case: no capability passed either way. pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize { - const r = sc.systemCall5(.ipc_call, h, @intFromPtr(message.ptr), message.len, @intFromPtr(reply.ptr), reply.len); - return if (failed(r)) error.Failed else r; + return (try callCap(h, message, reply, null)).len; } /// Set in `Received.badge` when what arrived is an asynchronous notification — a @@ -55,11 +84,12 @@ pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize { /// GSI. See `isNotification`. pub const notify_badge_bit: u64 = abi.notify_badge_bit; -/// The result of a `replyWait`: the request length and the sender's badge (a -/// task id, or an IRQ notification if the high bit is set). +/// The result of a `replyWait`: the request length, the sender's badge (a task id, or +/// an IRQ notification if the high bit is set), and any capability the request carried. pub const Received = struct { len: usize, badge: u64, + cap: ?Handle, /// True if this wake-up was a device interrupt, not a client request. A driver's /// event loop branches on this; there is no reply owed on the notification path. @@ -73,22 +103,25 @@ pub const Received = struct { } }; -/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if -/// any), then block until the next request arrives in `receive`. Returns its length -/// and the sender badge. This system_call returns two values — the length in rax and -/// the badge in rdx — so it needs a hand-written stub: rdx is a read-write -/// operand (input = reply length, arg #3; output = badge). -pub fn replyWait(h: Handle, reply: []const u8, receive: []u8) Received { +/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any, +/// optionally handing it `send_cap`), then block until the next request arrives in +/// `receive`. Returns its length, the sender badge, and any capability the request +/// carried (in `.cap`). Three return values — length in rax, badge in rdx, received +/// handle in r8 — so it needs a hand-written stub: rdx is read-write (in: reply length, +/// arg #3; out: badge) and r8 is read-write (in: receive capacity, arg #4; out: handle). +pub fn replyWait(h: Handle, reply: []const u8, receive: []u8, send_cap: ?Handle) Received { var rax: usize = undefined; - var rdx: usize = reply.len; // in: reply_len (arg #3 -> rdx); out: badge + var rdx: usize = reply.len; // in: reply_len (arg #3); out: badge + var r8: usize = receive.len; // in: receive capacity (arg #4); out: received capability handle asm volatile ("syscall" : [rax] "={rax}" (rax), [rdx] "+{rdx}" (rdx), + [r8] "+{r8}" (r8), : [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_reply_wait)), [a0] "{rdi}" (h), [a1] "{rsi}" (@intFromPtr(reply.ptr)), [a3] "{r10}" (@intFromPtr(receive.ptr)), - [a4] "{r8}" (receive.len), + [a5] "{r9}" (send_cap orelse abi.no_cap), : .{ .rcx = true, .r11 = true, .memory = true }); - return .{ .len = rax, .badge = rdx }; + return .{ .len = rax, .badge = rdx, .cap = if (r8 == abi.no_cap) null else r8 }; } diff --git a/system/abi.zig b/system/abi.zig index aeb8633..00909e3 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -59,3 +59,8 @@ pub const ServiceId = enum(u32) { pub const prot_read: u64 = 1; pub const prot_write: u64 = 2; pub const prot_exec: u64 = 4; + +/// `send_cap` / `received_cap` sentinel meaning "no capability" on the `ipc_call` / +/// `ipc_reply_wait` cap-passing path (M13). `~0`, like `no_parent` — a real handle is +/// a small index, so it can never collide. +pub const no_cap: u64 = ~@as(u64, 0); diff --git a/system/drivers/hpet/hpet.zig b/system/drivers/hpet/hpet.zig index 676806e..8372fd2 100644 --- a/system/drivers/hpet/hpet.zig +++ b/system/drivers/hpet/hpet.zig @@ -152,7 +152,7 @@ pub fn main() void { while (count < target_ticks) { // Blocked here. The task is `.blocked` and off every scheduler queue; the // next line runs only because the HPET raised its line. - const r = ipc.replyWait(endpoint, &.{}, &receive); + const r = ipc.replyWait(endpoint, &.{}, &receive, null); if (!r.isNotification()) continue; // a client request, not our IRQ // Quiet the device: write 1 to timer 0's status bit. Until this lands, the diff --git a/system/kernel/architecture/x86_64/cpu.zig b/system/kernel/architecture/x86_64/cpu.zig index b82dfa7..2fafbb3 100644 --- a/system/kernel/architecture/x86_64/cpu.zig +++ b/system/kernel/architecture/x86_64/cpu.zig @@ -87,6 +87,14 @@ pub fn setSystemCallResult2(state: *CpuState, value: u64) void { state.rdx = value; } +/// Write a *third* system_call return value (r8 here). r8 is an input argument +/// register (arg #4), but the syscall/int-0x80 stubs push and pop it around the +/// dispatch, so a value written into the frame is restored to the user on return. +/// Used by the IPC cap-passing calls to hand back the received capability handle. +pub fn setSystemCallResult3(state: *CpuState, value: u64) void { + state.r8 = value; +} + /// Bring up the serial port (the kernel's machine-readable log). No dependencies, /// so it can be the very first thing called. pub fn serialInit() void { diff --git a/system/kernel/ipc-synchronous.zig b/system/kernel/ipc-synchronous.zig index 13408c7..fec1f7c 100644 --- a/system/kernel/ipc-synchronous.zig +++ b/system/kernel/ipc-synchronous.zig @@ -157,10 +157,28 @@ pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool { // --- the two IPC operations ------------------------------------------------- +/// Share the capability named by handle `cap` in `from`'s table into `to`'s table, +/// bumping the endpoint's refcount (the sender keeps its handle — this is a copy, not +/// a move). Returns the handle it landed at in `to` (>= 0), or `-EBADF` if `cap` names +/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when +/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message. +fn shareCapability(from: *Task, to: *Task, cap: u64) i64 { + const endpoint = resolveHandle(from, cap) orelse return -EBADF; + endpoint.refcount += 1; + const handle = installHandle(to, endpoint); + if (handle < 0) { + dropRef(endpoint); // undo the bump; the receiver had no room + return -ENOSPC; + } + return handle; +} + /// Client side of IPC_Call: send `[message_ptr, message_len)` to `endpoint` and block until a /// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a -/// negative errno. Runs as the current task. -pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64) i64 { +/// negative errno. `send_cap` (a handle, or `no_cap`) is an endpoint transferred to the +/// server with the request; `out_received_cap` receives the handle of an endpoint the +/// server sent back in its reply, or `no_cap`. Runs as the current task. +pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64, send_cap: u64, out_received_cap: *u64) i64 { if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG; const flags = sync.enter(); defer sync.leave(flags); @@ -170,12 +188,15 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: me.ipc_send_len = message_len; me.ipc_reply_ptr = reply_ptr; me.ipc_reply_cap = reply_cap; + me.ipc_send_cap = send_cap; + me.ipc_received_cap = abi.no_cap; me.ipc_status = 0; enqueueSender(endpoint, me); // join the FIFO, then... scheduler.wakeLocked(&endpoint.receive_wait_queue); // ...wake a waiting server (no-op if none) scheduler.blockCurrentLocked(); // block until the reply readies us again + out_received_cap.* = me.ipc_received_cap; // a capability the replier sent back, or no_cap return me.ipc_status; // reply length or -errno, written by the replier } @@ -185,21 +206,33 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: /// to `out_badge` and returns the request length, or a negative errno. A pending /// notification is delivered ahead of client requests (length 0, badge with /// `notify_badge_bit` set, no reply owed). -pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, out_badge: *u64) i64 { +pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, send_cap: u64, out_badge: *u64, out_received_cap: *u64) i64 { if (reply_len > MESSAGE_MAXIMUM or receive_cap > MESSAGE_MAXIMUM) return -E2BIG; const flags = sync.enter(); defer sync.leave(flags); const me = scheduler.current(); + out_received_cap.* = abi.no_cap; // no capability received unless a request delivers one - // (1) Reply to the client we're still holding, if any. + // (1) Reply to the client we're still holding, if any — carrying `send_cap` to it. if (me.ipc_client) |client| { me.ipc_client = null; const n = @min(reply_len, client.ipc_reply_cap); - if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) { - client.ipc_status = @intCast(n); - } else { + client.ipc_received_cap = abi.no_cap; + if (!copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) { client.ipc_status = -EFAULT; + } else if (send_cap != abi.no_cap) { + // Transfer the reply's capability into the client. A failure fails the + // client's `call` rather than delivering a reply without its promised cap. + const shared = shareCapability(me, client, send_cap); + if (shared < 0) { + client.ipc_status = shared; // -EBADF (bad handle) or -ENOSPC (client table full) + } else { + client.ipc_received_cap = @intCast(shared); + client.ipc_status = @intCast(n); + } + } else { + client.ipc_status = @intCast(n); } scheduler.readyLocked(client); // its `call` now returns } @@ -208,7 +241,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt while (true) { if (popNotify(endpoint)) |badge| { out_badge.* = badge | notify_badge_bit; - return 0; // notification: no payload, no reply owed + return 0; // notification: no payload, no reply owed, no cap } if (dequeueSender(endpoint)) |caller| { const n = @min(caller.ipc_send_len, receive_cap); @@ -217,6 +250,17 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt scheduler.readyLocked(caller); continue; } + // Install the capability the caller sent, if any, into my table. A failure + // fails the caller's `call` and does not deliver — no half-delivered cap. + if (caller.ipc_send_cap != abi.no_cap) { + const shared = shareCapability(caller, me, caller.ipc_send_cap); + if (shared < 0) { + caller.ipc_status = shared; // -EBADF or -ENOSPC + scheduler.readyLocked(caller); + continue; + } + out_received_cap.* = @intCast(shared); + } me.ipc_client = caller; // remember who to reply to out_badge.* = caller.id; return @intCast(n); diff --git a/system/kernel/process.zig b/system/kernel/process.zig index c5f10ed..351362c 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -197,8 +197,10 @@ fn systemIpcLookup(state: *architecture.CpuState) void { /// stack, so it survives the block and receives the result on resume. fn systemIpcCall(state: *architecture.CpuState) void { const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); - const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4)); + var received_cap: u64 = abi.no_cap; + const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), architecture.systemCallArg(state, 5), &received_cap); architecture.setSystemCallResult(state, @bitCast(r)); + architecture.setSystemCallResult3(state, received_cap); } /// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len, @@ -206,9 +208,11 @@ fn systemIpcCall(state: *architecture.CpuState) void { fn systemIpcReplyWait(state: *architecture.CpuState) void { const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); var badge: u64 = 0; - const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), &badge); + var received_cap: u64 = abi.no_cap; + const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), architecture.systemCallArg(state, 5), &badge, &received_cap); architecture.setSystemCallResult(state, @bitCast(r)); architecture.setSystemCallResult2(state, badge); + architecture.setSystemCallResult3(state, received_cap); } /// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index 2705908..22e1480 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -66,6 +66,8 @@ pub const Task = struct { ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr) ipc_reply_cap: u64 = 0, ipc_status: i64 = 0, // client: reply length / -errno, written by the replier + ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none) + ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none) next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link) }; diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index c89a120..12e8424 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -82,6 +82,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { ipcTest(); } else if (eql(case, "ipc-call")) { ipcCallTest(); + } else if (eql(case, "ipc-cap")) { + capabilityTest(); } else if (eql(case, "smp")) { smpTest(); } else if (eql(case, "affinity")) { @@ -808,9 +810,10 @@ fn ipcServer() void { var reply_buffer: [8]u8 = undefined; var reply_len: u64 = 0; var badge: u64 = 0; + var received_cap: u64 = abi.no_cap; while (true) { var receive: [8]u8 = undefined; - const n = ipcsync.replyWait(ipc_endpoint, @intFromPtr(&reply_buffer), reply_len, @intFromPtr(&receive), receive.len, &badge); + const n = ipcsync.replyWait(ipc_endpoint, @intFromPtr(&reply_buffer), reply_len, @intFromPtr(&receive), receive.len, abi.no_cap, &badge, &received_cap); if (n < 0) scheduler.exit(); const v = std.mem.readInt(u64, receive[0..8], .little); std.mem.writeInt(u64, reply_buffer[0..8], v + 1, .little); @@ -826,7 +829,8 @@ fn ipcClient() void { var message: [8]u8 = undefined; std.mem.writeInt(u64, message[0..8], i, .little); var reply: [8]u8 = undefined; - const n = ipcsync.call(ipc_endpoint, @intFromPtr(&message), 8, @intFromPtr(&reply), reply.len); + var received_cap: u64 = abi.no_cap; + const n = ipcsync.call(ipc_endpoint, @intFromPtr(&message), 8, @intFromPtr(&reply), reply.len, abi.no_cap, &received_cap); if (n != 8 or std.mem.readInt(u64, reply[0..8], .little) != i + 1) ok = false; } ipc_replies_ok = ok; @@ -856,6 +860,81 @@ fn ipcCallTest() void { result(); } +// --- IPC capability passing (M13) ------------------------------------------- + +var cap_endpoint: *ipcsync.Endpoint = undefined; +var cap_ep_x: *ipcsync.Endpoint = undefined; // client mints, sends to the server +var cap_ep_y: *ipcsync.Endpoint = undefined; // server mints, sends back to the client +var cap_server_got_x: bool = false; +var cap_client_got_y: bool = false; +var cap_done: bool = false; + +/// Server half of the "open" pattern: receive one request carrying a capability, +/// verify it, then reply handing back a capability of its own. +fn capServer() void { + const me = scheduler.current(); + cap_ep_y = ipcsync.createEndpoint().?; + const h_y = ipcsync.installHandle(me, cap_ep_y); // the handle to send back in the reply + + var reply_buffer: [8]u8 = .{0} ** 8; + var receive: [8]u8 = undefined; + var badge: u64 = 0; + var received: u64 = abi.no_cap; + + // Phase 1: no reply owed yet — receive the client's request, which carries ep_x. + _ = ipcsync.replyWait(cap_endpoint, @intFromPtr(&reply_buffer), 0, @intFromPtr(&receive), receive.len, abi.no_cap, &badge, &received); + cap_server_got_x = received != abi.no_cap and + ipcsync.resolveHandle(me, received) == cap_ep_x and + cap_ep_x.refcount == 2; // shared (client's handle + this one), not moved + + // Phase 2: reply to the held client, handing it ep_y; then block for a next + // request that never comes (so this replyWait does not return). + _ = ipcsync.replyWait(cap_endpoint, @intFromPtr(&reply_buffer), 8, @intFromPtr(&receive), receive.len, @intCast(h_y), &badge, &received); + scheduler.exit(); +} + +/// Client half of "open": mint a capability, send it in a call, receive one back. +fn capClient() void { + const me = scheduler.current(); + cap_ep_x = ipcsync.createEndpoint().?; + const h_x = ipcsync.installHandle(me, cap_ep_x); + + var message: [8]u8 = .{0} ** 8; + var reply: [8]u8 = undefined; + var received: u64 = abi.no_cap; + _ = ipcsync.call(cap_endpoint, @intFromPtr(&message), 8, @intFromPtr(&reply), reply.len, @intCast(h_x), &received); + cap_client_got_y = received != abi.no_cap and + ipcsync.resolveHandle(me, received) == cap_ep_y and + cap_ep_y.refcount == 2; + + cap_done = true; + scheduler.exit(); +} + +/// IPC capability passing: a client hands the server an endpoint in a `call`, and the +/// server hands one back in its reply — the primitive that lets a bus driver give a +/// class driver a private channel to one device (M13). Two kernel tasks (no user ELF); +/// each verifies the endpoint it received is the *same* object the peer sent (resolves +/// equal) and was *shared*, not moved (refcount bumped to 2). +fn capabilityTest() void { + log("DANOS-TEST-BEGIN: ipc-cap\n", .{}); + cap_endpoint = ipcsync.createEndpoint().?; + cap_server_got_x = false; + cap_client_got_y = false; + cap_done = false; + scheduler.spawn(capServer, 5); + scheduler.spawn(capClient, 5); + + const done: *volatile bool = &cap_done; + var spins: u64 = 0; + while (!done.* and spins < 100_000_000) : (spins += 1) scheduler.yield(); + + check("capability test completed", cap_done); + check("server received the client's endpoint (same object, shared not moved)", cap_server_got_x); + check("client received the server's endpoint back (same object, shared not moved)", cap_client_got_y); + result(); +} + var proc_worker_run: bool = true; var proc_worker_ran: bool = false; diff --git a/system/services/vfs/vfs.zig b/system/services/vfs/vfs.zig index 4c7b660..2eb5313 100644 --- a/system/services/vfs/vfs.zig +++ b/system/services/vfs/vfs.zig @@ -128,7 +128,7 @@ pub fn main() void { var reply_len: usize = 0; var receive: [protocol.message_maximum]u8 = undefined; while (true) { - const got = runtime.ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive); + const got = runtime.ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null); // Ignore notifications (none expected here); handle a request. reply_len = handle(receive[0..got.len], &reply_buffer); } diff --git a/test/qemu_test.py b/test/qemu_test.py index ef0e41e..ec5b41d 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -125,6 +125,11 @@ CASES = [ {"name": "ipc-call", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # IPC capability passing (M13): a client hands the server an endpoint in a call, + # the server hands one back in its reply; each is verified same-object + shared. + {"name": "ipc-cap", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # Parallelism: needs more than one core, so this case boots with -smp 4. {"name": "smp", "smp": 4,