M13: IPC capability passing

ipc_call and ipc_reply_wait grow a `send_cap` argument (r9) and a `received_cap`
return (r8): an endpoint travels alongside a message, installed into the receiver's
handle table. The transfer is a share, not a move — the endpoint's refcount is bumped
and the sender keeps its handle. If the receiver's table is full the call fails
-ENOSPC and the message is NOT delivered (a half-delivered capability is worse than a
failed send); a bad handle fails -EBADF. Both directions carry a cap: a client's call
hands one to the server (seen in the server's replyWait), and the server's reply hands
one back (seen in the client's call return).

This is the "open" primitive the driver model was blocked on: a bus driver mints a
per-device endpoint and hands it to a class driver, giving it a private channel to one
device without the 8-slot global name registry.

Kernel: shareCapability in ipc-synchronous.zig at both copy points; new
setSystemCallResult3 (r8, saved/restored by the syscall stub); Task gains
ipc_send_cap / ipc_received_cap. Runtime: callCap + Reply, replyWait gains send_cap
and Received.cap; plain call/replyWait delegate with no_cap. New abi.no_cap.

New ipc-cap test (two kernel tasks exercise both directions, each verifying the
endpoint it received is the same object shared, refcount bumped to 2). No class driver
consumes callCap yet — it lands with the first one. Suite 37/37 plus host tests.
This commit is contained in:
Daniel Samson
2026-07-10 19:23:19 +01:00
parent 8eb4210251
commit a581712b09
12 changed files with 243 additions and 53 deletions
@@ -87,6 +87,14 @@ pub fn setSystemCallResult2(state: *CpuState, value: u64) void {
state.rdx = value;
}
/// Write a *third* system_call return value (r8 here). r8 is an input argument
/// register (arg #4), but the syscall/int-0x80 stubs push and pop it around the
/// dispatch, so a value written into the frame is restored to the user on return.
/// Used by the IPC cap-passing calls to hand back the received capability handle.
pub fn setSystemCallResult3(state: *CpuState, value: u64) void {
state.r8 = value;
}
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
/// so it can be the very first thing called.
pub fn serialInit() void {
+52 -8
View File
@@ -157,10 +157,28 @@ pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
// --- the two IPC operations -------------------------------------------------
/// Share the capability named by handle `cap` in `from`'s table into `to`'s table,
/// bumping the endpoint's refcount (the sender keeps its handle — this is a copy, not
/// a move). Returns the handle it landed at in `to` (>= 0), or `-EBADF` if `cap` names
/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when
/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message.
fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
const endpoint = resolveHandle(from, cap) orelse return -EBADF;
endpoint.refcount += 1;
const handle = installHandle(to, endpoint);
if (handle < 0) {
dropRef(endpoint); // undo the bump; the receiver had no room
return -ENOSPC;
}
return handle;
}
/// Client side of IPC_Call: send `[message_ptr, message_len)` to `endpoint` and block until a
/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a
/// negative errno. Runs as the current task.
pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64) i64 {
/// negative errno. `send_cap` (a handle, or `no_cap`) is an endpoint transferred to the
/// server with the request; `out_received_cap` receives the handle of an endpoint the
/// server sent back in its reply, or `no_cap`. Runs as the current task.
pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64, send_cap: u64, out_received_cap: *u64) i64 {
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
const flags = sync.enter();
defer sync.leave(flags);
@@ -170,12 +188,15 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr:
me.ipc_send_len = message_len;
me.ipc_reply_ptr = reply_ptr;
me.ipc_reply_cap = reply_cap;
me.ipc_send_cap = send_cap;
me.ipc_received_cap = abi.no_cap;
me.ipc_status = 0;
enqueueSender(endpoint, me); // join the FIFO, then...
scheduler.wakeLocked(&endpoint.receive_wait_queue); // ...wake a waiting server (no-op if none)
scheduler.blockCurrentLocked(); // block until the reply readies us again
out_received_cap.* = me.ipc_received_cap; // a capability the replier sent back, or no_cap
return me.ipc_status; // reply length or -errno, written by the replier
}
@@ -185,21 +206,33 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr:
/// to `out_badge` and returns the request length, or a negative errno. A pending
/// notification is delivered ahead of client requests (length 0, badge with
/// `notify_badge_bit` set, no reply owed).
pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, out_badge: *u64) i64 {
pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, send_cap: u64, out_badge: *u64, out_received_cap: *u64) i64 {
if (reply_len > MESSAGE_MAXIMUM or receive_cap > MESSAGE_MAXIMUM) return -E2BIG;
const flags = sync.enter();
defer sync.leave(flags);
const me = scheduler.current();
out_received_cap.* = abi.no_cap; // no capability received unless a request delivers one
// (1) Reply to the client we're still holding, if any.
// (1) Reply to the client we're still holding, if any — carrying `send_cap` to it.
if (me.ipc_client) |client| {
me.ipc_client = null;
const n = @min(reply_len, client.ipc_reply_cap);
if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
client.ipc_status = @intCast(n);
} else {
client.ipc_received_cap = abi.no_cap;
if (!copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
client.ipc_status = -EFAULT;
} else if (send_cap != abi.no_cap) {
// Transfer the reply's capability into the client. A failure fails the
// client's `call` rather than delivering a reply without its promised cap.
const shared = shareCapability(me, client, send_cap);
if (shared < 0) {
client.ipc_status = shared; // -EBADF (bad handle) or -ENOSPC (client table full)
} else {
client.ipc_received_cap = @intCast(shared);
client.ipc_status = @intCast(n);
}
} else {
client.ipc_status = @intCast(n);
}
scheduler.readyLocked(client); // its `call` now returns
}
@@ -208,7 +241,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
while (true) {
if (popNotify(endpoint)) |badge| {
out_badge.* = badge | notify_badge_bit;
return 0; // notification: no payload, no reply owed
return 0; // notification: no payload, no reply owed, no cap
}
if (dequeueSender(endpoint)) |caller| {
const n = @min(caller.ipc_send_len, receive_cap);
@@ -217,6 +250,17 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
scheduler.readyLocked(caller);
continue;
}
// Install the capability the caller sent, if any, into my table. A failure
// fails the caller's `call` and does not deliver — no half-delivered cap.
if (caller.ipc_send_cap != abi.no_cap) {
const shared = shareCapability(caller, me, caller.ipc_send_cap);
if (shared < 0) {
caller.ipc_status = shared; // -EBADF or -ENOSPC
scheduler.readyLocked(caller);
continue;
}
out_received_cap.* = @intCast(shared);
}
me.ipc_client = caller; // remember who to reply to
out_badge.* = caller.id;
return @intCast(n);
+6 -2
View File
@@ -197,8 +197,10 @@ fn systemIpcLookup(state: *architecture.CpuState) void {
/// stack, so it survives the block and receives the result on resume.
fn systemIpcCall(state: *architecture.CpuState) void {
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4));
var received_cap: u64 = abi.no_cap;
const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), architecture.systemCallArg(state, 5), &received_cap);
architecture.setSystemCallResult(state, @bitCast(r));
architecture.setSystemCallResult3(state, received_cap);
}
/// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len,
@@ -206,9 +208,11 @@ fn systemIpcCall(state: *architecture.CpuState) void {
fn systemIpcReplyWait(state: *architecture.CpuState) void {
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
var badge: u64 = 0;
const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), &badge);
var received_cap: u64 = abi.no_cap;
const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), architecture.systemCallArg(state, 5), &badge, &received_cap);
architecture.setSystemCallResult(state, @bitCast(r));
architecture.setSystemCallResult2(state, badge);
architecture.setSystemCallResult3(state, received_cap);
}
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
+2
View File
@@ -66,6 +66,8 @@ pub const Task = struct {
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
ipc_reply_cap: u64 = 0,
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
};
+81 -2
View File
@@ -82,6 +82,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
ipcTest();
} else if (eql(case, "ipc-call")) {
ipcCallTest();
} else if (eql(case, "ipc-cap")) {
capabilityTest();
} else if (eql(case, "smp")) {
smpTest();
} else if (eql(case, "affinity")) {
@@ -808,9 +810,10 @@ fn ipcServer() void {
var reply_buffer: [8]u8 = undefined;
var reply_len: u64 = 0;
var badge: u64 = 0;
var received_cap: u64 = abi.no_cap;
while (true) {
var receive: [8]u8 = undefined;
const n = ipcsync.replyWait(ipc_endpoint, @intFromPtr(&reply_buffer), reply_len, @intFromPtr(&receive), receive.len, &badge);
const n = ipcsync.replyWait(ipc_endpoint, @intFromPtr(&reply_buffer), reply_len, @intFromPtr(&receive), receive.len, abi.no_cap, &badge, &received_cap);
if (n < 0) scheduler.exit();
const v = std.mem.readInt(u64, receive[0..8], .little);
std.mem.writeInt(u64, reply_buffer[0..8], v + 1, .little);
@@ -826,7 +829,8 @@ fn ipcClient() void {
var message: [8]u8 = undefined;
std.mem.writeInt(u64, message[0..8], i, .little);
var reply: [8]u8 = undefined;
const n = ipcsync.call(ipc_endpoint, @intFromPtr(&message), 8, @intFromPtr(&reply), reply.len);
var received_cap: u64 = abi.no_cap;
const n = ipcsync.call(ipc_endpoint, @intFromPtr(&message), 8, @intFromPtr(&reply), reply.len, abi.no_cap, &received_cap);
if (n != 8 or std.mem.readInt(u64, reply[0..8], .little) != i + 1) ok = false;
}
ipc_replies_ok = ok;
@@ -856,6 +860,81 @@ fn ipcCallTest() void {
result();
}
// --- IPC capability passing (M13) -------------------------------------------
var cap_endpoint: *ipcsync.Endpoint = undefined;
var cap_ep_x: *ipcsync.Endpoint = undefined; // client mints, sends to the server
var cap_ep_y: *ipcsync.Endpoint = undefined; // server mints, sends back to the client
var cap_server_got_x: bool = false;
var cap_client_got_y: bool = false;
var cap_done: bool = false;
/// Server half of the "open" pattern: receive one request carrying a capability,
/// verify it, then reply handing back a capability of its own.
fn capServer() void {
const me = scheduler.current();
cap_ep_y = ipcsync.createEndpoint().?;
const h_y = ipcsync.installHandle(me, cap_ep_y); // the handle to send back in the reply
var reply_buffer: [8]u8 = .{0} ** 8;
var receive: [8]u8 = undefined;
var badge: u64 = 0;
var received: u64 = abi.no_cap;
// Phase 1: no reply owed yet — receive the client's request, which carries ep_x.
_ = ipcsync.replyWait(cap_endpoint, @intFromPtr(&reply_buffer), 0, @intFromPtr(&receive), receive.len, abi.no_cap, &badge, &received);
cap_server_got_x = received != abi.no_cap and
ipcsync.resolveHandle(me, received) == cap_ep_x and
cap_ep_x.refcount == 2; // shared (client's handle + this one), not moved
// Phase 2: reply to the held client, handing it ep_y; then block for a next
// request that never comes (so this replyWait does not return).
_ = ipcsync.replyWait(cap_endpoint, @intFromPtr(&reply_buffer), 8, @intFromPtr(&receive), receive.len, @intCast(h_y), &badge, &received);
scheduler.exit();
}
/// Client half of "open": mint a capability, send it in a call, receive one back.
fn capClient() void {
const me = scheduler.current();
cap_ep_x = ipcsync.createEndpoint().?;
const h_x = ipcsync.installHandle(me, cap_ep_x);
var message: [8]u8 = .{0} ** 8;
var reply: [8]u8 = undefined;
var received: u64 = abi.no_cap;
_ = ipcsync.call(cap_endpoint, @intFromPtr(&message), 8, @intFromPtr(&reply), reply.len, @intCast(h_x), &received);
cap_client_got_y = received != abi.no_cap and
ipcsync.resolveHandle(me, received) == cap_ep_y and
cap_ep_y.refcount == 2;
cap_done = true;
scheduler.exit();
}
/// IPC capability passing: a client hands the server an endpoint in a `call`, and the
/// server hands one back in its reply — the primitive that lets a bus driver give a
/// class driver a private channel to one device (M13). Two kernel tasks (no user ELF);
/// each verifies the endpoint it received is the *same* object the peer sent (resolves
/// equal) and was *shared*, not moved (refcount bumped to 2).
fn capabilityTest() void {
log("DANOS-TEST-BEGIN: ipc-cap\n", .{});
cap_endpoint = ipcsync.createEndpoint().?;
cap_server_got_x = false;
cap_client_got_y = false;
cap_done = false;
scheduler.spawn(capServer, 5);
scheduler.spawn(capClient, 5);
const done: *volatile bool = &cap_done;
var spins: u64 = 0;
while (!done.* and spins < 100_000_000) : (spins += 1) scheduler.yield();
check("capability test completed", cap_done);
check("server received the client's endpoint (same object, shared not moved)", cap_server_got_x);
check("client received the server's endpoint back (same object, shared not moved)", cap_client_got_y);
result();
}
var proc_worker_run: bool = true;
var proc_worker_ran: bool = false;