kernel: shm cross-process shared memory capability (v2 V2)

Generalize capability passing from endpoints to memory objects. The per-task
handle table now holds kind-tagged entries (scheduler.HandleObject{kind, ptr});
closeHandles and shareCapability dispatch by kind, so a shared-memory object
rides an ipc_call send_cap exactly like an endpoint and is refcount-freed only
when its last capability drops.

- shm_create(len) -> vaddr, handle: contiguous, zeroed, cacheable frames wrapped
  in a refcounted ShmObject, mapped into the caller's shm arena (PML4[230]).
- shm_map(cap) -> vaddr: map the same physical pages into a receiver that got the
  capability. mapUserSharedInto maps WB-cacheable + device_grant, so a sharer's
  teardown never frees the shared frames — the object owns them.
- runtime.shm: create(len) -> Region{ptr, handle, len}, map(handle) -> ptr.

Gate: qemu_test.py shm — shm-client creates a region, writes a pattern, passes
its capability to shm-server, which maps it and reads the same bytes back
(shm: shared 4096 bytes ok). ipc/ipc-call/ipc-cap/supervision/dma/usermem/
display-service and host tests all still pass — the handle change broke no IPC.
This commit is contained in:
Daniel Samson
2026-07-14 10:57:14 +01:00
parent 9333d0572f
commit 88ad432758
14 changed files with 439 additions and 31 deletions
@@ -188,6 +188,13 @@ pub fn mapUserDmaInto(root: u64, virtual: u64, physical: u64, len: u64) void {
paging.mapUserDmaInto(root, virtual, physical, len);
}
/// Map shared cacheable RAM into address space `root`: write-back cacheable, RW+NX, and
/// marked so teardown won't free the frames (they're owned by a refcounted shm object,
/// freed when its last capability drops). For shm_create/shm_map.
pub fn mapUserSharedInto(root: u64, virtual: u64, physical: u64, len: u64) void {
paging.mapUserSharedInto(root, virtual, physical, len);
}
/// Map a page into the kernel address space (non-executable). For the heap, etc.
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
paging.map(virtual, physical, writable);
@@ -393,6 +393,30 @@ pub fn leafIsWriteCombining(pml4: u64, virtual: u64) ?bool {
return (e & pte_pat != 0) and (e & pcd == 0) and (e & pwt == 0);
}
/// Map `[physical, physical+len)` into the user half rooted at `pml4` as **shared cacheable
/// RAM**: write-back cacheable (RW + NX) for CPU compositing, and carrying `device_grant`
/// so teardown (`freeSubtree`) does **not** return the frames to the allocator. The frames
/// are owned by a refcounted shared-memory object (system/kernel/ipc-synchronous.zig) and
/// freed only when its last capability drops — not when one sharer's address space dies, or
/// the other sharers would be left mapping freed RAM. The caller aligns `virtual`/`physical`.
pub fn mapUserSharedInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
const flags: u64 = present | user | writable | no_execute | device_grant; // WB cacheable
const first = physical & ~@as(u64, page_size - 1);
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
var off: u64 = 0;
while (first + off <= last) : (off += page_size) {
const v = virtual + off;
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
const pdpt = descendUser(pml4e);
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
const pd = descendUser(pdpte);
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
const pt = descendUser(pde);
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
invalidate(v);
}
}
/// Create a new address space: a fresh PML4 with an empty user half and the
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
+101 -17
View File
@@ -28,6 +28,7 @@ const architecture = @import("architecture");
const scheduler = @import("scheduler.zig");
const sync = @import("sync.zig");
const heap = @import("heap.zig");
const pmm = @import("pmm.zig");
const page_size = abi.page_size;
const Task = scheduler.Task;
@@ -123,6 +124,43 @@ pub fn dropRef(endpoint: *Endpoint) void {
}
}
// --- capability objects: what a handle-table entry can name ------------------
/// The `kind` tag on a `scheduler.HandleObject` — which capability object a handle names.
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
pub const handle_kind_endpoint: u8 = 0;
pub const handle_kind_shm: u8 = 1;
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
/// capability handles across processes and freed when the last one drops. `phys` is its
/// contiguous physical base, `pages` its length. A sharer's address-space teardown never
/// reclaims these frames (the mapping carries `device_grant`); this object owns them.
pub const ShmObject = struct {
refcount: u32 = 1,
phys: u64,
pages: usize,
};
/// Wrap `pages` contiguous frames at `phys` (already allocated + zeroed by the caller) in a
/// refcounted shm object, or null if the heap is out of room.
pub fn createShm(phys: u64, pages: usize) ?*ShmObject {
const shm = heap.allocator().create(ShmObject) catch return null;
shm.* = .{ .phys = phys, .pages = pages };
return shm;
}
/// Drop a shared-memory reference; when the last one goes, return its frames to the
/// allocator and free the object. (The mappings themselves are torn down with each
/// sharer's address space; `device_grant` keeps that from freeing the frames early.)
pub fn dropShmRef(shm: *ShmObject) void {
if (shm.refcount > 1) {
shm.refcount -= 1;
} else {
for (0..shm.pages) |i| pmm.free(shm.phys + i * page_size);
heap.allocator().destroy(shm);
}
}
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
@@ -222,11 +260,25 @@ pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when
/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message.
fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
const endpoint = resolveHandle(from, cap) orelse return -EBADF;
endpoint.refcount += 1;
const handle = installHandle(to, endpoint);
if (cap >= from.handles.len) return -EBADF;
const entry = from.handles[@intCast(cap)] orelse return -EBADF;
// Bump the named object's refcount (a copy, not a move — the sender keeps its handle),
// dispatching by kind so both endpoints and shared-memory regions can travel with a
// message.
switch (entry.kind) {
handle_kind_endpoint => {
const e: *Endpoint = @ptrCast(@alignCast(entry.ptr));
e.refcount += 1;
},
handle_kind_shm => {
const s: *ShmObject = @ptrCast(@alignCast(entry.ptr));
s.refcount += 1;
},
else => return -EBADF,
}
const handle = installEntry(to, entry);
if (handle < 0) {
dropRef(endpoint); // undo the bump; the receiver had no room
dropEntry(entry); // undo the bump; the receiver had no room
return -ENOSPC;
}
return handle;
@@ -416,36 +468,68 @@ pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
// --- per-process handle table + name registry -------------------------------
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
/// Install a capability object (kind + pointer) in task `t`'s handle table; returns the
/// small-int handle or -ENOSPC. The caller has already taken/holds the reference the slot
/// represents.
fn installEntry(t: *Task, entry: scheduler.HandleObject) i64 {
for (&t.handles, 0..) |*slot, i| {
if (slot.* == null) {
slot.* = @ptrCast(endpoint);
slot.* = entry;
return @intCast(i);
}
}
return -ENOSPC;
}
/// Resolve a handle to its endpoint, or null if out of range / unused.
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
if (h >= t.handles.len) return null;
const slot = t.handles[@intCast(h)] orelse return null;
return @ptrCast(@alignCast(slot));
/// Install an endpoint handle. The common case; keeps the endpoint callers' signature.
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
return installEntry(t, .{ .kind = handle_kind_endpoint, .ptr = @ptrCast(endpoint) });
}
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
/// exit path so a dead server's endpoints don't linger referenced.
/// Install a shared-memory handle.
pub fn installShmHandle(t: *Task, shm: *ShmObject) i64 {
return installEntry(t, .{ .kind = handle_kind_shm, .ptr = @ptrCast(shm) });
}
/// Resolve a handle to its endpoint, or null if out of range, unused, or a different kind
/// (e.g. an shm handle used where an endpoint is expected).
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
if (h >= t.handles.len) return null;
const entry = t.handles[@intCast(h)] orelse return null;
if (entry.kind != handle_kind_endpoint) return null;
return @ptrCast(@alignCast(entry.ptr));
}
/// Resolve a handle to its shared-memory object, or null if out of range, unused, or not
/// an shm handle.
pub fn resolveShm(t: *Task, h: u64) ?*ShmObject {
if (h >= t.handles.len) return null;
const entry = t.handles[@intCast(h)] orelse return null;
if (entry.kind != handle_kind_shm) return null;
return @ptrCast(@alignCast(entry.ptr));
}
/// Drop every capability reference an exiting task holds, dispatching by kind so a dead
/// task's endpoints *and* shared-memory regions are released correctly. Called from the
/// scheduler exit path.
pub fn closeHandles(t: *Task) void {
for (&t.handles) |*slot| {
if (slot.*) |p| {
dropRef(@ptrCast(@alignCast(p)));
if (slot.*) |entry| {
dropEntry(entry);
slot.* = null;
}
}
}
/// Drop the reference a handle-table entry represents, by kind.
fn dropEntry(entry: scheduler.HandleObject) void {
switch (entry.kind) {
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
handle_kind_shm => dropShmRef(@ptrCast(@alignCast(entry.ptr))),
else => {},
}
}
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
+76
View File
@@ -81,6 +81,18 @@ pub const device_arena_end: u64 = device_arena_base + (4 << 30);
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
/// The shared-memory arena: where `shm_create`/`shm_map` place shared cacheable regions, in
/// PML4[230] — a user-exclusive region distinct from the DMA arena. The frames are owned by
/// a refcounted shm object and freed when its last capability drops, not on teardown, so the
/// mapping carries `device_grant`. Per-process cursor in `Task.shm_map_next` (docs/display-v2.md).
pub const shm_arena_base: u64 = 0x0000_7300_0000_0000;
pub const shm_arena_end: u64 = shm_arena_base + (256 << 20); // 256 MiB per process
/// Largest single `shm_create`, in pages (32 MiB) — enough for a 4K framebuffer surface;
/// also an overflow guard on the page count. shm frames are contiguous (like DMA), so this
/// bounds the contiguous allocation asked of the frame allocator.
const maximum_shm_pages = 8192;
/// Largest single `mmap` grant, in pages (32 MiB). Big enough for a display service's
/// back buffer at up to 4K (3840x2160x4 ≈ 8100 pages); the user heap otherwise grows in
/// small chunks. `systemMmap` maps page by page with rollback, so this is only a sanity
@@ -212,6 +224,8 @@ fn system_call(state: *architecture.CpuState) void {
.timer_bind => systemTimerBind(state),
.klog_read => systemKlogRead(state),
.wall_clock => systemWallClock(state),
.shm_create => systemShmCreate(state),
.shm_map => systemShmMap(state),
_ => fail(state),
}
}
@@ -460,6 +474,68 @@ fn systemDmaFree(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, 0);
}
/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
/// its frames are owned by a refcounted object: the handle is passed to another process as
/// an `ipc_call` send_cap, that process `shm_map`s it, and the frames free only when the
/// last capability drops (docs/display-v2.md — the compositor↔native-driver and
/// app↔compositor surface path).
fn systemShmCreate(state: *architecture.CpuState) void {
const len = architecture.systemCallArg(state, 0);
const t = scheduler.current();
if (t.aspace == 0 or len == 0) return fail(state);
const pages: usize = @intCast((len + page_size - 1) / page_size);
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
// Reserve arena virtual space up front, so a mapping failure needs no rollback.
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
const base_v = t.shm_map_next;
if (base_v + pages * page_size > shm_arena_end) return fail(state); // arena exhausted
const phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse return fail(state);
// Zero through the physmap (the frames aren't mapped in the caller yet).
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
@memset(kernel_view[0 .. pages * page_size], 0);
const shm = ipc.createShm(phys, pages) orelse {
for (0..pages) |i| pmm.free(phys + i * page_size);
return fail(state);
};
const handle = ipc.installShmHandle(t, shm);
if (handle < 0) {
ipc.dropShmRef(shm); // last ref: frees the object and its frames
return fail(state);
}
architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size);
t.shm_map_next = base_v + pages * page_size;
architecture.setSystemCallResult(state, base_v); // vaddr for the CPU
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
}
/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
/// creator sees — returning the virtual address. The handle already holds a reference (taken
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
fn systemShmMap(state: *architecture.CpuState) void {
const cap = architecture.systemCallArg(state, 0);
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
const base_v = t.shm_map_next;
const size = shm.pages * page_size;
if (base_v + size > shm_arena_end) return fail(state);
architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size);
t.shm_map_next = base_v + size;
architecture.setSystemCallResult(state, base_v);
}
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
/// this process has claimed. The bus-driver primitive: a process that owns a bus
/// enumerates it and hands each device it finds to the table, where a class driver
+13 -3
View File
@@ -86,9 +86,11 @@ pub const Task = struct {
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
device_map_next: u64 = 0,
// --- synchronous IPC (ipc_sync.zig) ---
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
// opaque here so the scheduler and IPC modules don't import each other.
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
// Per-process handle table: a small-int handle names a kernel capability object.
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
// close/exit and cap-passing paths reclaim the right type. Kept opaque here so the
// scheduler and IPC modules don't import each other (ipc_sync.zig owns the kinds).
handles: [ipc_maximum_handles]?HandleObject = .{null} ** ipc_maximum_handles,
// A server holds the caller it currently owes a reply to (set by ReplyWait's
// receive, cleared when it replies). A client, while blocked in Call, records
// its message + reply buffers here and its result lands in `ipc_status`.
@@ -99,6 +101,7 @@ pub const Task = struct {
ipc_reply_cap: u64 = 0,
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
shm_map_next: u64 = 0, // bump pointer into this task's shared-memory arena (0 = unseeded)
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
@@ -125,6 +128,13 @@ pub const maximum_task_name = abi.maximum_process_name;
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
pub const ipc_maximum_handles = 16;
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the
/// capability-passing path reclaim/share the right type. The `kind` values are defined by
/// ipc_sync.zig (`handle_kind_*`); kept an opaque `u8` here so the scheduler doesn't import
/// the IPC module.
pub const HandleObject = struct { kind: u8, ptr: *anyopaque };
var tasks = [_]Task{.{}} ** maximum_tasks;
var next_id: u32 = 1;
+32
View File
@@ -101,6 +101,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
displayServiceTest(boot_information);
} else if (eql(case, "display-demo")) {
displayDemoTest(boot_information);
} else if (eql(case, "shm")) {
shmTest(boot_information);
} else if (eql(case, "clock")) {
clockTest();
} else if (eql(case, "smp")) {
@@ -2377,6 +2379,36 @@ fn displayDemoTest(boot_information: *const BootInformation) void {
while (true) scheduler.yield();
}
/// V2 — cross-process shared memory (docs/display-v2.md). Spawn shm-server and shm-client:
/// the client shm_creates a region, writes a pattern, and passes the region's capability to
/// the server as an ipc_call send_cap; the server shm_maps it and confirms the pattern is
/// visible — proving the two processes share the same physical pages, and that the extended
/// capability-passing (endpoints → memory objects) works. Its `shm: shared 4096 bytes ok`
/// heartbeat is the marker.
fn shmTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: shm\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over an initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
const rd = initial_ramdisk.Reader.init(image) orelse {
check("initial_ramdisk image is valid", false);
result();
return;
};
if (!spawnNamed(rd, "shm-server")) {
log("shm: could not spawn shm-server\n", .{});
result();
return;
}
_ = spawnNamed(rd, "shm-client");
scheduler.setPriority(1); // below the two, so they run
while (true) scheduler.yield();
}
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument