kernel: shm cross-process shared memory capability (v2 V2)

Generalize capability passing from endpoints to memory objects. The per-task
handle table now holds kind-tagged entries (scheduler.HandleObject{kind, ptr});
closeHandles and shareCapability dispatch by kind, so a shared-memory object
rides an ipc_call send_cap exactly like an endpoint and is refcount-freed only
when its last capability drops.

- shm_create(len) -> vaddr, handle: contiguous, zeroed, cacheable frames wrapped
  in a refcounted ShmObject, mapped into the caller's shm arena (PML4[230]).
- shm_map(cap) -> vaddr: map the same physical pages into a receiver that got the
  capability. mapUserSharedInto maps WB-cacheable + device_grant, so a sharer's
  teardown never frees the shared frames — the object owns them.
- runtime.shm: create(len) -> Region{ptr, handle, len}, map(handle) -> ptr.

Gate: qemu_test.py shm — shm-client creates a region, writes a pattern, passes
its capability to shm-server, which maps it and reads the same bytes back
(shm: shared 4096 bytes ok). ipc/ipc-call/ipc-cap/supervision/dma/usermem/
display-service and host tests all still pass — the handle change broke no IPC.
This commit is contained in:
Daniel Samson
2026-07-14 10:57:14 +01:00
parent 9333d0572f
commit 88ad432758
14 changed files with 439 additions and 31 deletions
+76
View File
@@ -81,6 +81,18 @@ pub const device_arena_end: u64 = device_arena_base + (4 << 30);
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
/// The shared-memory arena: where `shm_create`/`shm_map` place shared cacheable regions, in
/// PML4[230] — a user-exclusive region distinct from the DMA arena. The frames are owned by
/// a refcounted shm object and freed when its last capability drops, not on teardown, so the
/// mapping carries `device_grant`. Per-process cursor in `Task.shm_map_next` (docs/display-v2.md).
pub const shm_arena_base: u64 = 0x0000_7300_0000_0000;
pub const shm_arena_end: u64 = shm_arena_base + (256 << 20); // 256 MiB per process
/// Largest single `shm_create`, in pages (32 MiB) — enough for a 4K framebuffer surface;
/// also an overflow guard on the page count. shm frames are contiguous (like DMA), so this
/// bounds the contiguous allocation asked of the frame allocator.
const maximum_shm_pages = 8192;
/// Largest single `mmap` grant, in pages (32 MiB). Big enough for a display service's
/// back buffer at up to 4K (3840x2160x4 ≈ 8100 pages); the user heap otherwise grows in
/// small chunks. `systemMmap` maps page by page with rollback, so this is only a sanity
@@ -212,6 +224,8 @@ fn system_call(state: *architecture.CpuState) void {
.timer_bind => systemTimerBind(state),
.klog_read => systemKlogRead(state),
.wall_clock => systemWallClock(state),
.shm_create => systemShmCreate(state),
.shm_map => systemShmMap(state),
_ => fail(state),
}
}
@@ -460,6 +474,68 @@ fn systemDmaFree(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, 0);
}
/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
/// its frames are owned by a refcounted object: the handle is passed to another process as
/// an `ipc_call` send_cap, that process `shm_map`s it, and the frames free only when the
/// last capability drops (docs/display-v2.md — the compositor↔native-driver and
/// app↔compositor surface path).
fn systemShmCreate(state: *architecture.CpuState) void {
const len = architecture.systemCallArg(state, 0);
const t = scheduler.current();
if (t.aspace == 0 or len == 0) return fail(state);
const pages: usize = @intCast((len + page_size - 1) / page_size);
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
// Reserve arena virtual space up front, so a mapping failure needs no rollback.
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
const base_v = t.shm_map_next;
if (base_v + pages * page_size > shm_arena_end) return fail(state); // arena exhausted
const phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse return fail(state);
// Zero through the physmap (the frames aren't mapped in the caller yet).
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
@memset(kernel_view[0 .. pages * page_size], 0);
const shm = ipc.createShm(phys, pages) orelse {
for (0..pages) |i| pmm.free(phys + i * page_size);
return fail(state);
};
const handle = ipc.installShmHandle(t, shm);
if (handle < 0) {
ipc.dropShmRef(shm); // last ref: frees the object and its frames
return fail(state);
}
architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size);
t.shm_map_next = base_v + pages * page_size;
architecture.setSystemCallResult(state, base_v); // vaddr for the CPU
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
}
/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
/// creator sees — returning the virtual address. The handle already holds a reference (taken
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
fn systemShmMap(state: *architecture.CpuState) void {
const cap = architecture.systemCallArg(state, 0);
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
const base_v = t.shm_map_next;
const size = shm.pages * page_size;
if (base_v + size > shm_arena_end) return fail(state);
architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size);
t.shm_map_next = base_v + size;
architecture.setSystemCallResult(state, base_v);
}
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
/// this process has claimed. The bus-driver primitive: a process that owns a bus
/// enumerates it and hands each device it finds to the table, where a class driver