iommu: DMA-region capabilities — per-grant reachability, protocol flag-day
Replaces L2's interim DMA pool (every buffer reachable by every claimed device) with true per-grant confinement: a device reaches only buffers whose capability was delegated to its driver. Kernel: - DmaRegionObject (handle kind 2): a delegation token naming a dma_alloc'd region, passable across processes on the IPC cap slot like an endpoint or shared-memory object. Frames stay owned by the allocating address space (freed on dma_free/teardown as before); the token carries a `dead` flag so a stale downstream handle can no longer bind a freed region. - dma_alloc gains the dma_shareable flag: it returns a capability handle in r8 and every region is tracked in a registry. A task's own regions auto-bind into the devices it claims (its rings just work); foreign buffers are bound explicitly. - dma_bind / dma_unbind / handle_close syscalls (51-53). dma_bind maps a held region (or shared-memory) capability into a claimed device's domain; it is idempotent. handle_close reclaims a table slot (raised 16 -> 32). - dma_free and task death unmap a region from every domain and invalidate BEFORE its frames return to the allocator — the stale-IOTLB use-after- free window, closed structurally. Protocols (flag-day): block gains attach, usb-transfer gains dma_attach — each carries a region capability on the cap slot. fat allocates its bounce buffer shareable and attaches it; usb-storage allocates its transport buffers shareable, attaches them to the controller, and forwards fat's capability downstream; usb-xhci-bus binds and closes; virtio-gpu binds its shared scanout surface. The physical addresses on the wire are unchanged (identity IOVA), so no register-programming code moved. Cross-process DMA (fat -> usb-storage -> xHC) now flows only through delegated capabilities. iommu-usb-storage / iommu-usb-hid / iommu-fault all green under per-grant enforcement; 104/104 overall (fail-open paths unchanged).
This commit is contained in:
+159
-12
@@ -252,6 +252,9 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.fs_mount => systemFsMount(state),
|
||||
.fs_unmount => systemFsUnmount(state),
|
||||
.iommu_fault_drain => systemIommuFaultDrain(state),
|
||||
.dma_bind => systemDmaBind(state),
|
||||
.dma_unbind => systemDmaUnbind(state),
|
||||
.handle_close => systemHandleClose(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
.shared_memory_create => systemSharedMemoryCreate(state),
|
||||
.shared_memory_map => systemSharedMemoryMap(state),
|
||||
@@ -398,10 +401,14 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
// since the whole point is that claiming a DMA device is no longer equivalent
|
||||
// to ring 0. No-op when no IOMMU exists (fail-open).
|
||||
if (devices_broker.pciAddressOf(device_id)) |bdf| {
|
||||
if (!iommu.confineDevice(device_id, bdf, scheduler.current().id)) {
|
||||
_ = devices_broker.unclaim(device_id, scheduler.current().id);
|
||||
const owner = scheduler.current().id;
|
||||
if (!iommu.confineDevice(device_id, bdf, owner)) {
|
||||
_ = devices_broker.unclaim(device_id, owner);
|
||||
return fail(state);
|
||||
}
|
||||
// Bind the buffers this task allocated before claiming the device (a driver
|
||||
// that dma_alloc'd its rings, then claimed the controller).
|
||||
dmaBindOwnerRegionsInto(owner, device_id);
|
||||
}
|
||||
// A display service just took the framebuffer — quiesce the bootstrap console
|
||||
// so the kernel and the service don't scribble over each other's pixels. The
|
||||
@@ -510,6 +517,72 @@ fn systemIoWrite(state: *architecture.CpuState) void {
|
||||
/// their physical address is never disclosed. `dma_below_4g` caps the physical address
|
||||
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
|
||||
/// until PAT is programmed. See docs/driver-model.md (M14).
|
||||
// --- DMA-region registry ---------------------------------------------------------------
|
||||
// Every dma_alloc'd region is tracked here so it can be (a) auto-bound into the devices
|
||||
// its owner claims, (b) delegated across processes as a capability and bound into a
|
||||
// device's IOMMU domain by dma_bind, and (c) unmapped from every domain before its frames
|
||||
// return to the allocator. Only *shareable* regions carry a heap `object` (the delegation
|
||||
// token); a driver's private rings are tracked without one. All access under the big lock.
|
||||
const DmaRegistryEntry = struct {
|
||||
active: bool = false,
|
||||
object: ?*ipc.DmaRegionObject = null,
|
||||
physical: u64 = 0,
|
||||
len: u64 = 0,
|
||||
owner: u32 = 0,
|
||||
};
|
||||
const maximum_dma_regions = 256;
|
||||
var dma_registry: [maximum_dma_regions]DmaRegistryEntry = .{DmaRegistryEntry{}} ** maximum_dma_regions;
|
||||
|
||||
fn dmaRegistryAdd(object: ?*ipc.DmaRegionObject, physical: u64, len: u64, owner: u32) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (!e.active) {
|
||||
e.* = .{ .active = true, .object = object, .physical = physical, .len = len, .owner = owner };
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retire the region based at `physical` owned by `owner`: unmap it from every device
|
||||
/// domain (before the frames are freed), mark its token dead so a stale downstream handle
|
||||
/// can no longer bind it, drop the registry's reference, and clear the slot.
|
||||
fn dmaRegistryRemove(owner: u32, physical: u64) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner and e.physical == physical) {
|
||||
iommu.unmapRegionEverywhere(e.physical, e.len);
|
||||
if (e.object) |object| {
|
||||
object.dead = true;
|
||||
ipc.dropDmaRegionReference(object);
|
||||
}
|
||||
e.* = .{};
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retire every region owned by `owner` (task death) — same discipline as a per-region
|
||||
/// free, run before the address space is torn down and its DMA frames reclaimed.
|
||||
fn dmaRegistryReleaseOwner(owner: u32) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner) {
|
||||
iommu.unmapRegionEverywhere(e.physical, e.len);
|
||||
if (e.object) |object| {
|
||||
object.dead = true;
|
||||
ipc.dropDmaRegionReference(object);
|
||||
}
|
||||
e.* = .{};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Bind every region `owner` allocated into the domain of the device it just claimed
|
||||
/// (its rings allocated before the claim). Regions allocated *after* the claim are bound
|
||||
/// by dma_alloc's own auto-bind.
|
||||
fn dmaBindOwnerRegionsInto(owner: u32, device_id: u64) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner) _ = iommu.mapForDevice(device_id, e.physical, e.len);
|
||||
}
|
||||
}
|
||||
|
||||
/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults,
|
||||
/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects
|
||||
/// its device faulted can force the fault records to be logged now rather than waiting
|
||||
@@ -521,6 +594,60 @@ fn systemIommuFaultDrain(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, count);
|
||||
}
|
||||
|
||||
/// Resolve a capability handle the caller holds to the physical range it names — either
|
||||
/// a DMA-region token or a shared-memory object (both are bindable buffers). null if the
|
||||
/// handle is neither, or names a region already freed by its allocator.
|
||||
fn bindableRange(t: *scheduler.Task, handle: u64) ?struct { physical: u64, len: u64 } {
|
||||
if (ipc.resolveDmaRegion(t, handle)) |region| {
|
||||
if (region.dead) return null;
|
||||
return .{ .physical = region.phys, .len = region.pages * page_size };
|
||||
}
|
||||
if (ipc.resolveSharedMemory(t, handle)) |shared| {
|
||||
return .{ .physical = shared.phys, .len = shared.pages * page_size };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// dma_bind(device_id, handle) -> 0/-errno: map the buffer named by `handle` into the
|
||||
/// claimed device's IOMMU domain. The caller must own the device (the mmio_map gate) and
|
||||
/// hold the handle. Idempotent: re-binding is a harmless success, so a driver may re-bind
|
||||
/// after a restart without tracking what it already bound. Success (no-op) without an IOMMU.
|
||||
fn systemDmaBind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const handle = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
|
||||
const range = bindableRange(t, handle) orelse return fail(state);
|
||||
if (!iommu.mapForDevice(device_id, range.physical, range.len)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// dma_unbind(device_id, handle) -> 0/-errno: unmap a previously bound buffer from the
|
||||
/// device's domain and invalidate.
|
||||
fn systemDmaUnbind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const handle = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
|
||||
const range = bindableRange(t, handle) orelse return fail(state);
|
||||
iommu.unmapForDevice(device_id, range.physical, range.len);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// handle_close(handle) -> 0/-errno: drop one capability handle and free its slot.
|
||||
fn systemHandleClose(state: *architecture.CpuState) void {
|
||||
const handle = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (ipc.closeHandle(t, handle) < 0) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const flags = architecture.systemCallArg(state, 1);
|
||||
@@ -567,15 +694,33 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
// Publish the region to the DMA pool: it becomes reachable to every claimed device
|
||||
// (the interim rule until DMA-region capabilities land). No-op without an IOMMU.
|
||||
// Register the region and bind it into every device this task already drives (its
|
||||
// own buffers reach its own devices). When `dma_shareable` is set, also wrap it in a
|
||||
// capability token and return a handle so it can be delegated to another driver and
|
||||
// dma_bound there. All under the lock (the registry + IOMMU tables are shared).
|
||||
var handle: u64 = abi.no_cap;
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
iommu.poolAdd(phys, pages * page_size);
|
||||
sync.leave(lock_flags);
|
||||
defer sync.leave(lock_flags);
|
||||
var object: ?*ipc.DmaRegionObject = null;
|
||||
if (flags & abi.dma_shareable != 0) {
|
||||
if (ipc.createDmaRegion(phys, pages, t.id)) |region| {
|
||||
const h = ipc.installDmaRegionHandle(t, region);
|
||||
if (h >= 0) {
|
||||
region.refcount += 1; // the handle's reference (registry holds the first)
|
||||
handle = @intCast(h);
|
||||
object = region;
|
||||
} else {
|
||||
ipc.dropDmaRegionReference(region); // no table slot; drop it
|
||||
}
|
||||
}
|
||||
}
|
||||
dmaRegistryAdd(object, phys, pages * page_size, t.id);
|
||||
iommu.mapRegionForOwner(t.id, phys, pages * page_size);
|
||||
}
|
||||
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
||||
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
||||
architecture.setSystemCallResult3(state, handle); // capability handle (no_cap unless shareable)
|
||||
}
|
||||
|
||||
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||
@@ -591,14 +736,13 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
||||
|
||||
// Pull the region out of every device's domain and invalidate BEFORE any frame
|
||||
// returns to the allocator — a device still translating to a reallocated frame is a
|
||||
// use-after-free. dma_alloc's frames are contiguous, so the base translation names
|
||||
// the whole region.
|
||||
// Retire the region — unmap it from every device domain and mark its token dead —
|
||||
// BEFORE any frame returns to the allocator, so no device can still translate to a
|
||||
// reallocated frame. dma_alloc's frames are contiguous, so the base names the region.
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
if (architecture.translate(t.address_space, base_v)) |base_phys|
|
||||
iommu.poolRemove(base_phys, pages * page_size);
|
||||
dmaRegistryRemove(t.id, base_phys);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
|
||||
@@ -1036,8 +1180,11 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
// Detach the task's devices from their IOMMU domains BEFORE the broker clears the
|
||||
// claims (the detach reads ownership) and before the address space is torn down and
|
||||
// its DMA frames return to the allocator — a device must stop translating to a frame
|
||||
// before that frame can be handed to someone else.
|
||||
// before that frame can be handed to someone else. Then retire the buffers this task
|
||||
// allocated, unmapping them from any *other* driver's domain they were granted into,
|
||||
// before those frames are freed too.
|
||||
iommu.releaseAllOwnedBy(t.id);
|
||||
dmaRegistryReleaseOwner(t.id);
|
||||
devices_broker.releaseAllOwnedBy(t.id);
|
||||
// If that dropped the framebuffer claim (this task was the display service), let the
|
||||
// bootstrap console draw again — the screen is nobody's now, so panics/status land.
|
||||
|
||||
Reference in New Issue
Block a user