iommu: DMA-region capabilities — per-grant reachability, protocol flag-day
Replaces L2's interim DMA pool (every buffer reachable by every claimed device) with true per-grant confinement: a device reaches only buffers whose capability was delegated to its driver. Kernel: - DmaRegionObject (handle kind 2): a delegation token naming a dma_alloc'd region, passable across processes on the IPC cap slot like an endpoint or shared-memory object. Frames stay owned by the allocating address space (freed on dma_free/teardown as before); the token carries a `dead` flag so a stale downstream handle can no longer bind a freed region. - dma_alloc gains the dma_shareable flag: it returns a capability handle in r8 and every region is tracked in a registry. A task's own regions auto-bind into the devices it claims (its rings just work); foreign buffers are bound explicitly. - dma_bind / dma_unbind / handle_close syscalls (51-53). dma_bind maps a held region (or shared-memory) capability into a claimed device's domain; it is idempotent. handle_close reclaims a table slot (raised 16 -> 32). - dma_free and task death unmap a region from every domain and invalidate BEFORE its frames return to the allocator — the stale-IOTLB use-after- free window, closed structurally. Protocols (flag-day): block gains attach, usb-transfer gains dma_attach — each carries a region capability on the cap slot. fat allocates its bounce buffer shareable and attaches it; usb-storage allocates its transport buffers shareable, attaches them to the controller, and forwards fat's capability downstream; usb-xhci-bus binds and closes; virtio-gpu binds its shared scanout surface. The physical addresses on the wire are unchanged (identity IOVA), so no register-programming code moved. Cross-process DMA (fat -> usb-storage -> xHC) now flows only through delegated capabilities. iommu-usb-storage / iommu-usb-hid / iommu-fault all green under per-grant enforcement; 104/104 overall (fail-open paths unchanged).
This commit is contained in:
+39
-39
@@ -128,29 +128,18 @@ pub fn init() void {
|
||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
||||
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
||||
|
||||
/// The interim DMA pool: the set of `dma_alloc`'d regions. Every such region is mapped
|
||||
/// into every claimed device's domain, so a device reaches DMA buffers (including one
|
||||
/// another's — the honest limit of this stage) but NOT the kernel, page tables, process
|
||||
/// heaps, or arbitrary RAM. The capability layer (L3/L4) narrows this to per-grant.
|
||||
const PoolRegion = struct { active: bool = false, physical: u64 = 0, len: u64 = 0 };
|
||||
var pool: [maximum_pool]PoolRegion = .{PoolRegion{}} ** maximum_pool;
|
||||
const maximum_pool = 256;
|
||||
|
||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
||||
/// it a private empty domain, seed it with the current DMA pool and the device's own
|
||||
/// firmware reserved region, and attach. false only if a domain can't be allocated —
|
||||
/// the caller rolls the claim back (a claim that can't be confined must not stand). No-op
|
||||
/// success when no IOMMU exists (fail-open).
|
||||
/// it a private empty domain, seed it with the device's own firmware reserved region,
|
||||
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
|
||||
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
|
||||
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
|
||||
/// the claim back (a claim that can't be confined must not stand). No-op success when no
|
||||
/// IOMMU exists (fail-open).
|
||||
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
if (kind == .none) return true;
|
||||
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||
const domain = domainCreate(owner, bdf) orelse return false;
|
||||
|
||||
// Seed with every pooled DMA region so the device's own rings/buffers and the
|
||||
// cross-process buffers handed to it (fat's bounce buffer via usb-storage) resolve.
|
||||
for (&pool) |*r| {
|
||||
if (r.active) _ = map(domain, r.physical, r.len);
|
||||
}
|
||||
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
||||
const info = platform.platformInformation();
|
||||
var i: usize = 0;
|
||||
@@ -164,34 +153,45 @@ pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Record a newly `dma_alloc`'d region and map it into every claimed device's domain
|
||||
/// (the interim pool rule). Called from the dma_alloc syscall.
|
||||
pub fn poolAdd(physical: u64, len: u64) void {
|
||||
/// The confined record for `device_id`, or null if the device is not confined.
|
||||
fn confinedOf(device_id: u64) ?*Confined {
|
||||
if (device_id >= confined.len) return null;
|
||||
const c = &confined[@intCast(device_id)];
|
||||
return if (c.active) c else null;
|
||||
}
|
||||
|
||||
/// Map a DMA region into a specific claimed device's domain (the device owner binding a
|
||||
/// granted buffer). false if the device is not confined. No-op success without an IOMMU.
|
||||
pub fn mapForDevice(device_id: u64, physical: u64, len: u64) bool {
|
||||
if (kind == .none) return true;
|
||||
const c = confinedOf(device_id) orelse return false;
|
||||
return map(c.domain, physical, len);
|
||||
}
|
||||
|
||||
/// Unmap a DMA region from a specific claimed device's domain. No-op if not confined.
|
||||
pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
const c = confinedOf(device_id) orelse return;
|
||||
unmap(c.domain, physical, len);
|
||||
}
|
||||
|
||||
/// Map a region into every claimed device owned by `owner` — the auto-bind of a task's
|
||||
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
||||
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
for (&pool) |*r| {
|
||||
if (!r.active) {
|
||||
r.* = .{ .active = true, .physical = physical, .len = len };
|
||||
break;
|
||||
}
|
||||
} else return; // pool full; region stays unmapped and its device DMA will fault
|
||||
for (&confined) |*c| {
|
||||
if (c.active) _ = map(c.domain, physical, len);
|
||||
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
/// Unmap a freed DMA region from every claimed device's domain and forget it. MUST run
|
||||
/// before the frames return to pmm — a device translating to a reallocated frame is the
|
||||
/// use-after-free this prevents. Called from the dma_free syscall.
|
||||
pub fn poolRemove(physical: u64, len: u64) void {
|
||||
/// Unmap a region from EVERY claimed device's domain — the freed-region sweep. MUST run
|
||||
/// before the frames return to pmm: a device translating to a reallocated frame is the
|
||||
/// use-after-free this prevents. Cross-device because a granted buffer may be bound in a
|
||||
/// domain other than its owner's.
|
||||
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
||||
if (kind == .none) return;
|
||||
for (&pool) |*r| {
|
||||
if (r.active and r.physical == physical and r.len == len) {
|
||||
for (&confined) |*c| {
|
||||
if (c.active) unmap(c.domain, physical, len);
|
||||
}
|
||||
r.* = .{};
|
||||
return;
|
||||
}
|
||||
for (&confined) |*c| {
|
||||
if (c.active) unmap(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user