2 Commits
Author SHA1 Message Date
Daniel Samson 125a3b4993 M14b: DMA memory (dma_alloc / dma_free)
An HCD programs a bus-master engine: it needs a descriptor ring that is physically
contiguous, at a physical address it knows, uncacheable, and pinned. mmap gives none
of those. Add dma_alloc(len, flags) -> vaddr (rax), paddr (rdx) and dma_free(vaddr,
len): grant contiguous, zeroed, pinned, strong-uncacheable memory in a per-process DMA
arena (PML4[228]) and hand back both addresses.

Pieces: pmm.allocContiguous(count, max_phys) finds a run of contiguous free frames
below a cap (dma_below_4g for 32-bit engines); mapUserDmaInto maps them uncacheable
(PCD|PWT) but WITHOUT device_grant, so unlike an MMIO grant these frames are real RAM
and freeSubtree returns them on teardown — a driver that dies leaks nothing. dma_free
is bounded to the DMA arena so it can never unmap the caller's stack/heap/MMIO.
dma_write_combining is accepted but falls back to coherent (WC needs PAT programming).

Runtime: runtime.dma.alloc/free (a two-return-value stub, like replyWait). New `dma`
kernel test drives the mechanism directly — contiguity, the below-4G cap, coherent
mapping, and reclaim-on-teardown (no leak). The thin syscall wrappers follow the tested
mmap/mmio_map shape and land their first real use with the first DMA driver. Suite
38/38 plus host tests.
2026-07-10 19:43:59 +01:00
Daniel Samson e7c7e7b94c M14a: memory-ordering / MMIO layer (library/mmio)
The tree had zero memory barriers — correct-by-accident on x86 (TSO + strong-
uncacheable MMIO), but a landmine for the first DMA driver and for ARM, which is the
win condition. Add /lib/mmio: typed volatile register access (read/write) plus mb /
rmb / wmb, lowered per-architecture (mfence/lfence/sfence on x86_64, dsb sy/ld/st on
aarch64) so the ordering rules are a named primitive, not scattered `asm volatile`.
`volatile` is not a barrier — it says nothing about ordinary stores (a DMA descriptor
in WB RAM) relative to a volatile doorbell write; wmb() between them is the fix.

Prove it on the one existing caller: hpet now does its register access through
mmio.read/write. It needs no barriers itself (pure MMIO, no DMA, UC grant on x86) —
the point is the typed, arch-portable access every driver should use; the barriers are
there for the DMA drivers to come.

New `mmio` module injected into addUserBinary; host test asserts the barriers assemble
and a register round-trips. Suite 37/37 plus host tests.
2026-07-10 19:32:57 +01:00
14 changed files with 372 additions and 22 deletions
+17 -6
View File
@@ -59,6 +59,7 @@ fn addUserBinary(
target: std.Build.ResolvedTarget,
runtime_module: *std.Build.Module,
posix_module: *std.Build.Module,
mmio_module: *std.Build.Module,
name: []const u8,
root: []const u8,
) *std.Build.Step.Compile {
@@ -78,6 +79,8 @@ fn addUserBinary(
// POSIX/C compatibility layer, available to any program that wants it
// (danos-native code uses `runtime` directly). See library/posix/.
.{ .name = "posix", .module = posix_module },
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
.{ .name = "mmio", .module = mmio_module },
},
}),
});
@@ -181,6 +184,13 @@ pub fn build(b: *std.Build) void {
},
});
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
const mmio_module = b.addModule("mmio", .{
.root_source_file = b.path("library/mmio/mmio.zig"),
});
// The POSIX / C compatibility layer, a separate library layered strictly over the
// runtime (it calls the runtime's IPC/heap, never system calls directly). This is
// the one place POSIX/C spellings are allowed verbatim — see docs/coding-standards.md
@@ -264,7 +274,7 @@ pub fn build(b: *std.Build) void {
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
// linked into the kernel's user region against the `runtime` runtime library, and
// started in ring 3 by the kernel's user-ELF loader.
const init_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "init", "system/services/init/init.zig");
const init_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "init", "system/services/init/init.zig");
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
b.getInstallStep().dependOn(&init_install.step);
@@ -272,11 +282,11 @@ pub fn build(b: *std.Build) void {
// Each is built by the same user-binary recipe, then packed into one image by
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "vfs", "system/services/vfs/vfs.zig");
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "vfs-test", "system/services/vfs/vfs-test.zig");
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "hpet", "system/drivers/hpet/hpet.zig");
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "bus", "system/drivers/bus/bus.zig");
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "device-manager", "system/services/device-manager/device-manager.zig");
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "vfs", "system/services/vfs/vfs.zig");
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "vfs-test", "system/services/vfs/vfs-test.zig");
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "hpet", "system/drivers/hpet/hpet.zig");
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "bus", "system/drivers/bus/bus.zig");
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "device-manager", "system/services/device-manager/device-manager.zig");
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
// (the container format is trivial, and Python sidesteps std API churn). Args:
@@ -429,6 +439,7 @@ pub fn build(b: *std.Build) void {
"system/boot-handoff.zig",
"system/abi.zig",
"system/devices/device-abi.zig",
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
}) |root| {
const mod_tests = b.addTest(.{
.root_module = b.createModule(.{
+13 -1
View File
@@ -138,6 +138,13 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
`callCap` and `replyWait(..., send_cap)`; no class driver consumes it yet.
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`;
no DMA driver consumes `dma_alloc` yet.
- **`system_spawn`** — a user-space supervisor starts a driver: `system_spawn(name)`
loads a binary bundled in the initial-ramdisk as a fresh ring-3 process. This is what
turned the device manager from "log the match" into "run the driver": the kernel now
@@ -194,7 +201,12 @@ const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
// now dev_ep is a private channel to that one device
```
## M14 — DMA memory and the memory-ordering contract, for HCDs
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
PAT, a small follow-up. The rest of this section is the original design note.*
**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the
device can read, which means memory that is (a) physically contiguous, (b) at a
+80
View File
@@ -0,0 +1,80 @@
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
//!
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
//! don't reorder it against *other volatile* accesses. It says nothing about ordinary
//! stores — the DMA descriptor you just filled in write-back RAM — which the compiler
//! (and, on weakly-ordered hardware, the CPU) may freely move past a volatile MMIO
//! write. The canonical bug:
//!
//! ring[i] = descriptor; // ordinary store to WB RAM
//! doorbell.* = i; // volatile store to UC MMIO
//! // nothing orders these; the device can read a stale descriptor
//!
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
//! reason they are a named primitive and not scattered `asm volatile`:
//!
//! x86_64 aarch64
//! mb() mfence dsb sy
//! rmb() lfence dsb ld
//! wmb() sfence dsb st
//!
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
//! abstraction exists now, while there is one caller (hpet) to get right. See
//! docs/driver-model.md (M14) for the full ordering contract.
const builtin = @import("builtin");
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
/// another volatile access.
pub inline fn read(comptime T: type, addr: usize) T {
return @as(*const volatile T, @ptrFromInt(addr)).*;
}
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
pub inline fn write(comptime T: type, addr: usize, value: T) void {
@as(*volatile T, @ptrFromInt(addr)).* = value;
}
/// Full barrier: all loads and stores before it are globally visible before any after
/// it. Use when an MMIO write must complete before a following read.
pub inline fn mb() void {
switch (builtin.target.cpu.arch) {
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
else => @compileError("mmio.mb: unsupported architecture"),
}
}
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
/// wake, before reading what the device wrote to shared memory.
pub inline fn rmb() void {
switch (builtin.target.cpu.arch) {
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
else => @compileError("mmio.rmb: unsupported architecture"),
}
}
/// Write barrier: stores before it become visible before stores after it. Use between
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
pub inline fn wmb() void {
switch (builtin.target.cpu.arch) {
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
else => @compileError("mmio.wmb: unsupported architecture"),
}
}
test "barriers emit and registers round-trip through a RAM cell" {
// The barriers must at least assemble for the host arch; ordering can't be unit
// tested, but a missing/mistyped mnemonic is caught here.
wmb();
rmb();
mb();
var cell: u64 = 0;
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
}
+48
View File
@@ -0,0 +1,48 @@
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
//! physically contiguous, at a physical address the driver knows, uncacheable, and
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
const abi = @import("abi");
const sc = @import("system-call.zig");
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
/// opt-in for specific hardware — see `abi`.
pub const coherent: usize = abi.dma_coherent;
pub const write_combining: usize = abi.dma_write_combining;
pub const below_4g: usize = abi.dma_below_4g;
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
/// to program into the device's descriptor-ring / base registers.
pub const Region = struct {
virtual: usize,
physical: usize,
};
inline fn failed(r: usize) bool {
return r > ~@as(usize, 0) - 4095;
}
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
/// return values — the virtual address in rax, the physical address in rdx — so it
/// needs a hand-written stub.
pub fn alloc(len: usize, flags: usize) ?Region {
var rax: usize = undefined;
var rdx: usize = undefined; // out: physical address
asm volatile ("syscall"
: [rax] "={rax}" (rax),
[rdx] "={rdx}" (rdx),
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
[a0] "{rdi}" (len),
[a1] "{rsi}" (flags),
: .{ .rcx = true, .r11 = true, .memory = true });
if (failed(rax)) return null;
return .{ .virtual = rax, .physical = rdx };
}
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
pub fn free(virtual: usize, len: usize) void {
_ = sc.systemCall2(.dma_free, virtual, len);
}
+2
View File
@@ -20,6 +20,8 @@ pub const vfs_protocol = @import("vfs-protocol");
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
/// Device access for drivers: enumerate/claim/mmioMap.
pub const device = @import("device.zig");
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
pub const dma = @import("dma.zig");
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
pub const panic = start.panic;
+9
View File
@@ -37,9 +37,18 @@ pub const SystemCall = enum(u64) {
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
system_spawn = 17, // system_spawn(name_ptr, name_len) -> 0: start a named initial-ramdisk binary as a new ring-3 process
dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory
dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc
_,
};
/// `dma_alloc` flags. `coherent` (uncacheable) is the portable default; the others are
/// opt-in for specific hardware. `write_combining` needs PAT programming (not yet — it
/// currently falls back to coherent); see docs/driver-model.md (M14).
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
/// **asynchronous notification** (today: a device interrupt bound with `irq_bind`)
/// rather than a message from a client. There is no payload and no reply owed; the
+23 -15
View File
@@ -29,6 +29,7 @@
//! 0x108 TIMER0_COMPARATOR
const runtime = @import("runtime");
const mmio = @import("mmio");
const device = runtime.device;
const ipc = runtime.ipc;
@@ -50,8 +51,15 @@ const tn_route_mask: u64 = 0x1F << tn_route_shift;
/// Interrupts to observe before declaring victory.
const target_ticks = 5;
fn register(base: usize, off: usize) *volatile u64 {
return @ptrFromInt(base + off);
/// Read/write a 64-bit HPET register through the typed volatile MMIO layer (/lib/mmio).
/// The HPET is pure MMIO with no DMA, and on x86 its grant is strong-uncacheable (so
/// UC writes are already ordered) — no barriers are needed here; the point is the
/// typed, arch-portable access every driver should use.
inline fn rd(base: usize, off: usize) u64 {
return mmio.read(u64, base + off);
}
inline fn wr(base: usize, off: usize, value: u64) void {
mmio.write(u64, base + off, value);
}
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
@@ -66,16 +74,16 @@ fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
// Skip comparator children a bus driver may have published below the block
// (see system/drivers/bus/bus.zig) — we want the register block itself.
if (d.parent != device.no_parent) continue;
var mmio: ?u64 = null;
var mmio_index: ?u64 = null;
var irq: ?u64 = null;
for (0..d.resource_count) |j| {
switch (d.resources[j].kind) {
@intFromEnum(device.ResourceKind.memory) => mmio = mmio orelse j,
@intFromEnum(device.ResourceKind.memory) => mmio_index = mmio_index orelse j,
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
else => {},
}
}
if (mmio) |m| if (irq) |i| {
if (mmio_index) |m| if (irq) |i| {
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
};
}
@@ -114,7 +122,7 @@ pub fn main() void {
// --- program the hardware ------------------------------------------------
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
const femtos_per_tick = register(base, register_general_cap).* >> 32;
const femtos_per_tick = rd(base, register_general_cap) >> 32;
if (femtos_per_tick == 0) {
_ = runtime.system.write("hpet: bad HPET period\n");
return;
@@ -122,21 +130,21 @@ pub fn main() void {
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
// Stop the counter and take the legacy route off while we reconfigure.
register(base, register_general_configuration).* &= ~(configuration_enable | configuration_leg_rt);
wr(base, register_general_configuration, rd(base, register_general_configuration) & ~(configuration_enable | configuration_leg_rt));
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
// we simply re-arm from the driver on each interrupt, which is what a tickless
// timer driver does anyway.
var t0 = register(base, register_timer0_configuration).*;
var t0 = rd(base, register_timer0_configuration);
t0 &= ~(tn_route_mask | tn_type_periodic);
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
register(base, register_timer0_configuration).* = t0;
wr(base, register_timer0_configuration, t0);
// Clear any stale assertion, then arm ~100 ms out and start the counter.
register(base, register_int_status).* = 1;
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
register(base, register_general_configuration).* |= configuration_enable;
wr(base, register_int_status, 1);
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
wr(base, register_general_configuration, rd(base, register_general_configuration) | configuration_enable);
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
_ = runtime.system.write("hpet: irq_bind failed\n");
@@ -157,17 +165,17 @@ pub fn main() void {
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
// line is still asserted and unmasking would refire immediately.
register(base, register_int_status).* = 1;
wr(base, register_int_status, 1);
count += 1;
if (count < target_ticks) {
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
wr(base, register_timer0_comparator, rd(base, register_main_counter) + ticks_per_ms * 100);
} else {
// Last one: stop the source rather than re-arming, so the line is left
// both quiet *and* unmasked by the ack below. Re-arming here would leave
// a pending interrupt that nobody is waiting for, and the ISR would mask
// the line again a moment later.
register(base, register_timer0_configuration).* &= ~tn_int_enb;
wr(base, register_timer0_configuration, rd(base, register_timer0_configuration) & ~tn_int_enb);
}
_ = runtime.system.write("hpet: irq\n");
@@ -168,6 +168,12 @@ pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void
paging.mapUserDeviceInto(root, virtual, physical, len);
}
/// Map coherent DMA RAM into address space `root`: strong-uncacheable, RW+NX, but
/// reclaimed on teardown (real RAM, not MMIO). For dma_alloc.
pub fn mapUserDmaInto(root: u64, virtual: u64, physical: u64, len: u64) void {
paging.mapUserDmaInto(root, virtual, physical, len);
}
/// Map a page into the kernel address space (non-executable). For the heap, etc.
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
paging.map(virtual, physical, writable);
@@ -273,6 +273,30 @@ pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void
}
}
/// Map `[physical, physical+len)` into the user half rooted at `pml4` as **coherent
/// DMA memory**: strong-uncacheable (PCD|PWT — a device reads/writes this RAM without
/// snooping the CPU caches) but, unlike `mapUserDeviceInto`, **without** `device_grant`
/// — because these frames are real RAM from `pmm.allocContiguous`, so teardown
/// (`freeSubtree`) must return them to the allocator like any other user page. RW + NX.
/// The caller aligns `virtual`/`physical` and places `virtual` in the DMA arena.
pub fn mapUserDmaInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
const flags: u64 = present | user | writable | no_execute | pcd | pwt;
const first = physical & ~@as(u64, page_size - 1);
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
var off: u64 = 0;
while (first + off <= last) : (off += page_size) {
const v = virtual + off;
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
const pdpt = descendUser(pml4e);
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
const pd = descendUser(pdpte);
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
const pt = descendUser(pde);
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
invalidate(v);
}
}
/// Create a new address space: a fresh PML4 with an empty user half and the
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
+28
View File
@@ -173,3 +173,31 @@ pub fn free(address: u64) void {
used_frames -= 1;
if (f < next_hint) next_hint = f;
}
/// Allocate `count` physically **contiguous** frames whose highest byte is below
/// `max_phys` (pass `~0` for no limit; use a real limit for DMA engines with 32-bit
/// addressing). Returns the physical base, or null if no free run of that size fits.
/// A DMA descriptor ring needs contiguity, a known physical address, and pinning —
/// none of which one-frame `alloc` gives. Linear scan for a run of clear bits: fine
/// for the small rings DMA needs; a buddy allocator is a later optimisation. The
/// frames are freed individually with `free`, so there is no bespoke free path.
pub fn allocContiguous(count: usize, max_phys: u64) ?u64 {
if (count == 0) return null;
const limit: usize = @intCast(@min(@as(u64, total_frames), max_phys / page_size));
var start: usize = 1; // frame 0 stays reserved as the "none" address
while (start + count <= limit) {
if (isUsed(start)) {
start += 1;
continue;
}
var run: usize = 0;
while (run < count and !isUsed(start + run)) : (run += 1) {}
if (run == count) {
for (0..count) |i| setUsed(start + i);
used_frames += count;
return @as(u64, start) * page_size;
}
start += run + 1; // the frame at start+run is used; skip past it
}
return null; // no contiguous run of `count` frames below max_phys
}
+67
View File
@@ -62,6 +62,13 @@ pub const user_half_end: u64 = 0x0000_8000_0000_0000;
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
/// The DMA arena: where `dma_alloc` places coherent DMA buffers, in PML4[228] — a
/// user-exclusive region distinct from the MMIO arena. Unlike MMIO grants these back
/// real RAM (contiguous frames), so they are reclaimed on teardown. Per-process cursor
/// in `Task.dma_map_next`.
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
/// chunks, so this bound is generous; it also caps the frame scratch array below.
const maximum_mmap_pages = 256;
@@ -150,6 +157,8 @@ fn system_call(state: *architecture.CpuState) void {
.irq_ack => systemIrqAck(state),
.device_register => systemDeviceRegister(state),
.system_spawn => systemSpawn(state),
.dma_alloc => systemDmaAlloc(state),
.dma_free => systemDmaFree(state),
_ => fail(state),
}
}
@@ -261,6 +270,64 @@ fn systemMmioMap(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
}
/// dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): grant `len` bytes (rounded up to
/// whole pages) of DMA-capable memory — physically contiguous, zeroed, pinned, and
/// strong-uncacheable (coherent) — mapping it into the caller's DMA arena and handing
/// back both the virtual address to touch and the physical address to program into the
/// device. This is what `sysMmap` can't do: mmap frames are scattered, cacheable, and
/// their physical address is never disclosed. `dma_below_4g` caps the physical address
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
/// until PAT is programmed. See docs/driver-model.md (M14).
fn systemDmaAlloc(state: *architecture.CpuState) void {
const len = architecture.systemCallArg(state, 0);
const flags = architecture.systemCallArg(state, 1);
const t = scheduler.current();
if (t.aspace == 0 or len == 0) return fail(state);
const pages: usize = @intCast((len + page_size - 1) / page_size);
const max_phys: u64 = if (flags & abi.dma_below_4g != 0) (@as(u64, 4) << 30) else ~@as(u64, 0);
const phys = pmm.allocContiguous(pages, max_phys) orelse return fail(state);
if (t.dma_map_next == 0) t.dma_map_next = dma_arena_base;
const base_v = t.dma_map_next;
if (base_v + pages * page_size > dma_arena_end) {
for (0..pages) |i| pmm.free(phys + i * page_size); // arena exhausted; give the frames back
return fail(state);
}
// Zero through the physmap (the frames aren't mapped in the caller yet), then map.
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
@memset(kernel_view[0 .. pages * page_size], 0);
architecture.mapUserDmaInto(t.aspace, base_v, phys, pages * page_size);
t.dma_map_next = base_v + pages * page_size;
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
architecture.setSystemCallResult2(state, phys); // physical address for the device
}
/// dma_free(vaddr, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
/// it can never unmap-and-free the caller's stack, heap, or an MMIO grant; only pages
/// actually mapped are freed (an unmapped hole is skipped). Teardown also reclaims any
/// DMA pages left mapped at exit (they carry no `device_grant`, so `freeSubtree` frees
/// them as ordinary RAM), so a driver that just dies leaks nothing.
fn systemDmaFree(state: *architecture.CpuState) void {
const base_v = architecture.systemCallArg(state, 0);
const len = architecture.systemCallArg(state, 1);
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const pages: usize = @intCast((len + page_size - 1) / page_size);
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
for (0..pages) |i| {
const va = base_v + i * page_size;
if (architecture.translate(t.aspace, va)) |phys| {
architecture.unmapUserPageInto(t.aspace, va);
pmm.free(phys);
}
}
architecture.setSystemCallResult(state, 0);
}
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
/// this process has claimed. The bus-driver primitive: a process that owns a bus
/// enumerates it and hands each device it finds to the table, where a class driver
+1
View File
@@ -66,6 +66,7 @@ pub const Task = struct {
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
ipc_reply_cap: u64 = 0,
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
+49
View File
@@ -84,6 +84,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
ipcCallTest();
} else if (eql(case, "ipc-cap")) {
capabilityTest();
} else if (eql(case, "dma")) {
dmaTest();
} else if (eql(case, "smp")) {
smpTest();
} else if (eql(case, "affinity")) {
@@ -935,6 +937,53 @@ fn capabilityTest() void {
result();
}
/// DMA memory (M14): the properties a bus-mastering driver needs — physically
/// contiguous, a known physical address, correct cacheability, pinned, and reclaimed
/// on teardown. Exercises the kernel mechanism directly (`pmm.allocContiguous` +
/// `mapUserDmaInto`); the `dma_alloc`/`dma_free` syscalls are thin wrappers over it,
/// following the tested `mmap`/`mmio_map` shape, and land their first real use with the
/// first DMA driver.
fn dmaTest() void {
log("DANOS-TEST-BEGIN: dma\n", .{});
const base_free = pmm.stats().free_frames;
// A contiguous run: aligned, and it consumed exactly that many frames.
const frames = 4;
const phys = pmm.allocContiguous(frames, ~@as(u64, 0)) orelse {
check("allocContiguous(4) succeeded", false);
result();
return;
};
check("contiguous run is page-aligned", phys % abi.page_size == 0);
check("contiguous run consumed 4 frames", pmm.stats().free_frames == base_free - frames);
// The below-4G cap is honoured (legacy 32-bit DMA engines).
const low = pmm.allocContiguous(2, @as(u64, 4) << 30) orelse 0;
check("below-4G run stays under 4 GiB", low != 0 and low + 2 * abi.page_size <= (@as(u64, 4) << 30));
// Map the run into a fresh address space as coherent DMA and translate each page
// back: the same physical run, in order — proving contiguity and the mapping.
const aspace = architecture.createAddressSpace().?;
architecture.mapUserDmaInto(aspace, process.dma_arena_base, phys, frames * abi.page_size);
var mapped_ok = true;
for (0..frames) |i| {
const va = process.dma_arena_base + i * abi.page_size;
const got = architecture.translate(aspace, va) orelse {
mapped_ok = false;
break;
};
if (got != phys + i * abi.page_size) mapped_ok = false;
}
check("DMA pages translate to the contiguous physical run", mapped_ok);
// Teardown must reclaim the DMA RAM (the leaves carry no device_grant, so
// freeSubtree frees them as ordinary frames) — a driver that just dies leaks none.
architecture.destroyAddressSpace(aspace);
for (0..2) |i| pmm.free(low + i * abi.page_size);
check("no frames leaked after DMA teardown", pmm.stats().free_frames == base_free);
result();
}
var proc_worker_run: bool = true;
var proc_worker_ran: bool = false;
+5
View File
@@ -130,6 +130,11 @@ CASES = [
{"name": "ipc-cap",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# DMA memory (M14): contiguous frame allocation, below-4G cap, coherent mapping,
# and reclaim on teardown.
{"name": "dma",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Parallelism: needs more than one core, so this case boots with -smp 4.
{"name": "smp",
"smp": 4,