The mechanism behind delegation, which device-manager.md named as the step after hello: the device manager claims what discovery seeded and hands each device to the driver it matched, so assignment stops being first-come-first-served. It is a MOVE, not a copy. A claim is exclusive (driver-model.md, invariant 1), so the giver stops holding the device the instant the receiver starts. That is why this is a new syscall rather than the M13 capability path, where a passed handle is shared refcounted — exclusivity cannot be expressed that way. The kernel's whole rule is that you may give away what you hold. It has no notion of which task is the device manager and deliberately gains none: a binary name inside the kernel is not something that cannot safely live in user space. A recipient that does not exist is refused, because a device moved to nobody would be unreachable for the rest of the boot — nothing un-holds a device but task death. Three errnos, each naming its own rule: ENODEV no such device, EPERM you do not hold it, ESRCH no such recipient. Nothing uses it yet. The five claimants move across one at a time in D4-D5, so the suite stays green throughout and a regression names the driver that caused it. Ten assertions, verified to discriminate: removing the ownership check flips four of them, including the giveaway that an illegal transfer then blocks the legitimate claim behind it. Suite 116 -> 117.
289 lines
14 KiB
Zig
289 lines
14 KiB
Zig
//! library/device/driver — the driver author's interface: enumerate the kernel's device
|
|
//! table, claim a device, map its MMIO, bind its interrupt (the claim is the capability the
|
|
//! kernel checks before mapping registers or routing an IRQ), and say `hello` to the device
|
|
//! manager at startup. The whole kernel + manager surface a driver needs, in one import.
|
|
|
|
const std = @import("std");
|
|
const abi = @import("abi");
|
|
const device_abi = @import("device-abi");
|
|
const sc = @import("system-call");
|
|
const channel = @import("channel");
|
|
const envelope = @import("envelope");
|
|
const ipc = @import("ipc");
|
|
const time = @import("time");
|
|
const device_manager_protocol = @import("device-manager-protocol");
|
|
|
|
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
|
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
|
pub const DeviceClass = device_abi.DeviceClass;
|
|
pub const ResourceKind = device_abi.ResourceKind;
|
|
|
|
inline fn failed(r: usize) bool {
|
|
return r > ~@as(usize, 0) - 4095;
|
|
}
|
|
|
|
/// The errno inside a failed return. Only meaningful when `failed(r)`.
|
|
inline fn errnoOf(r: usize) i64 {
|
|
return -@as(i64, @bitCast(r));
|
|
}
|
|
|
|
// The envelope restates the kernel's errno numbering by hand, because the
|
|
// `protocol` package deliberately depends on nothing (so it cannot import `abi`).
|
|
// This module is one of the few that can see both halves, so it is where they are
|
|
// held together: drift becomes a compile error here rather than a driver reporting
|
|
// the wrong reason for a refusal. Anything linking a driver compiles this.
|
|
comptime {
|
|
if (envelope.ENOENT != abi.ENOENT) @compileError("envelope.ENOENT has drifted from abi.ENOENT");
|
|
if (envelope.ENOSPC != abi.ENOSPC) @compileError("envelope.ENOSPC has drifted from abi.ENOSPC");
|
|
if (envelope.EPERM != abi.EPERM) @compileError("envelope.EPERM has drifted from abi.EPERM");
|
|
if (envelope.ENOSYS != abi.ENOSYS) @compileError("envelope.ENOSYS has drifted from abi.ENOSYS");
|
|
if (envelope.EPROTO != abi.EPROTO) @compileError("envelope.EPROTO has drifted from abi.EPROTO");
|
|
if (envelope.EBUSY != abi.EBUSY) @compileError("envelope.EBUSY has drifted from abi.EBUSY");
|
|
}
|
|
|
|
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
|
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
|
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
|
}
|
|
|
|
/// Why a `transfer` failed. `NotHeld` is the interesting one — it means the caller tried
|
|
/// to give away a device it does not have, which is the whole rule.
|
|
pub const TransferError = error{ NoSuchDevice, NotHeld, NoSuchTask, Refused };
|
|
|
|
/// Give device `id` to task `to`. **A move, not a copy** — a claim is exclusive, so the
|
|
/// caller stops holding it. This is how the device manager hands a driver the device it
|
|
/// matched, replacing first-come-first-served claiming with policy
|
|
/// (docs/os-development/device-authority.md).
|
|
pub fn transfer(id: u64, to: u32) TransferError!void {
|
|
const r = sc.systemCall2(.device_transfer, id, to);
|
|
if (!failed(r)) return;
|
|
return switch (errnoOf(r)) {
|
|
abi.ENODEV => error.NoSuchDevice,
|
|
abi.EPERM => error.NotHeld,
|
|
abi.ESRCH => error.NoSuchTask,
|
|
else => error.Refused,
|
|
};
|
|
}
|
|
|
|
/// Why a `claim` failed. Worth distinguishing: `AlreadyClaimed` means back off and
|
|
/// let the owner have it, `NoSuchDevice` means this id is stale and the caller should
|
|
/// re-enumerate, and `NotConfined` means the machine could not place the device under
|
|
/// IOMMU translation — the claim was rolled back, and that one is a fault report, not
|
|
/// a retry. `Refused` is an errno this library does not know a name for.
|
|
pub const ClaimError = error{ NoSuchDevice, AlreadyClaimed, NotConfined, Refused };
|
|
|
|
/// Take exclusive ownership of device `id`.
|
|
pub fn claim(id: u64) ClaimError!void {
|
|
const r = sc.systemCall1(.device_claim, id);
|
|
if (!failed(r)) return;
|
|
return switch (errnoOf(r)) {
|
|
abi.ENODEV => error.NoSuchDevice,
|
|
abi.EBUSY => error.AlreadyClaimed,
|
|
abi.ECONFINE => error.NotConfined,
|
|
else => error.Refused,
|
|
};
|
|
}
|
|
|
|
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
|
/// `device_id` into this address space; returns the register base virtual address.
|
|
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
|
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
|
return if (failed(r)) null else r;
|
|
}
|
|
|
|
/// `DeviceDescriptor.parent` for a device with no parent.
|
|
pub const no_parent = device_abi.no_parent;
|
|
|
|
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. Set this on
|
|
/// descriptors passed to `register` unless the child really is one.
|
|
pub const no_pci_class = device_abi.no_pci_class;
|
|
|
|
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
|
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
|
/// that class of device can `claim` it — that is how a bus hands off a device.
|
|
///
|
|
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
|
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
|
/// anything else, because a device descriptor is a licence to map physical memory and
|
|
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
|
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
|
/// through its controller, not by MMIO.
|
|
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) RegisterError!u64 {
|
|
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
|
if (!failed(r)) return r;
|
|
return switch (errnoOf(r)) {
|
|
abi.ENOSPC => error.TableFull,
|
|
abi.ENODEV => error.NoSuchParent,
|
|
abi.EPERM => error.NotYourParent,
|
|
abi.E2BIG => error.TooManyResources,
|
|
abi.ECHILDREN => error.ParentFull,
|
|
abi.ERANGE => error.NotContained,
|
|
abi.EFAULT => error.BadDescriptor,
|
|
else => error.Refused,
|
|
};
|
|
}
|
|
|
|
/// Why a `register` failed. These are not interchangeable and a bus driver should
|
|
/// say which one it hit: `ParentFull` and `TableFull` are different ceilings with
|
|
/// different fixes, and `NotContained` is not a ceiling at all — it means the child
|
|
/// resource escaped the window the parent actually owns. Reporting all of them as
|
|
/// one refusal is what made an AMD desktop boot with no USB and no storage, and gave
|
|
/// no way to tell which of three causes it was (docs/fixed-bounds-audit.md).
|
|
pub const RegisterError = error{
|
|
TableFull, // the kernel's device table is full, machine-wide
|
|
ParentFull, // this parent already holds as many children as it can
|
|
NoSuchParent, // no device with that id
|
|
NotYourParent, // that device exists but this process has not claimed it
|
|
TooManyResources, // the descriptor declares more resources than one device may hold
|
|
NotContained, // a resource escapes the parent's window
|
|
BadDescriptor, // the descriptor pointer did not read back
|
|
Refused, // an errno this library does not know a name for
|
|
};
|
|
|
|
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
|
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
|
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
|
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
|
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
|
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
|
}
|
|
|
|
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
|
/// status register holds its line asserted) — the kernel left the line masked
|
|
/// precisely because it could not do that for you. Skip it and the interrupt never
|
|
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
|
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
|
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
|
}
|
|
|
|
/// The Message-Signalled Interrupt address/data a driver programs into its device's
|
|
/// MSI capability. The device raises the interrupt by writing `data` to `address`.
|
|
pub const Msi = struct { address: u64, data: u32 };
|
|
|
|
/// Set up MSI for a claimed device: the kernel allocates a per-device edge-triggered
|
|
/// vector, binds it to `endpoint` (delivered like `irqBind`, but with no mask and no
|
|
/// `irqAck` cycle), and returns the (address, data) to write into the device's MSI
|
|
/// capability — found by mmio_mapping the device's ECAM config space (resource 0) and
|
|
/// walking its capability list. Returns null on failure. Two return values (address in
|
|
/// rax, data in rdx), so a hand-written stub.
|
|
pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
|
var rax: usize = undefined;
|
|
var rdx: usize = undefined;
|
|
asm volatile ("syscall"
|
|
: [rax] "={rax}" (rax),
|
|
[rdx] "={rdx}" (rdx),
|
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.msi_bind)),
|
|
[a0] "{rdi}" (device_id),
|
|
[a1] "{rsi}" (endpoint),
|
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
|
if (failed(rax)) return null;
|
|
return .{ .address = rax, .data = @intCast(rdx) };
|
|
}
|
|
|
|
/// Map a delegated DMA-region (or shared-memory) capability into a claimed device's
|
|
/// IOMMU domain, so the device may DMA to that buffer. The caller must own `device_id`
|
|
/// and hold `handle` (received over IPC or from its own `dma.alloc(.. | shareable)`).
|
|
/// Idempotent. Returns true on success (and trivially when no IOMMU is present).
|
|
pub fn dmaBind(device_id: u64, handle: usize) bool {
|
|
return !failed(sc.systemCall2(.dma_bind, device_id, handle));
|
|
}
|
|
|
|
/// Unmap a previously `dmaBind`'d buffer from the device's domain.
|
|
pub fn dmaUnbind(device_id: u64, handle: usize) bool {
|
|
return !failed(sc.systemCall2(.dma_unbind, device_id, handle));
|
|
}
|
|
|
|
/// Drain and log any pending IOMMU translation faults, returning the count seen. A
|
|
/// diagnostic: a driver that suspects its device attempted an out-of-domain DMA (or a
|
|
/// test proving enforcement) forces the hardware's fault records to the log now. Returns
|
|
/// 0 when no IOMMU is present.
|
|
pub fn iommuFaultDrain() usize {
|
|
return sc.systemCall0(.iommu_fault_drain);
|
|
}
|
|
|
|
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
|
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
|
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
|
/// access is a syscall, which is fine for the low-rate hardware that needs it. Returns
|
|
/// null if the capability check fails (device not claimed, wrong resource, out of
|
|
/// range). A device that decodes no data returns all-ones, which is a valid value, not
|
|
/// a failure.
|
|
pub fn ioRead(device_id: u64, resource_index: u64, offset: u64, width: u8) ?u32 {
|
|
const r = sc.systemCall4(.io_read, device_id, resource_index, offset, width);
|
|
return if (failed(r)) null else @intCast(r);
|
|
}
|
|
|
|
/// Write `value` (its low `width` bytes, 1/2/4) to a port in a claimed device's
|
|
/// `io_port` resource, at byte `offset`. Same capability gate as `ioRead`.
|
|
pub fn ioWrite(device_id: u64, resource_index: u64, offset: u64, width: u8, value: u32) bool {
|
|
return !failed(sc.systemCall5(.io_write, device_id, resource_index, offset, width, value));
|
|
}
|
|
|
|
/// Find DeviceDescription by hid
|
|
///
|
|
/// Utility function for driver development
|
|
pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const u8) ?DeviceDescriptor {
|
|
const total = enumerate(buffer);
|
|
const n = @min(total, buffer.len);
|
|
for (@as([]DeviceDescriptor, buffer[0..n])) |d| {
|
|
const hid_haystack = d.hid[0..@intCast(d.hid_len)];
|
|
if (std.mem.eql(u8, hid_haystack, hid_needle)) {
|
|
return d;
|
|
}
|
|
}
|
|
|
|
return null;
|
|
}
|
|
|
|
// --- device-manager handshake (folded in from the former device-manager.zig) ---
|
|
|
|
/// What kind of driver is announcing itself (a bus that reports children, or a leaf
|
|
/// device). Re-exported so callers name it without importing the protocol.
|
|
pub const Role = device_manager_protocol.Role;
|
|
|
|
const lookup_attempts: u32 = 100;
|
|
const lookup_pause_ms: u64 = 20;
|
|
|
|
/// Say hello to the device manager and return its endpoint, or null if there is no manager
|
|
/// (best-effort standalone bring-up) or it refused the handshake. Bus drivers keep the handle
|
|
/// to report children through; a driver that runs fine unsupervised discards it with `_ =`,
|
|
/// and one that requires supervision bails on null. Logs the outcome itself.
|
|
///
|
|
/// The device this driver was assigned is the packet's `Header.target` — the manager's
|
|
/// object addressing, so `no_device` here is a driver that serves none.
|
|
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
|
var attempts: u32 = 0;
|
|
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
|
if (channel.openEndpoint("device-manager")) |handle| break handle;
|
|
time.sleepMillis(lookup_pause_ms);
|
|
} else {
|
|
std.log.info("no device manager to hello", .{});
|
|
return null;
|
|
};
|
|
|
|
var packet: [device_manager_protocol.message_maximum]u8 = undefined;
|
|
const framed = device_manager_protocol.Protocol.encodeRequest(
|
|
.hello,
|
|
device_id,
|
|
.{ .role = @intFromEnum(role) },
|
|
&.{},
|
|
&packet,
|
|
) orelse return null;
|
|
|
|
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
|
const length = ipc.call(manager, framed, &reply) catch {
|
|
std.log.info("hello call failed", .{});
|
|
return null;
|
|
};
|
|
const status = envelope.statusOf(reply[0..length]) orelse {
|
|
std.log.info("hello answered nothing readable", .{});
|
|
return null;
|
|
};
|
|
if (status.status != 0) {
|
|
std.log.info("hello refused", .{});
|
|
return null;
|
|
}
|
|
std.log.info("hello acknowledged", .{});
|
|
return manager;
|
|
}
|