Re-organize the source tree as a monorepo mirroring the FHS
The source layout now mirrors the runtime filesystem hierarchy
(docs/danos-file-system-hierarchy-FSH.md): what lives under system/ in the
source is what a running danos represents under /system. Each service and
driver is a sub-project directory that is its own Zig module — cross-project
references go by module name, never by a path into another project's files.
Moves (all git mv, history preserved):
- src/ -> system/ (danos internals; the self-representation)
root.zig -> danos.zig (the kernel<->user contract module)
kernel/arch/ -> kernel/architecture/ (arch -> architecture)
device/ -> devices/ (what /system/devices reflects)
boot/ -> /boot (the loaders, top level)
- sbin/ -> split by role:
init, vfs -> system/services/<name>/<name>.zig
hpetd, busd -> system/drivers/<name>/<name>.zig
vfs-test -> system/services/vfs/vfs-test.zig (inside the vfs project)
- lib/ -> library/runtime/ (room for other libraries beside runtime)
The VFS wire protocol becomes its own module, system/services/vfs/protocol.zig
("vfs-protocol"): the vfs sub-project exposes its interface, and the runtime's
file layer imports it by name. First instance of the "protocol module" pattern
(docs/driver-model.md); usb/block will expose theirs the same way.
Also: fix a naming-standard violation in the protocol — Op -> Operation (and
req -> request, _pad -> _padding). Docs updated: /system/services added to the
FHS doc, a repository-layout section added to the docs index, and stale source
paths swept across comments and docs.
Runtime boot paths are unchanged (the bootloader still loads /sbin/init);
aligning the runtime filesystem to the FHS is a separate follow-up. Suite 35/35
plus host tests green.
This commit is contained in:
@@ -0,0 +1,256 @@
|
||||
//! Shared definitions that form the contract between a bootloader
|
||||
//! (boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (system/kernel/main.zig).
|
||||
//!
|
||||
//! Both binaries import this as the "danos" module, so the handoff layout is
|
||||
//! defined in exactly one place.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Calling convention for the bootloader→kernel jump. Pinned to SystemV so it does
|
||||
/// not depend on each binary's target default: the UEFI bootloader's C
|
||||
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
||||
/// SystemV (first arg in RDI). Both reference this to agree on where `*BootInformation`
|
||||
/// is passed.
|
||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||
|
||||
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
||||
pub const PixelFormat = enum(u32) {
|
||||
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
||||
rgbx,
|
||||
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
||||
bgrx,
|
||||
};
|
||||
|
||||
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
||||
/// `pitch` bytes between the start of one row and the next (which may be larger
|
||||
/// than `width * 4` due to hardware padding).
|
||||
///
|
||||
/// A `base` of 0 means **no framebuffer** — the firmware exposed no Graphics
|
||||
/// Output Protocol (a headless server, say). The kernel must treat on-screen
|
||||
/// output as optional and never assume a framebuffer exists.
|
||||
pub const Framebuffer = extern struct {
|
||||
base: usize, // the memory address where pixel data starts (0 = none)
|
||||
width: u32, // visible pixels per row (e.g. 1920)
|
||||
height: u32, // visible rows (e.g. 1080)
|
||||
pitch: u32, // bytes from the start of one row to the start of the next
|
||||
format: PixelFormat,
|
||||
|
||||
/// Whether a usable framebuffer was handed over.
|
||||
pub fn present(self: Framebuffer) bool {
|
||||
return self.base != 0 and self.width != 0 and self.height != 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// Page size the memory map is measured in. 4 KiB on every architecture danos
|
||||
/// targets so far.
|
||||
pub const page_size = 4096;
|
||||
|
||||
/// The kernel's virtual-memory layout (higher-half). The kernel is linked at
|
||||
/// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the
|
||||
/// device MMIO windows) is also mapped at `physmap_base + physical`, so the kernel
|
||||
/// can reach any physical address by adding a constant. The low half is left
|
||||
/// entirely to user space.
|
||||
///
|
||||
/// user image + stack : 0x0000_7000_0000_0000 (PML4[224], low half)
|
||||
/// kernel heap : 0xFFFF_8000_0000_0000 (PML4[256])
|
||||
/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + physical
|
||||
/// kernel image : 0xFFFF_FFFF_8000_0000 (PML4[511])
|
||||
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
||||
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
||||
|
||||
/// The kernel system_call numbers — the single source of truth shared by the kernel
|
||||
/// dispatcher (system/kernel/process.zig) and the user runtime library, so the two
|
||||
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||
pub const SystemCall = enum(u64) {
|
||||
exit = 0, // exit(code): end the calling process
|
||||
yield = 1, // yield(): give up the rest of this quantum
|
||||
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||
create_endpoint = 6, // create_endpoint() -> handle: a new IPC endpoint
|
||||
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||
_,
|
||||
};
|
||||
|
||||
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||
/// **asynchronous notification** (today: a device interrupt bound with `irq_bind`)
|
||||
/// rather than a message from a client. There is no payload and no reply owed; the
|
||||
/// low bits carry the source, a GSI. Shared so the kernel's ISR and the driver's
|
||||
/// event loop can't disagree about which bit means "the hardware spoke".
|
||||
pub const notify_badge_bit: u64 = 1 << 63;
|
||||
|
||||
/// A device class, mirroring system/devices/device-model.zig's `DeviceClass` **in order**
|
||||
/// (its `@intFromEnum` values cross the system_call boundary in `DeviceDescriptor.class`).
|
||||
/// Keep the two in sync.
|
||||
pub const DeviceClass = enum(u32) {
|
||||
root,
|
||||
processor,
|
||||
interrupt_controller,
|
||||
timer,
|
||||
pci_host_bridge,
|
||||
pci_device,
|
||||
acpi_device,
|
||||
unknown,
|
||||
};
|
||||
|
||||
/// A resource kind, mirroring system/devices/device-model.zig's `ResourceKind` in order.
|
||||
pub const ResourceKind = enum(u32) {
|
||||
memory,
|
||||
io_port,
|
||||
irq,
|
||||
bus_range,
|
||||
};
|
||||
|
||||
/// One device resource, as handed to a user-space driver (flat, extern).
|
||||
pub const ResourceDescriptor = extern struct {
|
||||
kind: u64, // a ResourceKind value
|
||||
start: u64,
|
||||
len: u64,
|
||||
};
|
||||
|
||||
pub const maximum_device_resources = 8;
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||
pub const no_parent: u64 = ~@as(u64, 0);
|
||||
|
||||
/// A device, as snapshotted for user space by `device_enumerate`. A driver scans
|
||||
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
||||
///
|
||||
/// `parent` makes the table a tree rather than a list, which is what a **bus driver**
|
||||
/// needs: it claims the bus, finds the devices below it, and publishes any it
|
||||
/// discovers itself with `device_register`. A registered child's resources must lie
|
||||
/// within its parent's (the kernel enforces this) — that containment is what makes
|
||||
/// delegation safe, since a device descriptor is otherwise a licence to map physical
|
||||
/// memory.
|
||||
pub const DeviceDescriptor = extern struct {
|
||||
id: u64,
|
||||
parent: u64, // a device id, or `no_parent`
|
||||
class: u64, // a DeviceClass value
|
||||
hid_len: u64,
|
||||
resource_count: u64,
|
||||
hid: [8]u8,
|
||||
resources: [maximum_device_resources]ResourceDescriptor,
|
||||
};
|
||||
|
||||
/// Well-known IPC service ids for the bootstrap name registry (create_endpoint +
|
||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||
pub const ServiceId = enum(u32) {
|
||||
vfs = 1,
|
||||
_,
|
||||
};
|
||||
|
||||
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||
pub const prot_read: u64 = 1;
|
||||
pub const prot_write: u64 = 2;
|
||||
pub const prot_exec: u64 = 4;
|
||||
|
||||
/// Physical address -> its virtual address in the physmap. The single way the
|
||||
/// kernel dereferences a physical address once paging is up.
|
||||
///
|
||||
/// **Hazard:** valid only once the (bootstrap or final) page tables are live.
|
||||
/// The bootloader may use the *constant* `physmap_base` to build those tables,
|
||||
/// but must not call this to dereference memory before its own CR3 is loaded —
|
||||
/// it runs under the firmware's identity map, where these addresses are unmapped.
|
||||
pub inline fn physicalToVirtual(physical: u64) u64 {
|
||||
return physical + physmap_base;
|
||||
}
|
||||
|
||||
/// Physmap virtual address -> physical. Inverse of `physicalToVirtual`; for producing
|
||||
/// the physical address of something the kernel holds a physmap pointer to
|
||||
/// (e.g. a page-table frame for CR3, a post-mortem breadcrumb's RAM location).
|
||||
pub inline fn virtualToPhysical(virtual: u64) u64 {
|
||||
return virtual - physmap_base;
|
||||
}
|
||||
|
||||
/// danos's own classification of a span of physical memory — deliberately not
|
||||
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
||||
/// native memory description into these kinds, so the kernel never learns what
|
||||
/// booted it. [[architecture]] keeps the same discipline for CPU code.
|
||||
pub const MemoryKind = enum(u32) {
|
||||
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
||||
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
||||
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
||||
/// has to know about boot-protocol-specific "reclaimable" states.
|
||||
usable,
|
||||
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
||||
/// never hand out.
|
||||
reserved,
|
||||
/// ACPI tables: parse, then reclaim.
|
||||
acpi_tables,
|
||||
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
||||
acpi_nvs,
|
||||
/// Not backed by RAM: memory-mapped device registers or a reserved
|
||||
/// address-space window (e.g. PCIe configuration space). Kept distinct from
|
||||
/// `reserved` so RAM accounting doesn't count device address space.
|
||||
mmio,
|
||||
};
|
||||
|
||||
/// One contiguous span of physical memory. Because danos defines this layout
|
||||
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
||||
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
||||
/// firmware's variable descriptor-stride to worry about.
|
||||
pub const MemoryRegion = extern struct {
|
||||
base: u64, // physical start address
|
||||
pages: u64, // length in `page_size` units
|
||||
kind: MemoryKind,
|
||||
_pad: u32 = 0,
|
||||
};
|
||||
|
||||
/// The physical memory layout handed to the kernel: a pointer to an array of
|
||||
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
||||
pub const MemoryMap = extern struct {
|
||||
regions: usize, // address of a `[len]MemoryRegion`
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
||||
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
||||
/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virtual` is the higher-half link address;
|
||||
/// `physical` is where the loader actually placed the segment (they differ once the
|
||||
/// kernel links high — the loader records the real load address here).
|
||||
pub const KernelSegment = extern struct {
|
||||
virtual: u64,
|
||||
physical: u64,
|
||||
pages: u64,
|
||||
flags: u32,
|
||||
_pad: u32 = 0,
|
||||
};
|
||||
|
||||
/// Handoff structure the bootloader fills in and passes to the kernel's
|
||||
/// `_start` in RDI (the first argument under the SystemV AMD64 C ABI).
|
||||
pub const BootInformation = extern struct {
|
||||
framebuffer: Framebuffer,
|
||||
memory_map: MemoryMap,
|
||||
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
||||
kernel_segments: [8]KernelSegment,
|
||||
kernel_segment_count: u32,
|
||||
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
||||
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||
acpi_rsdp: u64 = 0,
|
||||
/// The raw `/sbin/init` ELF image, read off the boot volume by the loader
|
||||
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||
/// kernel boots without user space. Grows into a full initrd handoff later.
|
||||
init_base: u64 = 0,
|
||||
init_len: u64 = 0,
|
||||
/// The initrd image (a bundle of extra user binaries — the VFS server and
|
||||
/// device drivers), read off the boot volume into memory that survives the
|
||||
/// handoff, same as `init` above. 0/0 = no initrd. See system/initrd.zig.
|
||||
initrd_base: u64 = 0,
|
||||
initrd_len: u64 = 0,
|
||||
};
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,208 @@
|
||||
//! AML (ACPI Machine Language) — the bytecode in the DSDT and SSDTs that describes
|
||||
//! the parts of the machine the static tables don't.
|
||||
//!
|
||||
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
||||
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
||||
//! method bodies and field/region layout. `interp.zig` then *evaluates* control
|
||||
//! methods on demand — running operators, control flow, and OperationRegion field
|
||||
//! access — so callers can resolve device status (`_STA`), current resource
|
||||
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
||||
|
||||
const std = @import("std");
|
||||
const opcode = @import("opcodes.zig");
|
||||
const parser = @import("parser.zig");
|
||||
|
||||
pub const Namespace = @import("namespace.zig").Namespace;
|
||||
pub const Node = @import("namespace.zig").Node;
|
||||
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||
|
||||
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
||||
/// enough for device discovery. See `interp.zig`.
|
||||
pub const Interpreter = @import("interp.zig").Interpreter;
|
||||
pub const Object = @import("interp.zig").Object;
|
||||
pub const EvaluateHal = @import("interp.zig").Hal;
|
||||
|
||||
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
||||
pub const SleepType = struct {
|
||||
slp_typ_a: u8,
|
||||
slp_typ_b: u8,
|
||||
};
|
||||
|
||||
pub const ParseResult = struct {
|
||||
namespace: Namespace,
|
||||
/// Bytes the parser consumed across all blocks...
|
||||
consumed: usize,
|
||||
/// ...out of this many. A clean full traversal has `consumed == total`.
|
||||
total: usize,
|
||||
};
|
||||
|
||||
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
||||
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
||||
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
||||
var namespace = try Namespace.init(allocator);
|
||||
var consumed: usize = 0;
|
||||
var total: usize = 0;
|
||||
for (blocks) |block| {
|
||||
var p = parser.Parser.init(block, &namespace);
|
||||
consumed += p.parseAll();
|
||||
total += block.len;
|
||||
}
|
||||
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
||||
}
|
||||
|
||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
||||
const segment = [4]u8{ '_', 'S', '0' + state, '_' };
|
||||
const node = namespace.resolve(namespace.root, false, 0, &.{segment}) orelse return null;
|
||||
if (node.kind != .name) return null;
|
||||
return parseSleepPackage(node.value);
|
||||
}
|
||||
|
||||
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
||||
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
||||
fn parseSleepPackage(value: []const u8) ?SleepType {
|
||||
if (value.len == 0 or value[0] != opcode.package_opcode) return null;
|
||||
var p: usize = 1;
|
||||
p += packageLengthSize(value, p) orelse return null;
|
||||
if (p >= value.len) return null;
|
||||
const number_elements = value[p];
|
||||
p += 1;
|
||||
|
||||
const a: u8 = if (number_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||
const b: u8 = if (number_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
||||
}
|
||||
|
||||
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
||||
fn packageLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||
if (p >= bytes.len) return null;
|
||||
const follow: usize = bytes[p] >> 6;
|
||||
if (p + 1 + follow > bytes.len) return null;
|
||||
return 1 + follow;
|
||||
}
|
||||
|
||||
/// Read one AML integer data object at `p`, advancing `p`.
|
||||
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
||||
if (p.* >= bytes.len) return null;
|
||||
const opcode_byte = bytes[p.*];
|
||||
p.* += 1;
|
||||
return switch (opcode_byte) {
|
||||
opcode.zero_opcode => 0,
|
||||
opcode.one_opcode => 1,
|
||||
opcode.ones_opcode => 0xFF,
|
||||
opcode.byte_prefix => readLittle(bytes, p, 1),
|
||||
opcode.word_prefix => readLittle(bytes, p, 2),
|
||||
opcode.dword_prefix => readLittle(bytes, p, 4),
|
||||
opcode.qword_prefix => readLittle(bytes, p, 8),
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
|
||||
fn readLittle(bytes: []const u8, p: *usize, n: usize) ?u64 {
|
||||
if (p.* + n > bytes.len) return null;
|
||||
var v: u64 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < n) : (k += 1) v |= @as(u64, bytes[p.* + k]) << @intCast(k * 8);
|
||||
p.* += n;
|
||||
return v;
|
||||
}
|
||||
|
||||
// --- tests ------------------------------------------------------------------
|
||||
|
||||
test "parses a nested namespace and finds the sleep package" {
|
||||
// A hand-assembled AML blob (all PkgLengths computed to be single-byte):
|
||||
// Name(_S5, Package(2){0x05, 0x00})
|
||||
// Scope(\_SB) { Device(PCI0) {
|
||||
// Name(_HID, 0x11)
|
||||
// Method(MTHD, 1) {}
|
||||
// Method(CALL, 0) { MTHD(Zero) } // invocation of a 1-arg method
|
||||
// } }
|
||||
// OperationRegion(DBG0, SystemIO, 0x0402, 1)
|
||||
// Field(DBG0, ...) { DBGB, 8 }
|
||||
const blob = [_]u8{
|
||||
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
||||
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
||||
// Scope(\_SB) packagelen=0x27
|
||||
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
||||
// Device(PCI0) packagelen=0x1F
|
||||
0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30,
|
||||
// Name(_HID, 0x11)
|
||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
||||
// Method(MTHD, flags=1) empty, packagelen=0x06
|
||||
0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01,
|
||||
// Method(CALL, flags=0) { MTHD(Zero) }, packagelen=0x0B
|
||||
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00,
|
||||
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
||||
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01,
|
||||
// Field(DBG0, flags=1) { DBGB, 8 }, packagelen=0x0B
|
||||
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08,
|
||||
};
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||
defer arena.deinit();
|
||||
var result = try parse(arena.allocator(), &.{&blob});
|
||||
|
||||
// Integrity: the parser consumed exactly the whole blob (no desync).
|
||||
try std.testing.expectEqual(blob.len, result.consumed);
|
||||
try std.testing.expectEqual(blob.len, result.total);
|
||||
|
||||
const namespace = &result.namespace;
|
||||
|
||||
// Expected top-level nodes.
|
||||
const sb = namespace.resolve(namespace.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
||||
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
||||
const pci0 = namespace.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
||||
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
||||
_ = namespace.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
||||
|
||||
// The 1-arg method's arg count was parsed from its flags byte.
|
||||
const mthd = namespace.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
||||
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
||||
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
||||
|
||||
// OperationRegion and the Field unit made it into the namespace.
|
||||
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
||||
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
||||
|
||||
// The sleep package decoded.
|
||||
const s5 = sleepState(namespace, 5) orelse return error.NoS5;
|
||||
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
||||
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
||||
}
|
||||
|
||||
fn noMap(physical: u64, _: u64, _: bool) u64 {
|
||||
return physical;
|
||||
}
|
||||
fn noRead(_: u8, _: u16) u32 {
|
||||
return 0;
|
||||
}
|
||||
fn noWrite(_: u8, _: u16, _: u32) void {}
|
||||
|
||||
test "interpreter runs a method with args, arithmetic, and control flow" {
|
||||
// Method(TST_, 1) {
|
||||
// Store(Arg0, Local0); Add(Local0, 5, Local0)
|
||||
// If (LGreater(Local0, 10)) { Return(One) }
|
||||
// Return(Zero)
|
||||
// }
|
||||
const blob = [_]u8{
|
||||
0x14, 0x18, 0x54, 0x53, 0x54, 0x5F, 0x01, // Method TST_, 1 arg
|
||||
0x70, 0x68, 0x60, // Store(Arg0, Local0)
|
||||
0x72, 0x60, 0x0A, 0x05, 0x60, // Add(Local0, 5, Local0)
|
||||
0xA0, 0x07, 0x94, 0x60, 0x0A, 0x0A, 0xA4, 0x01, // If(LGreater(Local0,10)) { Return(One) }
|
||||
0xA4, 0x00, // Return(Zero)
|
||||
};
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||
defer arena.deinit();
|
||||
var result = try parse(arena.allocator(), &.{&blob});
|
||||
const namespace = &result.namespace;
|
||||
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||
|
||||
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||
|
||||
const hi = try interpreter.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
||||
try std.testing.expectEqual(@as(u64, 1), try hi.asInteger());
|
||||
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
|
||||
}
|
||||
@@ -0,0 +1,736 @@
|
||||
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
||||
//! structural namespace. It executes control methods (their bodies captured by
|
||||
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
||||
//! device present), current resource settings (`_CRS`), and the operators, control
|
||||
//! flow, locals/args, and
|
||||
//! OperationRegion field access those methods reach for.
|
||||
//!
|
||||
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
||||
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
||||
//! access; CreateField buffer patching (the common current-resource-settings
|
||||
//! (`_CRS`) idiom); and field
|
||||
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
||||
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
||||
//! fall back — never a hard failure.
|
||||
|
||||
const std = @import("std");
|
||||
const opcode = @import("opcodes.zig");
|
||||
const Node = @import("namespace.zig").Node;
|
||||
const Namespace = @import("namespace.zig").Namespace;
|
||||
|
||||
/// Injected hardware access for OperationRegion reads/writes (the architecture VMM + pio).
|
||||
pub const Hal = struct {
|
||||
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||
pioRead: *const fn (width: u8, port: u16) u32,
|
||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||
};
|
||||
|
||||
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
||||
|
||||
/// A runtime AML value.
|
||||
pub const Object = union(enum) {
|
||||
uninitialized,
|
||||
integer: u64,
|
||||
buffer: []u8,
|
||||
string: []u8,
|
||||
package: []Object,
|
||||
reference: *Node,
|
||||
|
||||
pub fn asInteger(self: Object) Error!u64 {
|
||||
return switch (self) {
|
||||
.integer => |v| v,
|
||||
.buffer => |b| blk: {
|
||||
var v: u64 = 0;
|
||||
for (b, 0..) |byte, i| {
|
||||
if (i >= 8) break;
|
||||
v |= @as(u64, byte) << @intCast(i * 8);
|
||||
}
|
||||
break :blk v;
|
||||
},
|
||||
else => error.Unsupported,
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
const maximum_segments = 16;
|
||||
const NamePath = struct {
|
||||
rooted: bool = false,
|
||||
parents: u8 = 0,
|
||||
segments: [maximum_segments][4]u8 = undefined,
|
||||
count: usize = 0,
|
||||
fn slice(self: *const NamePath) []const [4]u8 {
|
||||
return self.segments[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
const Cursor = struct {
|
||||
b: []const u8,
|
||||
i: usize = 0,
|
||||
|
||||
fn eof(self: *Cursor) bool {
|
||||
return self.i >= self.b.len;
|
||||
}
|
||||
fn peek(self: *Cursor) ?u8 {
|
||||
return if (self.eof()) null else self.b[self.i];
|
||||
}
|
||||
fn byte(self: *Cursor) Error!u8 {
|
||||
if (self.eof()) return error.Truncated;
|
||||
const v = self.b[self.i];
|
||||
self.i += 1;
|
||||
return v;
|
||||
}
|
||||
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
||||
if (self.i + n > self.b.len) return error.Truncated;
|
||||
const s = self.b[self.i .. self.i + n];
|
||||
self.i += n;
|
||||
return s;
|
||||
}
|
||||
fn packageLength(self: *Cursor) Error!usize {
|
||||
const lead = try self.byte();
|
||||
const follow: usize = lead >> 6;
|
||||
if (follow == 0) return lead & 0x3F;
|
||||
var value: usize = lead & 0x0F;
|
||||
var k: usize = 0;
|
||||
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
||||
return value;
|
||||
}
|
||||
fn nameString(self: *Cursor) Error!NamePath {
|
||||
var name_path = NamePath{};
|
||||
if (self.peek() == opcode.root_char) {
|
||||
name_path.rooted = true;
|
||||
self.i += 1;
|
||||
} else {
|
||||
while (self.peek() == opcode.parent_prefix_char) : (self.i += 1) name_path.parents += 1;
|
||||
}
|
||||
const lead = self.peek() orelse return name_path;
|
||||
switch (lead) {
|
||||
0x00 => self.i += 1,
|
||||
opcode.dual_name_prefix => {
|
||||
self.i += 1;
|
||||
try self.segment(&name_path);
|
||||
try self.segment(&name_path);
|
||||
},
|
||||
opcode.multi_name_prefix => {
|
||||
self.i += 1;
|
||||
const count = try self.byte();
|
||||
var k: usize = 0;
|
||||
while (k < count) : (k += 1) try self.segment(&name_path);
|
||||
},
|
||||
else => try self.segment(&name_path),
|
||||
}
|
||||
return name_path;
|
||||
}
|
||||
fn segment(self: *Cursor, name_path: *NamePath) Error!void {
|
||||
const s = try self.take(4);
|
||||
if (name_path.count < maximum_segments) {
|
||||
name_path.segments[name_path.count] = s[0..4].*;
|
||||
name_path.count += 1;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const Frame = struct {
|
||||
args: [7]Object = .{.uninitialized} ** 7,
|
||||
locals: [8]Object = .{.uninitialized} ** 8,
|
||||
scope: *Node,
|
||||
ret: Object = .uninitialized,
|
||||
returned: bool = false,
|
||||
broke: bool = false,
|
||||
};
|
||||
|
||||
/// A CreateField binding: a name that indexes into a buffer object.
|
||||
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
|
||||
|
||||
pub const Interpreter = struct {
|
||||
namespace: *Namespace,
|
||||
hal: Hal,
|
||||
arena: std.mem.Allocator,
|
||||
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
||||
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||
/// CreateField bindings active for the current evaluation.
|
||||
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
|
||||
|
||||
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
|
||||
return .{ .namespace = namespace, .hal = hal, .arena = arena };
|
||||
}
|
||||
|
||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||
/// Field. Resets per-evaluation runtime state first.
|
||||
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||
self.dynamic_overrides.clearRetainingCapacity();
|
||||
self.fields.clearRetainingCapacity();
|
||||
return self.invoke(node, args);
|
||||
}
|
||||
|
||||
fn invoke(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||
switch (node.kind) {
|
||||
.method => {
|
||||
var frame = Frame{ .scope = node };
|
||||
for (args, 0..) |a, i| {
|
||||
if (i < frame.args.len) frame.args[i] = a;
|
||||
}
|
||||
var current = Cursor{ .b = node.value };
|
||||
try self.executeList(¤t, &frame);
|
||||
return frame.ret;
|
||||
},
|
||||
.name => {
|
||||
if (self.dynamic_overrides.get(node)) |o| return o;
|
||||
var current = Cursor{ .b = node.value };
|
||||
var frame = Frame{ .scope = node.parent orelse self.namespace.root };
|
||||
return self.term(¤t, &frame);
|
||||
},
|
||||
.field => return .{ .integer = try self.readField(node) },
|
||||
else => return .{ .reference = node },
|
||||
}
|
||||
}
|
||||
|
||||
/// Execute a TermList until it ends or the frame returns/breaks.
|
||||
fn executeList(self: *Interpreter, current: *Cursor, frame: *Frame) Error!void {
|
||||
while (!current.eof() and !frame.returned and !frame.broke) {
|
||||
_ = try self.term(current, frame);
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
||||
/// statements).
|
||||
fn term(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const lead = current.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) return self.nameReference(current, frame);
|
||||
_ = try current.byte();
|
||||
|
||||
return switch (lead) {
|
||||
opcode.zero_opcode => Object{ .integer = 0 },
|
||||
opcode.one_opcode => Object{ .integer = 1 },
|
||||
opcode.ones_opcode => Object{ .integer = ~@as(u64, 0) },
|
||||
opcode.byte_prefix => Object{ .integer = try self.readConstant(current, 1) },
|
||||
opcode.word_prefix => Object{ .integer = try self.readConstant(current, 2) },
|
||||
opcode.dword_prefix => Object{ .integer = try self.readConstant(current, 4) },
|
||||
opcode.qword_prefix => Object{ .integer = try self.readConstant(current, 8) },
|
||||
opcode.string_prefix => try self.readString(current),
|
||||
opcode.buffer_opcode => try self.buffer(current, frame),
|
||||
opcode.package_opcode, opcode.var_package_opcode => try self.package(current, frame, lead == opcode.var_package_opcode),
|
||||
|
||||
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode],
|
||||
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode],
|
||||
|
||||
opcode.return_opcode => blk: {
|
||||
frame.ret = try self.term(current, frame);
|
||||
frame.returned = true;
|
||||
break :blk .uninitialized;
|
||||
},
|
||||
opcode.break_opcode => blk: {
|
||||
frame.broke = true;
|
||||
break :blk .uninitialized;
|
||||
},
|
||||
opcode.continue_opcode, opcode.noop_opcode => .uninitialized,
|
||||
|
||||
opcode.if_opcode => try self.ifElse(current, frame),
|
||||
opcode.while_opcode => try self.whileLoop(current, frame),
|
||||
opcode.store_opcode => try self.store(current, frame),
|
||||
opcode.increment_opcode => try self.incDec(current, frame, 1),
|
||||
opcode.decrement_opcode => try self.incDec(current, frame, -1),
|
||||
|
||||
opcode.add_opcode => try self.binary(current, frame, .add),
|
||||
opcode.subtract_opcode => try self.binary(current, frame, .sub),
|
||||
opcode.multiply_opcode => try self.binary(current, frame, .mul),
|
||||
opcode.mod_opcode => try self.binary(current, frame, .mod),
|
||||
opcode.and_opcode => try self.binary(current, frame, .band),
|
||||
opcode.or_opcode => try self.binary(current, frame, .bor),
|
||||
opcode.xor_opcode => try self.binary(current, frame, .bxor),
|
||||
opcode.nand_opcode => try self.binary(current, frame, .nand),
|
||||
opcode.nor_opcode => try self.binary(current, frame, .nor),
|
||||
opcode.shift_left_opcode => try self.binary(current, frame, .shl),
|
||||
opcode.shift_right_opcode => try self.binary(current, frame, .shr),
|
||||
opcode.divide_opcode => try self.divide(current, frame),
|
||||
|
||||
opcode.land_opcode => try self.logic2(current, frame, .land),
|
||||
opcode.lor_opcode => try self.logic2(current, frame, .lor),
|
||||
opcode.lequal_opcode => try self.logic2(current, frame, .eq),
|
||||
opcode.lgreater_opcode => try self.logic2(current, frame, .gt),
|
||||
opcode.lless_opcode => try self.logic2(current, frame, .lt),
|
||||
opcode.lnot_opcode => try self.lnot(current, frame),
|
||||
|
||||
opcode.not_opcode => blk: {
|
||||
const v = try self.evaluateInteger(current, frame);
|
||||
const r = ~v;
|
||||
try self.storeTarget(current, frame, .{ .integer = r });
|
||||
break :blk .{ .integer = r };
|
||||
},
|
||||
|
||||
opcode.size_of_opcode => try self.sizeOf(current, frame),
|
||||
opcode.index_opcode => try self.index(current, frame),
|
||||
opcode.dereference_of_opcode => try self.dereferenceOf(current, frame),
|
||||
opcode.to_integer_opcode => blk: {
|
||||
const v = try self.evaluateInteger(current, frame);
|
||||
try self.storeTarget(current, frame, .{ .integer = v });
|
||||
break :blk .{ .integer = v };
|
||||
},
|
||||
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
|
||||
|
||||
opcode.extended_opcode_prefix => try self.ext(current, frame),
|
||||
|
||||
// CreateXField: source, index, name (bit widths differ by op)
|
||||
opcode.create_bit_field_opcode => try self.createField(current, frame, 1),
|
||||
opcode.create_byte_field_opcode => try self.createField(current, frame, 8),
|
||||
opcode.create_word_field_opcode => try self.createField(current, frame, 16),
|
||||
opcode.create_dword_field_opcode => try self.createField(current, frame, 32),
|
||||
opcode.create_qword_field_opcode => try self.createField(current, frame, 64),
|
||||
|
||||
else => error.Unsupported,
|
||||
};
|
||||
}
|
||||
|
||||
// --- name references ----------------------------------------------------
|
||||
|
||||
fn nameReference(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const name_path = try current.nameString();
|
||||
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse
|
||||
return .uninitialized; // unknown name -> treat as uninitialised
|
||||
switch (node.kind) {
|
||||
.method => {
|
||||
var argbuf: [7]Object = undefined;
|
||||
var i: usize = 0;
|
||||
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(current, frame);
|
||||
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
||||
},
|
||||
.field => return .{ .integer = try self.readField(node) },
|
||||
.name => return self.invoke(node, &.{}),
|
||||
else => return .{ .reference = node },
|
||||
}
|
||||
}
|
||||
|
||||
// --- data objects -------------------------------------------------------
|
||||
|
||||
fn readConstant(self: *Interpreter, current: *Cursor, n: usize) Error!u64 {
|
||||
_ = self;
|
||||
const bytes = try current.take(n);
|
||||
var v: u64 = 0;
|
||||
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
||||
return v;
|
||||
}
|
||||
|
||||
fn readString(self: *Interpreter, current: *Cursor) Error!Object {
|
||||
const start = current.i;
|
||||
while (current.peek()) |c| {
|
||||
current.i += 1;
|
||||
if (c == 0) break;
|
||||
}
|
||||
const raw = current.b[start .. current.i - 1];
|
||||
const s = try self.arena.dupe(u8, raw);
|
||||
return .{ .string = s };
|
||||
}
|
||||
|
||||
fn buffer(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = current.i;
|
||||
const len = try current.packageLength();
|
||||
const end = @min(start + len, current.b.len);
|
||||
const size = try self.evaluateInteger(current, frame);
|
||||
const data = current.b[@min(current.i, end)..end];
|
||||
const bytes = try self.arena.alloc(u8, @intCast(size));
|
||||
@memset(bytes, 0);
|
||||
@memcpy(bytes[0..@min(bytes.len, data.len)], data[0..@min(bytes.len, data.len)]);
|
||||
current.i = end;
|
||||
return .{ .buffer = bytes };
|
||||
}
|
||||
|
||||
fn package(self: *Interpreter, current: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
||||
const start = current.i;
|
||||
const len = try current.packageLength();
|
||||
const end = @min(start + len, current.b.len);
|
||||
const count: usize = if (variable) @intCast(try self.evaluateInteger(current, frame)) else try current.byte();
|
||||
const elems = try self.arena.alloc(Object, count);
|
||||
var i: usize = 0;
|
||||
while (i < count and current.i < end) : (i += 1) elems[i] = try self.term(current, frame);
|
||||
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
||||
current.i = end;
|
||||
return .{ .package = elems };
|
||||
}
|
||||
|
||||
// --- operators ----------------------------------------------------------
|
||||
|
||||
const BinaryOperation = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
||||
|
||||
fn binary(self: *Interpreter, current: *Cursor, frame: *Frame, kind: BinaryOperation) Error!Object {
|
||||
const a = try self.evaluateInteger(current, frame);
|
||||
const b = try self.evaluateInteger(current, frame);
|
||||
const r: u64 = switch (kind) {
|
||||
.add => a +% b,
|
||||
.sub => a -% b,
|
||||
.mul => a *% b,
|
||||
.mod => if (b == 0) return error.DivByZero else a % b,
|
||||
.band => a & b,
|
||||
.bor => a | b,
|
||||
.bxor => a ^ b,
|
||||
.nand => ~(a & b),
|
||||
.nor => ~(a | b),
|
||||
.shl => if (b >= 64) 0 else a << @intCast(b),
|
||||
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
||||
};
|
||||
try self.storeTarget(current, frame, .{ .integer = r });
|
||||
return .{ .integer = r };
|
||||
}
|
||||
|
||||
fn divide(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const a = try self.evaluateInteger(current, frame);
|
||||
const b = try self.evaluateInteger(current, frame);
|
||||
if (b == 0) return error.DivByZero;
|
||||
try self.storeTarget(current, frame, .{ .integer = a % b }); // remainder target
|
||||
try self.storeTarget(current, frame, .{ .integer = a / b }); // quotient target
|
||||
return .{ .integer = a / b };
|
||||
}
|
||||
|
||||
const LogicOperation = enum { land, lor, eq, gt, lt };
|
||||
|
||||
fn logic2(self: *Interpreter, current: *Cursor, frame: *Frame, kind: LogicOperation) Error!Object {
|
||||
const a = try self.evaluateInteger(current, frame);
|
||||
const b = try self.evaluateInteger(current, frame);
|
||||
const r = switch (kind) {
|
||||
.land => a != 0 and b != 0,
|
||||
.lor => a != 0 or b != 0,
|
||||
.eq => a == b,
|
||||
.gt => a > b,
|
||||
.lt => a < b,
|
||||
};
|
||||
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
||||
}
|
||||
|
||||
fn lnot(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
// 0x92 0x93/94/95 are the compound comparisons.
|
||||
const b = current.peek() orelse return error.Truncated;
|
||||
switch (b) {
|
||||
opcode.lnot.not_equal => {
|
||||
current.i += 1;
|
||||
const x = try self.evaluateInteger(current, frame);
|
||||
const y = try self.evaluateInteger(current, frame);
|
||||
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
opcode.lnot.less_equal => {
|
||||
current.i += 1;
|
||||
const x = try self.evaluateInteger(current, frame);
|
||||
const y = try self.evaluateInteger(current, frame);
|
||||
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
opcode.lnot.greater_equal => {
|
||||
current.i += 1;
|
||||
const x = try self.evaluateInteger(current, frame);
|
||||
const y = try self.evaluateInteger(current, frame);
|
||||
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
else => {
|
||||
const x = try self.evaluateInteger(current, frame);
|
||||
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn incDec(self: *Interpreter, current: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
||||
// Operand is a SuperName that is both read and written.
|
||||
const save = current.i;
|
||||
const current_value = try self.term(current, frame);
|
||||
const v = try current_value.asInteger();
|
||||
const r = if (delta > 0) v +% 1 else v -% 1;
|
||||
var tcur = Cursor{ .b = current.b, .i = save };
|
||||
try self.storeInto(&tcur, frame, .{ .integer = r });
|
||||
return .{ .integer = r };
|
||||
}
|
||||
|
||||
fn sizeOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(current, frame);
|
||||
return .{ .integer = switch (o) {
|
||||
.buffer => |b| b.len,
|
||||
.string => |s| s.len,
|
||||
.package => |p| p.len,
|
||||
else => 0,
|
||||
} };
|
||||
}
|
||||
|
||||
fn passThroughUnary(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(current, frame);
|
||||
try self.storeTarget(current, frame, o);
|
||||
return o;
|
||||
}
|
||||
|
||||
fn index(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const source = try self.term(current, frame);
|
||||
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||
// Optional target (a reference); we don't materialise references, so store
|
||||
// the indexed value if a target is present.
|
||||
const value: Object = switch (source) {
|
||||
.buffer => |b| .{ .integer = if (element_index < b.len) b[element_index] else 0 },
|
||||
.package => |p| if (element_index < p.len) p[element_index] else .uninitialized,
|
||||
.string => |s| .{ .integer = if (element_index < s.len) s[element_index] else 0 },
|
||||
else => .uninitialized,
|
||||
};
|
||||
try self.storeTarget(current, frame, value);
|
||||
return value;
|
||||
}
|
||||
|
||||
fn dereferenceOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(current, frame);
|
||||
return switch (o) {
|
||||
.reference => |n| self.invoke(n, &.{}),
|
||||
else => o,
|
||||
};
|
||||
}
|
||||
|
||||
// --- control flow -------------------------------------------------------
|
||||
|
||||
fn ifElse(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = current.i;
|
||||
const end = @min(start + try current.packageLength(), current.b.len);
|
||||
const cond = try self.evaluateInteger(current, frame);
|
||||
if (cond != 0) {
|
||||
var body = Cursor{ .b = current.b[0..end], .i = current.i };
|
||||
try self.executeList(&body, frame);
|
||||
current.i = end;
|
||||
// Skip a trailing Else.
|
||||
if (current.peek() == opcode.else_opcode) {
|
||||
current.i += 1;
|
||||
const es = current.i;
|
||||
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||
current.i = ee;
|
||||
}
|
||||
} else {
|
||||
current.i = end;
|
||||
if (current.peek() == opcode.else_opcode) {
|
||||
current.i += 1;
|
||||
const es = current.i;
|
||||
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||
var body = Cursor{ .b = current.b[0..ee], .i = current.i };
|
||||
try self.executeList(&body, frame);
|
||||
current.i = ee;
|
||||
}
|
||||
}
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
fn whileLoop(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = current.i;
|
||||
const end = @min(start + try current.packageLength(), current.b.len);
|
||||
const pred_at = current.i;
|
||||
var guard: usize = 0;
|
||||
while (guard < 100_000) : (guard += 1) {
|
||||
var pc = Cursor{ .b = current.b[0..end], .i = pred_at };
|
||||
const cond = try self.evaluateInteger(&pc, frame);
|
||||
if (cond == 0) break;
|
||||
var body = Cursor{ .b = current.b[0..end], .i = pc.i };
|
||||
try self.executeList(&body, frame);
|
||||
if (frame.returned) break;
|
||||
if (frame.broke) {
|
||||
frame.broke = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
current.i = end;
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
// --- store --------------------------------------------------------------
|
||||
|
||||
fn store(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const value = try self.term(current, frame);
|
||||
try self.storeInto(current, frame, value);
|
||||
return value;
|
||||
}
|
||||
|
||||
/// A Store *target* that may be NullName (no store).
|
||||
fn storeTarget(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
if (current.peek() == 0x00) {
|
||||
current.i += 1; // NullName
|
||||
return;
|
||||
}
|
||||
try self.storeInto(current, frame, value);
|
||||
}
|
||||
|
||||
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
const lead = current.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) {
|
||||
const name_path = try current.nameString();
|
||||
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return;
|
||||
if (self.fields.get(node)) |buffer_field| {
|
||||
try self.writeBufferField(buffer_field, try value.asInteger());
|
||||
} else if (node.kind == .field) {
|
||||
try self.writeField(node, try value.asInteger());
|
||||
} else {
|
||||
try self.dynamic_overrides.put(self.arena, node, value);
|
||||
}
|
||||
return;
|
||||
}
|
||||
_ = try current.byte();
|
||||
switch (lead) {
|
||||
0x00 => {}, // NullName
|
||||
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode] = value,
|
||||
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode] = value,
|
||||
opcode.index_opcode => {
|
||||
const source = try self.term(current, frame);
|
||||
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||
switch (source) {
|
||||
.buffer => |b| if (element_index < b.len) {
|
||||
b[element_index] = @truncate(try value.asInteger());
|
||||
},
|
||||
.package => |p| if (element_index < p.len) {
|
||||
p[element_index] = value;
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
},
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
// --- CreateField (buffer patching) --------------------------------------
|
||||
|
||||
fn createField(self: *Interpreter, current: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
||||
const source = try self.term(current, frame); // source buffer (as a reference or value)
|
||||
const bit_index = try self.evaluateInteger(current, frame);
|
||||
const name_path = try current.nameString();
|
||||
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return .uninitialized;
|
||||
|
||||
// Bind the new name to the source buffer's node so stores land in it.
|
||||
const buffer_node: *Node = switch (source) {
|
||||
.reference => |n| n,
|
||||
else => return .uninitialized,
|
||||
};
|
||||
// Materialise the buffer into `dynamic_overrides` so patches persist and are returned.
|
||||
if (self.dynamic_overrides.get(buffer_node) == null) {
|
||||
const value = try self.invoke(buffer_node, &.{});
|
||||
try self.dynamic_overrides.put(self.arena, buffer_node, value);
|
||||
}
|
||||
const byte_off: usize = @intCast(bit_index / 8);
|
||||
try self.fields.put(self.arena, node, .{ .buffer = buffer_node, .byte_off = byte_off, .bit_width = bit_width });
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
fn writeBufferField(self: *Interpreter, buffer_field: BufferField, value: u64) Error!void {
|
||||
const obj = self.dynamic_overrides.get(buffer_field.buffer) orelse return;
|
||||
const bytes = switch (obj) {
|
||||
.buffer => |b| b,
|
||||
else => return,
|
||||
};
|
||||
const byte_count = (buffer_field.bit_width + 7) / 8;
|
||||
var k: usize = 0;
|
||||
while (k < byte_count and buffer_field.byte_off + k < bytes.len) : (k += 1) {
|
||||
bytes[buffer_field.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
||||
}
|
||||
}
|
||||
|
||||
// --- OperationRegion field access ---------------------------------------
|
||||
|
||||
fn readField(self: *Interpreter, field: *Node) Error!u64 {
|
||||
const region = field.region orelse return error.Unsupported;
|
||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||
const base = try self.regionBase(region);
|
||||
const start_byte = base + field.bit_offset / 8;
|
||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||
const total = @as(usize, shift) + field.bit_width;
|
||||
const byte_count = (total + 7) / 8;
|
||||
var raw: u128 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < byte_count) : (k += 1) {
|
||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||
}
|
||||
const masked = (raw >> shift) & bitMask(field.bit_width);
|
||||
return @truncate(masked);
|
||||
}
|
||||
|
||||
fn writeField(self: *Interpreter, field: *Node, value: u64) Error!void {
|
||||
const region = field.region orelse return error.Unsupported;
|
||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||
const base = try self.regionBase(region);
|
||||
const start_byte = base + field.bit_offset / 8;
|
||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||
const total = @as(usize, shift) + field.bit_width;
|
||||
const byte_count = (total + 7) / 8;
|
||||
// Read-modify-write byte by byte.
|
||||
var raw: u128 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < byte_count) : (k += 1) {
|
||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||
}
|
||||
const mask = bitMask(field.bit_width) << shift;
|
||||
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
||||
k = 0;
|
||||
while (k < byte_count) : (k += 1) {
|
||||
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
||||
}
|
||||
}
|
||||
|
||||
fn regionBase(self: *Interpreter, region: *Node) Error!u64 {
|
||||
var current = Cursor{ .b = region.region_offset_aml };
|
||||
var frame = Frame{ .scope = region.parent orelse self.namespace.root };
|
||||
return (try self.term(¤t, &frame)).asInteger();
|
||||
}
|
||||
|
||||
fn readRegionByte(self: *Interpreter, space: u8, address: u64) Error!u8 {
|
||||
switch (space) {
|
||||
0 => { // SystemMemory
|
||||
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||
const p: *align(1) const volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||
return p.*;
|
||||
},
|
||||
1 => return @truncate(self.hal.pioRead(1, @intCast(address & 0xFFFF))), // SystemIO
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
fn writeRegionByte(self: *Interpreter, space: u8, address: u64, value: u8) Error!void {
|
||||
switch (space) {
|
||||
0 => {
|
||||
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||
const p: *align(1) volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||
p.* = value;
|
||||
},
|
||||
1 => self.hal.pioWrite(1, @intCast(address & 0xFFFF), value),
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
// --- extended opcodes ---------------------------------------------------
|
||||
|
||||
fn ext(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const e = try current.byte();
|
||||
switch (e) {
|
||||
opcode.extended.debug => return .uninitialized,
|
||||
opcode.extended.revision => return .{ .integer = 2 },
|
||||
opcode.extended.timer => return .{ .integer = 0 },
|
||||
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
||||
opcode.extended.acquire => {
|
||||
_ = try self.term(current, frame); // mutex SuperName
|
||||
_ = try current.take(2); // timeout
|
||||
return .{ .integer = 0 }; // acquired
|
||||
},
|
||||
opcode.extended.release, opcode.extended.reset, opcode.extended.signal => {
|
||||
_ = try self.term(current, frame);
|
||||
return .uninitialized;
|
||||
},
|
||||
opcode.extended.wait => {
|
||||
_ = try self.term(current, frame);
|
||||
_ = try self.term(current, frame);
|
||||
return .{ .integer = 0 };
|
||||
},
|
||||
opcode.extended.sleep, opcode.extended.stall => {
|
||||
_ = try self.term(current, frame);
|
||||
return .uninitialized;
|
||||
},
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
fn evaluateInteger(self: *Interpreter, current: *Cursor, frame: *Frame) Error!u64 {
|
||||
return (try self.term(current, frame)).asInteger();
|
||||
}
|
||||
};
|
||||
|
||||
fn bitMask(width: u32) u128 {
|
||||
if (width >= 128) return ~@as(u128, 0);
|
||||
return (@as(u128, 1) << @intCast(width)) - 1;
|
||||
}
|
||||
|
||||
fn isNameStart(b: u8) bool {
|
||||
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||
b == opcode.name_char_underscore or
|
||||
b == opcode.root_char or
|
||||
b == opcode.parent_prefix_char or
|
||||
b == opcode.dual_name_prefix or
|
||||
b == opcode.multi_name_prefix;
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
//! The ACPI namespace the AML parser builds: a tree of named nodes, plus the name
|
||||
//! resolution rules the parser needs while it walks (so a method invocation can be
|
||||
//! resolved to its declaration to learn its argument count).
|
||||
//!
|
||||
//! Nodes are individually allocated and linked intrusively (first-child /
|
||||
//! next-sibling), the same shape as the device tree in `device.zig`.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub const NodeKind = enum {
|
||||
root,
|
||||
scope,
|
||||
device,
|
||||
method,
|
||||
name,
|
||||
region, // OperationRegion
|
||||
field, // a Field unit
|
||||
mutex,
|
||||
event,
|
||||
processor,
|
||||
power_resource,
|
||||
thermal_zone,
|
||||
alias,
|
||||
external,
|
||||
other,
|
||||
};
|
||||
|
||||
pub const Node = struct {
|
||||
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
||||
/// all-zero.
|
||||
segment: [4]u8 = .{ 0, 0, 0, 0 },
|
||||
kind: NodeKind = .other,
|
||||
/// For Method / External: the declared argument count (0..7). Used to resolve
|
||||
/// how many TermArgs a method invocation consumes.
|
||||
arg_count: u8 = 0,
|
||||
/// For Name: the AML bytes of its DataReferenceObject (so a value like a sleep
|
||||
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
||||
/// interpreted on demand by the evaluator. Empty otherwise.
|
||||
value: []const u8 = &.{},
|
||||
|
||||
// OperationRegion metadata (kind == .region): the address space, plus the AML
|
||||
// of the offset/length expressions (evaluated lazily, usually constants).
|
||||
region_space: u8 = 0,
|
||||
region_offset_aml: []const u8 = &.{},
|
||||
region_len_aml: []const u8 = &.{},
|
||||
|
||||
// Field-unit metadata (kind == .field): which region it lives in and its bit
|
||||
// position/width/access, so the evaluator can read/write it.
|
||||
region: ?*Node = null,
|
||||
bit_offset: u32 = 0,
|
||||
bit_width: u32 = 0,
|
||||
access_type: u8 = 0,
|
||||
|
||||
parent: ?*Node = null,
|
||||
first_child: ?*Node = null,
|
||||
next_sibling: ?*Node = null,
|
||||
|
||||
/// Depth-first count of this node and everything under it.
|
||||
pub fn subtreeCount(self: *const Node) usize {
|
||||
var n: usize = 1;
|
||||
var c = self.first_child;
|
||||
while (c) |child| : (c = child.next_sibling) n += child.subtreeCount();
|
||||
return n;
|
||||
}
|
||||
};
|
||||
|
||||
pub const Namespace = struct {
|
||||
allocator: std.mem.Allocator,
|
||||
root: *Node,
|
||||
|
||||
pub fn init(allocator: std.mem.Allocator) !Namespace {
|
||||
const root = try allocator.create(Node);
|
||||
root.* = .{ .kind = .root };
|
||||
return .{ .allocator = allocator, .root = root };
|
||||
}
|
||||
|
||||
pub fn nodeCount(self: *const Namespace) usize {
|
||||
return self.root.subtreeCount();
|
||||
}
|
||||
|
||||
fn findChild(parent: *Node, segment: [4]u8) ?*Node {
|
||||
var c = parent.first_child;
|
||||
while (c) |child| : (c = child.next_sibling) {
|
||||
if (std.mem.eql(u8, &child.segment, &segment)) return child;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The direct child of `node` named `segment`, or null. Unlike `resolve`, this does
|
||||
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
||||
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
||||
pub fn childOf(node: *Node, segment: [4]u8) ?*Node {
|
||||
return findChild(node, segment);
|
||||
}
|
||||
|
||||
fn newChild(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||
const n = try self.allocator.create(Node);
|
||||
n.* = .{ .segment = segment, .kind = kind, .parent = parent };
|
||||
// Append at the tail so a dump reads in declaration order.
|
||||
if (parent.first_child == null) {
|
||||
parent.first_child = n;
|
||||
} else {
|
||||
var current = parent.first_child.?;
|
||||
while (current.next_sibling) |sib| current = sib;
|
||||
current.next_sibling = n;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Create a Field unit node directly under `scope` (field units live in the
|
||||
/// scope of the Field/IndexField/BankField, not under the region).
|
||||
pub fn newFieldUnit(self: *Namespace, scope: *Node, segment: [4]u8) !*Node {
|
||||
return self.findOrCreate(scope, segment, .field);
|
||||
}
|
||||
|
||||
fn findOrCreate(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||
if (findChild(parent, segment)) |existing| {
|
||||
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
||||
// specific kind rather than downgrading to a plain scope.
|
||||
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
||||
return existing;
|
||||
}
|
||||
return self.newChild(parent, segment, kind);
|
||||
}
|
||||
|
||||
/// The node a definition's NameString names, creating any intermediate scopes.
|
||||
/// The final segment is created (or found) with `kind`; intermediates are
|
||||
/// scopes. Returns the namespace root for a NullName (empty path).
|
||||
pub fn place(
|
||||
self: *Namespace,
|
||||
current: *Node,
|
||||
rooted: bool,
|
||||
parents: u8,
|
||||
segments: []const [4]u8,
|
||||
kind: NodeKind,
|
||||
) !*Node {
|
||||
var base = startNode(self, current, rooted, parents);
|
||||
if (segments.len == 0) return base;
|
||||
var i: usize = 0;
|
||||
while (i + 1 < segments.len) : (i += 1) {
|
||||
base = try self.findOrCreate(base, segments[i], .scope);
|
||||
}
|
||||
return self.findOrCreate(base, segments[segments.len - 1], kind);
|
||||
}
|
||||
|
||||
/// Resolve a NameString *reference* to an existing node, or null. A single
|
||||
/// relative segment uses the ACPI search rule (walk up the ancestors); any
|
||||
/// rooted, parented, or multi-segment path is resolved exactly.
|
||||
pub fn resolve(
|
||||
self: *Namespace,
|
||||
current: *Node,
|
||||
rooted: bool,
|
||||
parents: u8,
|
||||
segments: []const [4]u8,
|
||||
) ?*Node {
|
||||
if (segments.len == 0) return null;
|
||||
|
||||
if (!rooted and parents == 0 and segments.len == 1) {
|
||||
// Search rule: this scope, then each ancestor up to the root.
|
||||
var scope: ?*Node = current;
|
||||
while (scope) |s| : (scope = s.parent) {
|
||||
if (findChild(s, segments[0])) |n| return n;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
var base = startNode(self, current, rooted, parents);
|
||||
for (segments) |segment| {
|
||||
base = findChild(base, segment) orelse return null;
|
||||
}
|
||||
return base;
|
||||
}
|
||||
|
||||
fn startNode(self: *Namespace, current: *Node, rooted: bool, parents: u8) *Node {
|
||||
if (rooted) return self.root;
|
||||
var base = current;
|
||||
var up = parents;
|
||||
while (up > 0) : (up -= 1) base = base.parent orelse self.root;
|
||||
return base;
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,137 @@
|
||||
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
||||
//!
|
||||
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
||||
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
||||
//! opcodes are `lnot_opcode` (0x92) followed by a second byte (see `lnot`).
|
||||
|
||||
// --- name / path characters -------------------------------------------------
|
||||
pub const zero_opcode = 0x00;
|
||||
pub const one_opcode = 0x01;
|
||||
pub const alias_opcode = 0x06;
|
||||
pub const name_opcode = 0x08;
|
||||
pub const byte_prefix = 0x0A;
|
||||
pub const word_prefix = 0x0B;
|
||||
pub const dword_prefix = 0x0C;
|
||||
pub const string_prefix = 0x0D;
|
||||
pub const qword_prefix = 0x0E;
|
||||
pub const scope_opcode = 0x10;
|
||||
pub const buffer_opcode = 0x11;
|
||||
pub const package_opcode = 0x12;
|
||||
pub const var_package_opcode = 0x13;
|
||||
pub const method_opcode = 0x14;
|
||||
pub const external_opcode = 0x15;
|
||||
|
||||
pub const dual_name_prefix = 0x2E;
|
||||
pub const multi_name_prefix = 0x2F;
|
||||
pub const extended_opcode_prefix = 0x5B;
|
||||
pub const root_char = 0x5C;
|
||||
pub const parent_prefix_char = 0x5E;
|
||||
pub const name_char_underscore = 0x5F;
|
||||
|
||||
pub const digit_char_start = 0x30;
|
||||
pub const digit_char_end = 0x39;
|
||||
pub const name_char_start = 0x41; // 'A'
|
||||
pub const name_char_end = 0x5A; // 'Z'
|
||||
|
||||
// --- locals / args ----------------------------------------------------------
|
||||
pub const local0_opcode = 0x60;
|
||||
pub const local7_opcode = 0x67;
|
||||
pub const arg0_opcode = 0x68;
|
||||
pub const arg6_opcode = 0x6E;
|
||||
|
||||
// --- store / references / arithmetic ---------------------------------------
|
||||
pub const store_opcode = 0x70;
|
||||
pub const ref_of_opcode = 0x71;
|
||||
pub const add_opcode = 0x72;
|
||||
pub const concat_opcode = 0x73;
|
||||
pub const subtract_opcode = 0x74;
|
||||
pub const increment_opcode = 0x75;
|
||||
pub const decrement_opcode = 0x76;
|
||||
pub const multiply_opcode = 0x77;
|
||||
pub const divide_opcode = 0x78;
|
||||
pub const shift_left_opcode = 0x79;
|
||||
pub const shift_right_opcode = 0x7A;
|
||||
pub const and_opcode = 0x7B;
|
||||
pub const nand_opcode = 0x7C;
|
||||
pub const or_opcode = 0x7D;
|
||||
pub const nor_opcode = 0x7E;
|
||||
pub const xor_opcode = 0x7F;
|
||||
pub const not_opcode = 0x80;
|
||||
pub const find_set_left_bit_opcode = 0x81;
|
||||
pub const find_set_right_bit_opcode = 0x82;
|
||||
pub const dereference_of_opcode = 0x83;
|
||||
pub const concat_resource_opcode = 0x84;
|
||||
pub const mod_opcode = 0x85;
|
||||
pub const notify_opcode = 0x86;
|
||||
pub const size_of_opcode = 0x87;
|
||||
pub const index_opcode = 0x88;
|
||||
pub const match_opcode = 0x89;
|
||||
pub const create_dword_field_opcode = 0x8A;
|
||||
pub const create_word_field_opcode = 0x8B;
|
||||
pub const create_byte_field_opcode = 0x8C;
|
||||
pub const create_bit_field_opcode = 0x8D;
|
||||
pub const object_type_opcode = 0x8E;
|
||||
pub const create_qword_field_opcode = 0x8F;
|
||||
|
||||
pub const land_opcode = 0x90;
|
||||
pub const lor_opcode = 0x91;
|
||||
pub const lnot_opcode = 0x92; // may be followed by a second byte (see `lnot`)
|
||||
pub const lequal_opcode = 0x93;
|
||||
pub const lgreater_opcode = 0x94;
|
||||
pub const lless_opcode = 0x95;
|
||||
pub const to_buffer_opcode = 0x96;
|
||||
pub const to_decimal_string_opcode = 0x97;
|
||||
pub const to_hex_string_opcode = 0x98;
|
||||
pub const to_integer_opcode = 0x99;
|
||||
pub const to_string_opcode = 0x9C;
|
||||
pub const copy_object_opcode = 0x9D;
|
||||
pub const mid_opcode = 0x9E;
|
||||
pub const continue_opcode = 0x9F;
|
||||
pub const if_opcode = 0xA0;
|
||||
pub const else_opcode = 0xA1;
|
||||
pub const while_opcode = 0xA2;
|
||||
pub const noop_opcode = 0xA3;
|
||||
pub const return_opcode = 0xA4;
|
||||
pub const break_opcode = 0xA5;
|
||||
pub const break_point_opcode = 0xCC;
|
||||
pub const ones_opcode = 0xFF;
|
||||
|
||||
/// Second bytes of the `lnot_opcode` (0x92) compound comparison opcodes.
|
||||
pub const lnot = struct {
|
||||
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
||||
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
||||
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
||||
};
|
||||
|
||||
/// Second bytes of extended opcodes (prefixed by `extended_opcode_prefix`, 0x5B).
|
||||
pub const extended = struct {
|
||||
pub const mutex = 0x01;
|
||||
pub const event = 0x02;
|
||||
pub const conditional_reference_of = 0x12;
|
||||
pub const create_field = 0x13;
|
||||
pub const load_table = 0x1F;
|
||||
pub const load = 0x20;
|
||||
pub const stall = 0x21;
|
||||
pub const sleep = 0x22;
|
||||
pub const acquire = 0x23;
|
||||
pub const signal = 0x24;
|
||||
pub const wait = 0x25;
|
||||
pub const reset = 0x26;
|
||||
pub const release = 0x27;
|
||||
pub const from_bcd = 0x28;
|
||||
pub const to_bcd = 0x29;
|
||||
pub const unload = 0x2A;
|
||||
pub const revision = 0x30;
|
||||
pub const debug = 0x31;
|
||||
pub const fatal = 0x32;
|
||||
pub const timer = 0x33;
|
||||
pub const operation_region = 0x80;
|
||||
pub const field = 0x81;
|
||||
pub const device = 0x82;
|
||||
pub const processor = 0x83;
|
||||
pub const power_resource = 0x84;
|
||||
pub const thermal_zone = 0x85;
|
||||
pub const index_field = 0x86;
|
||||
pub const bank_field = 0x87;
|
||||
pub const data_region = 0x88;
|
||||
};
|
||||
@@ -0,0 +1,517 @@
|
||||
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
||||
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
||||
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
||||
//! cursor stays aligned and every named object is recorded.
|
||||
//!
|
||||
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
||||
//! operand position is a call whose argument count is only known from the method's
|
||||
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
||||
//! `resolve` finds that declaration and tells us how many operands to consume.
|
||||
//!
|
||||
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
||||
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
||||
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
||||
//! *within* one such object; the enclosing walk realigns at the boundary.
|
||||
|
||||
const std = @import("std");
|
||||
const opcode = @import("opcodes.zig");
|
||||
const Namespace = @import("namespace.zig").Namespace;
|
||||
const Node = @import("namespace.zig").Node;
|
||||
const NodeKind = @import("namespace.zig").NodeKind;
|
||||
|
||||
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
||||
|
||||
const maximum_segments = 64;
|
||||
|
||||
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
||||
/// of 4-byte segments.
|
||||
const NamePath = struct {
|
||||
rooted: bool = false,
|
||||
parents: u8 = 0,
|
||||
segments: [maximum_segments][4]u8 = undefined,
|
||||
count: usize = 0,
|
||||
|
||||
fn slice(self: *const NamePath) []const [4]u8 {
|
||||
return self.segments[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
pub const Parser = struct {
|
||||
aml: []const u8,
|
||||
position: usize = 0,
|
||||
namespace: *Namespace,
|
||||
|
||||
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
||||
return .{ .aml = aml, .namespace = namespace };
|
||||
}
|
||||
|
||||
/// Parse the whole block as a TermList under the namespace root. Returns the
|
||||
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
||||
pub fn parseAll(self: *Parser) usize {
|
||||
self.termList(self.aml.len, self.namespace.root);
|
||||
return self.position;
|
||||
}
|
||||
|
||||
// --- cursor primitives --------------------------------------------------
|
||||
|
||||
fn eof(self: *Parser) bool {
|
||||
return self.position >= self.aml.len;
|
||||
}
|
||||
|
||||
fn peek(self: *Parser) ?u8 {
|
||||
return if (self.eof()) null else self.aml[self.position];
|
||||
}
|
||||
|
||||
fn readByte(self: *Parser) Error!u8 {
|
||||
if (self.eof()) return error.Truncated;
|
||||
const b = self.aml[self.position];
|
||||
self.position += 1;
|
||||
return b;
|
||||
}
|
||||
|
||||
fn skip(self: *Parser, n: usize) Error!void {
|
||||
if (self.position + n > self.aml.len) return error.Truncated;
|
||||
self.position += n;
|
||||
}
|
||||
|
||||
fn skipCString(self: *Parser) Error!void {
|
||||
while (true) {
|
||||
const b = try self.readByte();
|
||||
if (b == 0) return;
|
||||
}
|
||||
}
|
||||
|
||||
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
||||
/// follow; the value counts from the start of the PkgLength field.
|
||||
fn readPackageLength(self: *Parser) Error!usize {
|
||||
const lead = try self.readByte();
|
||||
const follow: usize = lead >> 6;
|
||||
if (follow == 0) return lead & 0x3F;
|
||||
var value: usize = lead & 0x0F;
|
||||
var i: usize = 0;
|
||||
while (i < follow) : (i += 1) {
|
||||
const b = try self.readByte();
|
||||
value |= @as(usize, b) << @intCast(4 + i * 8);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
fn readNameSegment(self: *Parser) Error![4]u8 {
|
||||
if (self.position + 4 > self.aml.len) return error.Truncated;
|
||||
const segment = self.aml[self.position..][0..4].*;
|
||||
self.position += 4;
|
||||
return segment;
|
||||
}
|
||||
|
||||
fn readNameString(self: *Parser) Error!NamePath {
|
||||
var name_path = NamePath{};
|
||||
// A NameString is either root-anchored or parent-relative, not both.
|
||||
if (self.peek() == opcode.root_char) {
|
||||
name_path.rooted = true;
|
||||
self.position += 1;
|
||||
} else {
|
||||
while (self.peek() == opcode.parent_prefix_char) : (self.position += 1) name_path.parents += 1;
|
||||
}
|
||||
|
||||
const lead = self.peek() orelse return name_path;
|
||||
switch (lead) {
|
||||
0x00 => self.position += 1, // NullName
|
||||
opcode.dual_name_prefix => {
|
||||
self.position += 1;
|
||||
try self.appendSegment(&name_path);
|
||||
try self.appendSegment(&name_path);
|
||||
},
|
||||
opcode.multi_name_prefix => {
|
||||
self.position += 1;
|
||||
const count = try self.readByte();
|
||||
var i: usize = 0;
|
||||
while (i < count) : (i += 1) try self.appendSegment(&name_path);
|
||||
},
|
||||
else => {
|
||||
if (isNameStart(lead)) try self.appendSegment(&name_path);
|
||||
},
|
||||
}
|
||||
return name_path;
|
||||
}
|
||||
|
||||
fn appendSegment(self: *Parser, name_path: *NamePath) Error!void {
|
||||
const segment = try self.readNameSegment();
|
||||
if (name_path.count < maximum_segments) {
|
||||
name_path.segments[name_path.count] = segment;
|
||||
name_path.count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// --- term list / object -------------------------------------------------
|
||||
|
||||
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
||||
/// the boundary rather than propagating — containment for the rare desync.
|
||||
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
||||
while (self.position < end) {
|
||||
self.object(scope) catch break;
|
||||
}
|
||||
self.position = end;
|
||||
}
|
||||
|
||||
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
||||
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
||||
/// purpose of advancing the cursor).
|
||||
fn object(self: *Parser, scope: *Node) Error!void {
|
||||
const lead = self.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) return self.nameInvocation(scope);
|
||||
|
||||
_ = try self.readByte();
|
||||
switch (lead) {
|
||||
// constants and no-operand statements
|
||||
opcode.zero_opcode, opcode.one_opcode, opcode.ones_opcode => {},
|
||||
opcode.noop_opcode, opcode.continue_opcode, opcode.break_opcode, opcode.break_point_opcode => {},
|
||||
opcode.local0_opcode...opcode.local7_opcode => {},
|
||||
opcode.arg0_opcode...opcode.arg6_opcode => {},
|
||||
|
||||
// literal data
|
||||
opcode.byte_prefix => try self.skip(1),
|
||||
opcode.word_prefix => try self.skip(2),
|
||||
opcode.dword_prefix => try self.skip(4),
|
||||
opcode.qword_prefix => try self.skip(8),
|
||||
opcode.string_prefix => try self.skipCString(),
|
||||
|
||||
// data containers (contents skipped via their PkgLength)
|
||||
opcode.buffer_opcode, opcode.package_opcode, opcode.var_package_opcode => try self.skipPackage(),
|
||||
|
||||
// namespace modifiers / named objects
|
||||
opcode.name_opcode => try self.parseName(scope),
|
||||
opcode.alias_opcode => try self.parseAlias(scope),
|
||||
opcode.scope_opcode => try self.parseScopeLike(scope, .scope),
|
||||
opcode.method_opcode => try self.parseMethod(scope),
|
||||
opcode.external_opcode => try self.parseExternal(scope),
|
||||
opcode.extended_opcode_prefix => try self.parseExtended(scope),
|
||||
|
||||
// control flow
|
||||
opcode.if_opcode => try self.parseIf(scope),
|
||||
opcode.else_opcode => try self.parseElse(scope),
|
||||
opcode.while_opcode => try self.parseWhile(scope),
|
||||
opcode.return_opcode => try self.object(scope),
|
||||
opcode.notify_opcode => try self.args(scope, 2),
|
||||
|
||||
// stores / references / unary+target
|
||||
opcode.store_opcode => try self.args(scope, 2),
|
||||
opcode.ref_of_opcode, opcode.dereference_of_opcode, opcode.size_of_opcode, opcode.object_type_opcode => try self.args(scope, 1),
|
||||
opcode.increment_opcode, opcode.decrement_opcode => try self.args(scope, 1),
|
||||
opcode.not_opcode, opcode.find_set_left_bit_opcode, opcode.find_set_right_bit_opcode => try self.args(scope, 2),
|
||||
opcode.to_buffer_opcode, opcode.to_decimal_string_opcode, opcode.to_hex_string_opcode, opcode.to_integer_opcode => try self.args(scope, 2),
|
||||
opcode.copy_object_opcode => try self.args(scope, 2),
|
||||
|
||||
// binary + target
|
||||
opcode.add_opcode, opcode.subtract_opcode, opcode.multiply_opcode, opcode.mod_opcode => try self.args(scope, 3),
|
||||
opcode.and_opcode, opcode.nand_opcode, opcode.or_opcode, opcode.nor_opcode, opcode.xor_opcode => try self.args(scope, 3),
|
||||
opcode.shift_left_opcode, opcode.shift_right_opcode, opcode.concat_opcode, opcode.concat_resource_opcode, opcode.index_opcode => try self.args(scope, 3),
|
||||
opcode.divide_opcode => try self.args(scope, 4),
|
||||
opcode.to_string_opcode => try self.args(scope, 3),
|
||||
opcode.mid_opcode => try self.args(scope, 4),
|
||||
|
||||
// logical
|
||||
opcode.land_opcode, opcode.lor_opcode => try self.args(scope, 2),
|
||||
opcode.lequal_opcode, opcode.lgreater_opcode, opcode.lless_opcode => try self.args(scope, 2),
|
||||
opcode.lnot_opcode => try self.parseLnot(scope),
|
||||
|
||||
opcode.match_opcode => try self.parseMatch(scope),
|
||||
|
||||
// CreateXField: <source> <index> NameString
|
||||
opcode.create_dword_field_opcode,
|
||||
opcode.create_word_field_opcode,
|
||||
opcode.create_byte_field_opcode,
|
||||
opcode.create_bit_field_opcode,
|
||||
opcode.create_qword_field_opcode,
|
||||
=> try self.parseCreateField(scope, 2),
|
||||
|
||||
else => return error.Malformed,
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse `n` operands.
|
||||
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
||||
var i: usize = 0;
|
||||
while (i < n) : (i += 1) try self.object(scope);
|
||||
}
|
||||
|
||||
/// A NameString in operand/statement position: a method invocation (consuming
|
||||
/// the callee's declared argument count) or a plain name reference.
|
||||
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
if (self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice())) |node| {
|
||||
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
||||
try self.args(scope, node.arg_count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
||||
/// the contents are pure data, never namespace declarations.
|
||||
fn skipPackage(self: *Parser) Error!void {
|
||||
const start = self.position;
|
||||
const len = try self.readPackageLength();
|
||||
const end = start + len;
|
||||
if (end > self.aml.len) return error.Truncated;
|
||||
self.position = end;
|
||||
}
|
||||
|
||||
// --- namespace objects --------------------------------------------------
|
||||
|
||||
fn parseName(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
const value_start = self.position;
|
||||
try self.object(scope); // the DataReferenceObject value
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||
node.value = self.aml[value_start..self.position];
|
||||
}
|
||||
|
||||
fn parseAlias(self: *Parser, scope: *Node) Error!void {
|
||||
_ = try self.readNameString(); // source
|
||||
const name_path = try self.readNameString(); // the alias name
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .alias);
|
||||
}
|
||||
|
||||
fn parseMethod(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
const name_path = try self.readNameString();
|
||||
const flags = try self.readByte();
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .method);
|
||||
node.arg_count = flags & 0x7;
|
||||
// Capture the body for on-demand evaluation and skip it — objects declared
|
||||
// inside a method are created at *runtime*, not at load, so they must not
|
||||
// become permanent namespace nodes.
|
||||
node.value = self.aml[self.position..@min(end, self.aml.len)];
|
||||
self.position = end;
|
||||
}
|
||||
|
||||
fn parseExternal(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
_ = try self.readByte(); // object type
|
||||
const arg_count = try self.readByte();
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .external);
|
||||
node.arg_count = arg_count;
|
||||
}
|
||||
|
||||
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
||||
fn parseScopeLike(self: *Parser, scope: *Node, kind: NodeKind) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
const name_path = try self.readNameString();
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), kind);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
fn parseProcessor(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
const name_path = try self.readNameString();
|
||||
try self.skip(6); // ProcID(byte) + PblkAddress(dword) + PblkLen(byte)
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .processor);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
fn parsePowerResource(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
const name_path = try self.readNameString();
|
||||
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .power_resource);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
||||
/// The offset/length expressions are kept as AML for lazy evaluation.
|
||||
fn parseRegion(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
const space = try self.readByte();
|
||||
const off_start = self.position;
|
||||
try self.object(scope);
|
||||
const off_end = self.position;
|
||||
try self.object(scope);
|
||||
const len_end = self.position;
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||
node.region_space = space;
|
||||
node.region_offset_aml = self.aml[off_start..off_end];
|
||||
node.region_len_aml = self.aml[off_end..len_end];
|
||||
}
|
||||
|
||||
fn parseDataRegion(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||
}
|
||||
|
||||
fn parseMutex(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
try self.skip(1); // sync flags
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .mutex);
|
||||
}
|
||||
|
||||
fn parseEvent(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .event);
|
||||
}
|
||||
|
||||
/// CreateXField: `count` TermArgs then the new field's NameString.
|
||||
fn parseCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
||||
try self.args(scope, count);
|
||||
const name_path = try self.readNameString();
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||
}
|
||||
|
||||
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
||||
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
||||
/// Field, the first NameString is the backing region — captured so field units
|
||||
/// carry a region + bit position the evaluator can read/write.
|
||||
fn parseField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
var region: ?*Node = null;
|
||||
var i: u8 = 0;
|
||||
while (i < name_strings) : (i += 1) {
|
||||
const name_path = try self.readNameString();
|
||||
// Only a plain Field's single NameString denotes an OperationRegion.
|
||||
if (name_strings == 1) region = self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||
}
|
||||
if (bank) try self.object(scope); // bank value TermArg
|
||||
const flags = try self.readByte();
|
||||
self.fieldList(end, scope, region, flags & 0x0F);
|
||||
}
|
||||
|
||||
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
||||
var bit_offset: u32 = 0;
|
||||
var access = initial_access;
|
||||
while (self.position < end) {
|
||||
const lead = self.peek() orelse break;
|
||||
switch (lead) {
|
||||
0x00 => { // ReservedField: advances the bit position
|
||||
self.position += 1;
|
||||
const width = self.readPackageLength() catch break;
|
||||
bit_offset += @intCast(width);
|
||||
},
|
||||
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
||||
self.position += 1;
|
||||
const at = self.readByte() catch break;
|
||||
self.skip(1) catch break;
|
||||
access = at & 0x0F;
|
||||
},
|
||||
0x02 => { // ConnectField: NameString | BufferData
|
||||
self.position += 1;
|
||||
self.object(scope) catch break;
|
||||
},
|
||||
0x03 => { // ExtendedAccessField: type + attrib + length
|
||||
self.position += 1;
|
||||
self.skip(3) catch break;
|
||||
},
|
||||
else => { // NamedField: NameSegment + PkgLength (bit width)
|
||||
const segment = self.readNameSegment() catch break;
|
||||
const width = self.readPackageLength() catch break;
|
||||
const unit = self.namespace.newFieldUnit(scope, segment) catch break;
|
||||
unit.region = region;
|
||||
unit.bit_offset = bit_offset;
|
||||
unit.bit_width = @intCast(width);
|
||||
unit.access_type = access;
|
||||
bit_offset += @intCast(width);
|
||||
},
|
||||
}
|
||||
}
|
||||
self.position = end;
|
||||
}
|
||||
|
||||
// --- control flow -------------------------------------------------------
|
||||
|
||||
fn parseIf(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
try self.object(scope); // predicate
|
||||
self.termList(end, scope);
|
||||
if (self.peek() == opcode.else_opcode) {
|
||||
self.position += 1;
|
||||
try self.parseElse(scope);
|
||||
}
|
||||
}
|
||||
|
||||
fn parseElse(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
self.termList(end, scope);
|
||||
}
|
||||
|
||||
fn parseWhile(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
try self.object(scope); // predicate
|
||||
self.termList(end, scope);
|
||||
}
|
||||
|
||||
fn parseLnot(self: *Parser, scope: *Node) Error!void {
|
||||
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
||||
// otherwise it is a plain LNot of one operand.
|
||||
const b = self.peek() orelse return error.Truncated;
|
||||
switch (b) {
|
||||
opcode.lnot.not_equal, opcode.lnot.less_equal, opcode.lnot.greater_equal => {
|
||||
self.position += 1;
|
||||
try self.args(scope, 2);
|
||||
},
|
||||
else => try self.object(scope),
|
||||
}
|
||||
}
|
||||
|
||||
fn parseMatch(self: *Parser, scope: *Node) Error!void {
|
||||
try self.object(scope); // search package
|
||||
try self.skip(1); // match opcode 1
|
||||
try self.object(scope); // operand 1
|
||||
try self.skip(1); // match opcode 2
|
||||
try self.object(scope); // operand 2
|
||||
try self.object(scope); // start index
|
||||
}
|
||||
|
||||
// --- extended opcodes (0x5B xx) -----------------------------------------
|
||||
|
||||
fn parseExtended(self: *Parser, scope: *Node) Error!void {
|
||||
const e = try self.readByte();
|
||||
switch (e) {
|
||||
opcode.extended.mutex => try self.parseMutex(scope),
|
||||
opcode.extended.event => try self.parseEvent(scope),
|
||||
opcode.extended.operation_region => try self.parseRegion(scope),
|
||||
opcode.extended.data_region => try self.parseDataRegion(scope),
|
||||
opcode.extended.field => try self.parseField(scope, 1, false),
|
||||
opcode.extended.index_field => try self.parseField(scope, 2, false),
|
||||
opcode.extended.bank_field => try self.parseField(scope, 2, true),
|
||||
opcode.extended.device => try self.parseScopeLike(scope, .device),
|
||||
opcode.extended.thermal_zone => try self.parseScopeLike(scope, .thermal_zone),
|
||||
opcode.extended.processor => try self.parseProcessor(scope),
|
||||
opcode.extended.power_resource => try self.parsePowerResource(scope),
|
||||
|
||||
opcode.extended.conditional_reference_of => try self.args(scope, 2), // SuperName, Target
|
||||
opcode.extended.create_field => try self.parseCreateField(scope, 3),
|
||||
opcode.extended.load_table => try self.args(scope, 6),
|
||||
opcode.extended.load => try self.args(scope, 2), // NameString, Target
|
||||
opcode.extended.stall, opcode.extended.sleep => try self.args(scope, 1),
|
||||
opcode.extended.acquire => {
|
||||
try self.object(scope); // mutex SuperName
|
||||
try self.skip(2); // timeout WordData
|
||||
},
|
||||
opcode.extended.signal, opcode.extended.reset, opcode.extended.release, opcode.extended.unload => try self.args(scope, 1),
|
||||
opcode.extended.wait => try self.args(scope, 2),
|
||||
opcode.extended.from_bcd, opcode.extended.to_bcd => try self.args(scope, 2),
|
||||
opcode.extended.fatal => {
|
||||
try self.skip(5); // Type(byte) + Code(dword)
|
||||
try self.object(scope); // Arg TermArg
|
||||
},
|
||||
opcode.extended.revision, opcode.extended.debug, opcode.extended.timer => {},
|
||||
|
||||
else => return error.Malformed,
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
fn isNameStart(b: u8) bool {
|
||||
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||
b == opcode.name_char_underscore or
|
||||
b == opcode.root_char or
|
||||
b == opcode.parent_prefix_char or
|
||||
b == opcode.dual_name_prefix or
|
||||
b == opcode.multi_name_prefix;
|
||||
}
|
||||
@@ -0,0 +1,230 @@
|
||||
//! The backend-agnostic device model.
|
||||
//!
|
||||
//! Discovery backends (ACPI today, device-tree later) translate their native
|
||||
//! hardware description into this one shape, so the rest of the kernel walks a
|
||||
//! plain `Device` tree without knowing which firmware described the machine —
|
||||
//! the same discipline `root.zig`'s `MemoryKind` applies to memory and `architecture`
|
||||
//! applies to the CPU.
|
||||
//!
|
||||
//! This is deliberately minimal: enough to *describe* what was discovered (a
|
||||
//! named node, its class, and its hardware resources) and where it sits in the
|
||||
//! bus hierarchy. Driver matching, families, and probing are a later layer built
|
||||
//! on top of this — nothing here presumes them.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// The hardware primitives a discovery backend needs but can't express portably.
|
||||
/// The kernel injects an implementation (the architecture VMM + port I/O), so the device
|
||||
/// layer touches hardware without importing `architecture` — the same discipline that lets
|
||||
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
||||
pub const Hal = struct {
|
||||
/// Map a physical MMIO range and return the virtual address to reach it at.
|
||||
/// The device layer dereferences the returned address and never learns how
|
||||
/// the kernel places it (identity, physmap, a window — the kernel's choice).
|
||||
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||
pioRead: *const fn (width: u8, port: u16) u32,
|
||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||
};
|
||||
|
||||
/// The kind of hardware resource a device occupies.
|
||||
pub const ResourceKind = enum {
|
||||
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
||||
memory,
|
||||
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
||||
io_port,
|
||||
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
||||
irq,
|
||||
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
||||
bus_range,
|
||||
};
|
||||
|
||||
/// One hardware resource claimed by a device.
|
||||
pub const Resource = struct {
|
||||
kind: ResourceKind,
|
||||
start: u64,
|
||||
len: u64,
|
||||
};
|
||||
|
||||
/// A coarse classification of a device, independent of the describing firmware.
|
||||
/// Kept small on purpose; refine as real drivers arrive.
|
||||
pub const DeviceClass = enum {
|
||||
/// The synthetic root every discovered device hangs beneath.
|
||||
root,
|
||||
processor,
|
||||
interrupt_controller,
|
||||
timer,
|
||||
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
||||
pci_host_bridge,
|
||||
/// A single PCI function.
|
||||
pci_device,
|
||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||
acpi_device,
|
||||
unknown,
|
||||
};
|
||||
|
||||
/// Firmware-independent identity. Each backend fills only the fields it knows;
|
||||
/// the rest stay null. The generic layer never branches on *how* an id was
|
||||
/// obtained, only on its value.
|
||||
pub const Ids = struct {
|
||||
/// The device's ACPI hardware ID (`_HID`), EISA-encoded into 4 bytes, when applicable.
|
||||
acpi_hid: ?u32 = null,
|
||||
/// PCI configuration-space identity, when this node is a PCI function.
|
||||
pci_vendor: ?u16 = null,
|
||||
pci_device: ?u16 = null,
|
||||
/// PCI class/subclass/prog-if packed as 0xCCSSPP.
|
||||
pci_class: ?u24 = null,
|
||||
/// PCI bus/device/function packed as (bus << 8) | (device << 3) | function — the key
|
||||
/// the ACPI address (`_ADR`) merge uses to match a namespace device to this node.
|
||||
pci_bdf: ?u16 = null,
|
||||
};
|
||||
|
||||
/// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is
|
||||
/// the busy case). Stored inline so a device is a single allocation.
|
||||
pub const maximum_resources = 8;
|
||||
|
||||
/// One node in the device tree. Nodes are individually heap-allocated and linked
|
||||
/// intrusively (first-child / next-sibling), the classic device-tree layout —
|
||||
/// no per-node dynamic arrays to manage.
|
||||
pub const Device = struct {
|
||||
name_buffer: [24]u8 = undefined,
|
||||
name_len: u8 = 0,
|
||||
class: DeviceClass = .unknown,
|
||||
ids: Ids = .{},
|
||||
/// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like
|
||||
/// `name`; empty when unset. The generic layer stores/prints it without knowing
|
||||
/// how a backend encoded it.
|
||||
hid_buffer: [8]u8 = undefined,
|
||||
hid_len: u8 = 0,
|
||||
resources: [maximum_resources]Resource = undefined,
|
||||
resource_count: u8 = 0,
|
||||
|
||||
parent: ?*Device = null,
|
||||
first_child: ?*Device = null,
|
||||
next_sibling: ?*Device = null,
|
||||
|
||||
/// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline
|
||||
/// buffer, so it stays valid for the life of the node with no extra allocation.
|
||||
pub fn name(self: *const Device) []const u8 {
|
||||
return self.name_buffer[0..self.name_len];
|
||||
}
|
||||
|
||||
fn setName(self: *Device, s: []const u8) void {
|
||||
const n: u8 = @intCast(@min(s.len, self.name_buffer.len));
|
||||
@memcpy(self.name_buffer[0..n], s[0..n]);
|
||||
self.name_len = n;
|
||||
}
|
||||
|
||||
/// The device's hardware id string, or empty if none is set.
|
||||
pub fn hid(self: *const Device) []const u8 {
|
||||
return self.hid_buffer[0..self.hid_len];
|
||||
}
|
||||
|
||||
pub fn setHid(self: *Device, s: []const u8) void {
|
||||
const n: u8 = @intCast(@min(s.len, self.hid_buffer.len));
|
||||
@memcpy(self.hid_buffer[0..n], s[0..n]);
|
||||
self.hid_len = n;
|
||||
}
|
||||
|
||||
/// Record a resource. Silently drops beyond `maximum_resources` — discovery logs
|
||||
/// the truncation rather than failing the whole tree.
|
||||
pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool {
|
||||
if (self.resource_count >= maximum_resources) return false;
|
||||
self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len };
|
||||
self.resource_count += 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The device's first resource of `kind`, or null — e.g. a timer's MMIO base.
|
||||
pub fn firstResource(self: *const Device, kind: ResourceKind) ?Resource {
|
||||
for (self.resources[0..self.resource_count]) |r| {
|
||||
if (r.kind == kind) return r;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
/// Owns the discovered device tree and the allocator its nodes came from.
|
||||
pub const DeviceTree = struct {
|
||||
allocator: std.mem.Allocator,
|
||||
root: *Device,
|
||||
|
||||
/// Create a tree with just the synthetic root node.
|
||||
pub fn init(allocator: std.mem.Allocator) !DeviceTree {
|
||||
const root = try allocator.create(Device);
|
||||
root.* = .{ .class = .root };
|
||||
root.setName("root");
|
||||
return .{ .allocator = allocator, .root = root };
|
||||
}
|
||||
|
||||
/// The first device of `class` anywhere in the tree (depth-first), or null —
|
||||
/// how the kernel pulls e.g. the HPET or IOAPIC MMIO base out of discovery.
|
||||
pub fn firstOfClass(self: *const DeviceTree, class: DeviceClass) ?*Device {
|
||||
return firstOfClassIn(self.root, class);
|
||||
}
|
||||
|
||||
/// Allocate a device and append it under `parent`, returning it so the caller
|
||||
/// can attach resources/ids. Appended at the tail so a dump reads in the order
|
||||
/// devices were discovered.
|
||||
pub fn addChild(
|
||||
self: *DeviceTree,
|
||||
parent: *Device,
|
||||
class: DeviceClass,
|
||||
device_name: []const u8,
|
||||
) !*Device {
|
||||
const d = try self.allocator.create(Device);
|
||||
d.* = .{ .class = class, .parent = parent };
|
||||
d.setName(device_name);
|
||||
if (parent.first_child == null) {
|
||||
parent.first_child = d;
|
||||
} else {
|
||||
var current = parent.first_child.?;
|
||||
while (current.next_sibling) |sib| current = sib;
|
||||
current.next_sibling = d;
|
||||
}
|
||||
return d;
|
||||
}
|
||||
|
||||
/// Walk the tree depth-first, emitting an indented, human-readable listing.
|
||||
/// `emit` is a raw byte sink (e.g. the serial `debugWrite`), so this stays
|
||||
/// independent of the kernel console.
|
||||
pub fn dump(self: *const DeviceTree, emit: *const fn ([]const u8) void) void {
|
||||
dumpNode(self.root, 0, emit);
|
||||
}
|
||||
};
|
||||
|
||||
fn firstOfClassIn(node: *Device, class: DeviceClass) ?*Device {
|
||||
var child = node.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) {
|
||||
if (c.class == class) return c;
|
||||
if (firstOfClassIn(c, class)) |found| return found;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn dumpNode(device: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
||||
const indent = @min(depth * 2, 40);
|
||||
|
||||
var buffer: [200]u8 = undefined;
|
||||
@memset(buffer[0..indent], ' ');
|
||||
const body = if (device.hid_len != 0)
|
||||
std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s}\n", .{ device.name(), @tagName(device.class), device.hid() }) catch return
|
||||
else
|
||||
std.fmt.bufPrint(buffer[indent..], "{s} [{s}]\n", .{ device.name(), @tagName(device.class) }) catch return;
|
||||
emit(buffer[0 .. indent + body.len]);
|
||||
|
||||
for (device.resources[0..device.resource_count]) |r| {
|
||||
var rbuf: [200]u8 = undefined;
|
||||
const pad = @min(indent + 2, 42);
|
||||
@memset(rbuf[0..pad], ' ');
|
||||
const rline = std.fmt.bufPrint(
|
||||
rbuf[pad..],
|
||||
"- {s} 0x{x} len 0x{x}\n",
|
||||
.{ @tagName(r.kind), r.start, r.len },
|
||||
) catch continue;
|
||||
emit(rbuf[0 .. pad + rline.len]);
|
||||
}
|
||||
|
||||
var child = device.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
//! Device-tree (Flattened Device Tree / FDT) discovery backend — stub.
|
||||
//!
|
||||
//! This is the second backend the platform facade dispatches to, for machines
|
||||
//! that describe hardware with a device-tree blob instead of ACPI (typically
|
||||
//! ARM). It is intentionally unimplemented: the bootloader has no DTB handoff
|
||||
//! field yet, so this path is currently unreachable. It exists so the facade
|
||||
//! already routes to a backend rather than hard-coding ACPI — wiring the FDT
|
||||
//! parser in later is a local change here, not an architectural one.
|
||||
|
||||
const device_model = @import("device-model.zig");
|
||||
|
||||
/// Populate `device_tree` from a device-tree blob. Not implemented yet.
|
||||
pub fn discover(device_tree: *device_model.DeviceTree) !void {
|
||||
_ = device_tree;
|
||||
return error.Unsupported;
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
//! The firmware-agnostic discovery facade.
|
||||
//!
|
||||
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
||||
//! without ever naming ACPI or device-tree — the same way it imports `architecture`
|
||||
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
||||
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
||||
//! because a single image — a future ARM kernel especially — may boot under
|
||||
//! either firmware. That's a deliberate divergence from `architecture`, which is a
|
||||
//! compile-time choice.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const device_model = @import("device-model.zig");
|
||||
const acpi = @import("acpi.zig");
|
||||
const power = @import("power.zig");
|
||||
const devicetree = @import("device-tree.zig");
|
||||
|
||||
pub const DeviceTree = device_model.DeviceTree;
|
||||
pub const Device = device_model.Device;
|
||||
pub const DeviceClass = device_model.DeviceClass;
|
||||
pub const Resource = device_model.Resource;
|
||||
pub const ResourceKind = device_model.ResourceKind;
|
||||
pub const Hal = device_model.Hal;
|
||||
pub const PowerInformation = acpi.PowerInformation;
|
||||
pub const AmlStats = acpi.AmlStats;
|
||||
pub const PlatformInformation = acpi.PlatformInformation;
|
||||
pub const RegisterAccess = acpi.RegisterAccess;
|
||||
pub const IsoEntry = acpi.IsoEntry;
|
||||
pub const Cpu = acpi.Cpu;
|
||||
|
||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||
pub fn powerInformation() PowerInformation {
|
||||
return acpi.power_information;
|
||||
}
|
||||
|
||||
/// The scalar firmware facts the architecture layer needs to avoid legacy assumptions
|
||||
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
||||
pub fn platformInformation() PlatformInformation {
|
||||
return acpi.platform_information;
|
||||
}
|
||||
|
||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||
pub fn amlStats() AmlStats {
|
||||
return acpi.aml_stats;
|
||||
}
|
||||
|
||||
/// The usable logical processors discovered during enumeration — one entry per
|
||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||
/// same instant once the application processors are started. Today only the
|
||||
/// bootstrap processor is actually running, so starting the rest is the pending SMP
|
||||
/// step (see docs/smp.md). Borrowed from static storage populated by `discover`.
|
||||
pub fn cpus() []const Cpu {
|
||||
return acpi.cpu_information.cpus[0..acpi.cpu_information.count];
|
||||
}
|
||||
|
||||
/// Non-zero only if enumeration found more processors than the static pool holds
|
||||
/// (the surplus were dropped from `cpus()`); surfaced so the cap is never silent.
|
||||
pub fn cpusDropped() usize {
|
||||
return acpi.cpu_information.dropped;
|
||||
}
|
||||
|
||||
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
||||
/// primitives the backend needs (MMIO mapping for PCIe configuration space, port I/O for
|
||||
/// ACPI registers); pass the architecture implementation. Errors leave nothing to clean up
|
||||
/// beyond the tree's own allocations.
|
||||
pub fn discover(
|
||||
boot_information: *const danos.BootInformation,
|
||||
allocator: std.mem.Allocator,
|
||||
hal: Hal,
|
||||
) !DeviceTree {
|
||||
var device_tree = try DeviceTree.init(allocator);
|
||||
|
||||
if (boot_information.acpi_rsdp != 0) {
|
||||
try acpi.discover(boot_information.acpi_rsdp, &device_tree, hal);
|
||||
} else {
|
||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
||||
// path is a stub, so this reports the machine described itself no way we
|
||||
// understand yet.
|
||||
try devicetree.discover(&device_tree);
|
||||
}
|
||||
|
||||
return device_tree;
|
||||
}
|
||||
|
||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
||||
pub fn reboot(hal: Hal) void {
|
||||
power.reboot(hal);
|
||||
}
|
||||
|
||||
/// Power the machine off (ACPI S5). Never returns on success.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
power.shutdown(hal);
|
||||
}
|
||||
@@ -0,0 +1,104 @@
|
||||
//! Machine power control: enter ACPI mode, reboot, and power off (ACPI S5).
|
||||
//!
|
||||
//! Built entirely on the register map `acpi` extracted from the FADT plus the
|
||||
//! sleep-state (`_Sx`) types the AML submodule pulled from the DSDT, driven through the
|
||||
//! injected `Hal` (port I/O and MMIO). Nothing here is x86-specific beyond the
|
||||
//! well-known legacy reset fallbacks, which are guarded behind the ACPI methods.
|
||||
//!
|
||||
//! S3 (suspend-to-RAM) is stubbed: it needs a wake trampoline and device
|
||||
//! re-initialisation, a milestone of its own.
|
||||
|
||||
const acpi = @import("acpi.zig");
|
||||
const device_model = @import("device-model.zig");
|
||||
const Hal = device_model.Hal;
|
||||
|
||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
||||
|
||||
/// Switch the platform into ACPI mode if it isn't already, so the PM1 control
|
||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
||||
pub fn enable(hal: Hal) void {
|
||||
const pi = acpi.power_information;
|
||||
if (!pi.pm1a_cnt.present()) return;
|
||||
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
||||
|
||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
||||
var spins: usize = 0;
|
||||
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||
}
|
||||
|
||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
const pi = acpi.power_information;
|
||||
|
||||
// 1. The FADT reset register, when the firmware advertises support.
|
||||
if (pi.reset_supported and pi.reset.present()) {
|
||||
writeRegister(hal, pi.reset, pi.reset_value);
|
||||
delay();
|
||||
}
|
||||
// 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYSTEM_RST).
|
||||
hal.pioWrite(1, 0xCF9, 0x0E);
|
||||
hal.pioWrite(1, 0xCF9, 0x06);
|
||||
delay();
|
||||
// 3. Pulse the 8042 keyboard controller's reset line.
|
||||
hal.pioWrite(1, 0x64, 0xFE);
|
||||
delay();
|
||||
}
|
||||
|
||||
/// Power the machine off via ACPI S5. Requires the soft-off (`_S5`) sleep type; if
|
||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
enable(hal);
|
||||
const pi = acpi.power_information;
|
||||
const s5 = pi.s5 orelse return;
|
||||
|
||||
if (pi.pm1a_cnt.present()) {
|
||||
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||
}
|
||||
if (pi.pm1b_cnt.present()) {
|
||||
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||
}
|
||||
delay();
|
||||
}
|
||||
|
||||
/// S3 suspend-to-RAM — not implemented (needs a wake path + device re-init).
|
||||
pub fn sleepS3(hal: Hal) error{Unsupported}!void {
|
||||
_ = hal;
|
||||
return error.Unsupported;
|
||||
}
|
||||
|
||||
/// The PM1 control write that requests sleep type `slp_typ`: SLP_TYP in bits
|
||||
/// [12:10], SLP_EN in bit 13.
|
||||
fn sleepValue(slp_typ: u8) u32 {
|
||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
||||
}
|
||||
|
||||
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
return p.*;
|
||||
}
|
||||
return hal.pioRead(register.width, @intCast(register.address));
|
||||
}
|
||||
|
||||
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
p.* = value;
|
||||
} else {
|
||||
hal.pioWrite(register.width, @intCast(register.address), value);
|
||||
}
|
||||
}
|
||||
|
||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
||||
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||
/// from being optimised away.
|
||||
fn delay() void {
|
||||
var i: usize = 0;
|
||||
while (i < 50_000_000) : (i += 1) {
|
||||
asm volatile ("" ::: .{ .memory = true });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,210 @@
|
||||
//! /sbin/busd — a user-space **bus driver**, and the smallest honest example of one.
|
||||
//!
|
||||
//! A bus driver owns a device that *contains other devices*, enumerates them by some
|
||||
//! bus-specific protocol, and publishes each one into the kernel's device table so a
|
||||
//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here
|
||||
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
|
||||
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
|
||||
//!
|
||||
//! It's a toy bus, but nothing about the mechanism is: `busd` reads how many children
|
||||
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
|
||||
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
|
||||
//! every one of those resources is contained in what `busd` was granted. A comparator
|
||||
//! driver then claims a child and maps only *its* registers — not the whole block.
|
||||
//!
|
||||
//! It also proves the negative: registering a child whose window escapes the parent's
|
||||
//! is refused. Without that check, `device_register` would be a system_call for mapping
|
||||
//! arbitrary physical memory.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const device = runtime.device;
|
||||
|
||||
const register_general_cap = 0x000;
|
||||
|
||||
/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes.
|
||||
fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor {
|
||||
return .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = hpet_base + 0x100 + 0x20 * n,
|
||||
.len = 0x20,
|
||||
};
|
||||
}
|
||||
|
||||
fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
const total = device.enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
if (d.parent != device.no_parent) continue; // the block, not a comparator child
|
||||
for (0..d.resource_count) |j| {
|
||||
if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The parent's MMIO resource, and its IRQ if it has one.
|
||||
fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } {
|
||||
var mmio: device.ResourceDescriptor = undefined;
|
||||
var irq: ?device.ResourceDescriptor = null;
|
||||
for (0..d.resource_count) |j| {
|
||||
const r = d.resources[j];
|
||||
if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r;
|
||||
if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r;
|
||||
}
|
||||
return .{ .mmio = mmio, .irq = irq };
|
||||
}
|
||||
|
||||
fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 {
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.parent == parent_id) return d.id;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("busd: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const parent = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("busd: no HPET\n");
|
||||
return;
|
||||
};
|
||||
const resource = resourcesOf(parent);
|
||||
|
||||
// Claim the bus. Everything below is subdivision of what this claim granted.
|
||||
//
|
||||
// Claims are exclusive, and at a normal boot the kernel spawns every initrd
|
||||
// binary — so hpetd may own the HPET already. That's not an error, it's the
|
||||
// capability model working: exit quietly and leave the device to its owner. The
|
||||
// `bus` test spawns busd alone, so there it wins the claim.
|
||||
if (!device.claim(parent.id)) {
|
||||
_ = runtime.system.write("busd: HPET already claimed by another driver, nothing to do\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Enumerate the bus: ask the hardware how many children it has.
|
||||
const base = device.mmioMap(parent.id, 0) orelse {
|
||||
_ = runtime.system.write("busd: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
|
||||
const n_children = ((cap.* >> 8) & 0x1F) + 1;
|
||||
|
||||
// Publish one child per comparator, each owning only its own window.
|
||||
var published: u64 = 0;
|
||||
var n: u64 = 0;
|
||||
while (n < n_children) : (n += 1) {
|
||||
var child = std.mem.zeroes(device.DeviceDescriptor);
|
||||
child.class = @intFromEnum(device.DeviceClass.timer);
|
||||
child.hid_len = 6;
|
||||
child.hid[0..6].* = "hpet-t".*;
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = timerWindow(resource.mmio.start, n);
|
||||
// Comparators share the block's interrupt line; only one child can bind it,
|
||||
// but all of them may legitimately name it.
|
||||
if (resource.irq) |i| {
|
||||
child.resources[child.resource_count] = i;
|
||||
child.resource_count += 1;
|
||||
}
|
||||
|
||||
if (device.register(parent.id, &child) == null) {
|
||||
_ = runtime.system.write("busd: register failed\n");
|
||||
return;
|
||||
}
|
||||
published += 1;
|
||||
}
|
||||
|
||||
// The negative case. A window one byte past the end of the parent's must be
|
||||
// refused — otherwise device_register would be "map any physical page you like".
|
||||
// Confirm the table did not grow, not merely that the call returned null: null
|
||||
// also means NoSpace/BadParent, so a size check is what actually proves the
|
||||
// *containment* rule fired.
|
||||
const before = device.enumerate(buffer);
|
||||
var rogue = std.mem.zeroes(device.DeviceDescriptor);
|
||||
rogue.class = @intFromEnum(device.DeviceClass.unknown);
|
||||
rogue.resource_count = 1;
|
||||
rogue.resources[0] = .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = resource.mmio.start + resource.mmio.len,
|
||||
.len = 0x1000,
|
||||
};
|
||||
if (device.register(parent.id, &rogue) != null) {
|
||||
_ = runtime.system.write("busd: FAIL out-of-window child was accepted\n");
|
||||
return;
|
||||
}
|
||||
if (device.enumerate(buffer) != before) {
|
||||
_ = runtime.system.write("busd: FAIL rogue child leaked into the table\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// And confirm the children came back with the right parent and a *narrower*
|
||||
// window than the bus — read from the table, not from our own memory.
|
||||
const total = device.enumerate(buffer);
|
||||
var seen: u64 = 0;
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.parent != parent.id) continue;
|
||||
const w = d.resources[0];
|
||||
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
|
||||
_ = runtime.system.write("busd: FAIL child window is not inside the bus\n");
|
||||
return;
|
||||
}
|
||||
seen += 1;
|
||||
}
|
||||
if (seen != published) {
|
||||
_ = runtime.system.write("busd: FAIL child count mismatch\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Delegation, end to end: claim a child and map *it*. A real class driver would be
|
||||
// a different process; here busd plays both parts, which exercises the same path.
|
||||
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
|
||||
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
|
||||
//
|
||||
// (mmio_map rounds to a page, so the child's mapping physically covers the whole
|
||||
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
|
||||
// The *resource* is narrow even though the page isn't.)
|
||||
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
|
||||
_ = runtime.system.write("busd: FAIL no child to claim\n");
|
||||
return;
|
||||
};
|
||||
if (!device.claim(child_id)) {
|
||||
_ = runtime.system.write("busd: FAIL could not claim own child\n");
|
||||
return;
|
||||
}
|
||||
const child_base = device.mmioMap(child_id, 0) orelse {
|
||||
_ = runtime.system.write("busd: FAIL child mmio_map refused\n");
|
||||
return;
|
||||
};
|
||||
const via_child: *volatile u64 = @ptrFromInt(child_base);
|
||||
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
|
||||
if (via_child.* != via_bus.*) {
|
||||
_ = runtime.system.write("busd: FAIL child window does not alias the bus register\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// A descriptor pointer into an unmapped page must fail the call, not fault the
|
||||
// kernel. Grab a page, free it, and register through the stale address: if the
|
||||
// kernel dereferenced it raw (rather than copying in through the page tables) this
|
||||
// would triple-fault QEMU and the test would time out instead of printing ok.
|
||||
const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE);
|
||||
if (!runtime.system.mmapFailed(scratch)) {
|
||||
_ = runtime.system.munmap(scratch, 0x1000);
|
||||
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
|
||||
if (device.register(parent.id, descriptor) != null) {
|
||||
_ = runtime.system.write("busd: FAIL register accepted an unmapped descriptor\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("busd: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,187 @@
|
||||
//! /sbin/hpetd — a user-space HPET driver. It proves the whole driver model end to
|
||||
//! end: enumerate the device table, find the HPET, claim it, map its registers into
|
||||
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
|
||||
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
|
||||
//!
|
||||
//! Nothing here polls. Between interrupts the process is `.blocked` and off every
|
||||
//! scheduler queue; the core runs other work or idles. That is the point of the
|
||||
//! exercise — a driver is a process that sleeps until its device has something to
|
||||
//! say (see docs/drivers.md).
|
||||
//!
|
||||
//! The comparator is configured **level-triggered** on purpose. Edge would be
|
||||
//! simpler, but level is the discipline every real device line needs, and it forces
|
||||
//! the full cycle to be correct:
|
||||
//!
|
||||
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
|
||||
//! hpetd wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
||||
//! hpetd irq_ack -> kernel unmasks the GSI
|
||||
//!
|
||||
//! Clear the status bit *before* acking, or the line is still asserted when the
|
||||
//! kernel unmasks and the I/O APIC redelivers forever.
|
||||
//!
|
||||
//! Register map (HPET spec 1.0a):
|
||||
//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1
|
||||
//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF
|
||||
//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear)
|
||||
//! 0x0F0 MAIN_COUNTER
|
||||
//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic)
|
||||
//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP
|
||||
//! 0x108 TIMER0_COMPARATOR
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const register_general_cap = 0x000;
|
||||
const register_general_configuration = 0x010;
|
||||
const register_int_status = 0x020;
|
||||
const register_main_counter = 0x0F0;
|
||||
const register_timer0_configuration = 0x100;
|
||||
const register_timer0_comparator = 0x108;
|
||||
|
||||
const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF
|
||||
const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF
|
||||
const tn_int_type_level: u64 = 1 << 1;
|
||||
const tn_int_enb: u64 = 1 << 2;
|
||||
const tn_type_periodic: u64 = 1 << 3;
|
||||
const tn_route_shift = 9;
|
||||
const tn_route_mask: u64 = 0x1F << tn_route_shift;
|
||||
|
||||
/// Interrupts to observe before declaring victory.
|
||||
const target_ticks = 5;
|
||||
|
||||
fn register(base: usize, off: usize) *volatile u64 {
|
||||
return @ptrFromInt(base + off);
|
||||
}
|
||||
|
||||
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
|
||||
/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`.
|
||||
const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 };
|
||||
|
||||
fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
||||
const total = device.enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
// Skip comparator children a bus driver may have published below the block
|
||||
// (see system/drivers/busd/busd.zig) — we want the register block itself.
|
||||
if (d.parent != device.no_parent) continue;
|
||||
var mmio: ?u64 = null;
|
||||
var irq: ?u64 = null;
|
||||
for (0..d.resource_count) |j| {
|
||||
switch (d.resources[j].kind) {
|
||||
@intFromEnum(device.ResourceKind.memory) => mmio = mmio orelse j,
|
||||
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
if (mmio) |m| if (irq) |i| {
|
||||
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
|
||||
_ = runtime.system.write("hpetd: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const hpet = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("hpetd: no HPET with an IRQ\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (!device.claim(hpet.device_id)) {
|
||||
_ = runtime.system.write("hpetd: claim failed\n");
|
||||
return;
|
||||
}
|
||||
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
|
||||
_ = runtime.system.write("hpetd: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator
|
||||
// to raise exactly this line — the kernel will only bind the one it recorded.
|
||||
const gsi = hpet.gsi;
|
||||
|
||||
const endpoint = ipc.createEndpoint() orelse {
|
||||
_ = runtime.system.write("hpetd: create_endpoint failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// --- program the hardware ------------------------------------------------
|
||||
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
|
||||
const femtos_per_tick = register(base, register_general_cap).* >> 32;
|
||||
if (femtos_per_tick == 0) {
|
||||
_ = runtime.system.write("hpetd: bad HPET period\n");
|
||||
return;
|
||||
}
|
||||
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
|
||||
|
||||
// Stop the counter and take the legacy route off while we reconfigure.
|
||||
register(base, register_general_configuration).* &= ~(configuration_enable | configuration_leg_rt);
|
||||
|
||||
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
|
||||
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
|
||||
// we simply re-arm from the driver on each interrupt, which is what a tickless
|
||||
// timer driver does anyway.
|
||||
var t0 = register(base, register_timer0_configuration).*;
|
||||
t0 &= ~(tn_route_mask | tn_type_periodic);
|
||||
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
|
||||
register(base, register_timer0_configuration).* = t0;
|
||||
|
||||
// Clear any stale assertion, then arm ~100 ms out and start the counter.
|
||||
register(base, register_int_status).* = 1;
|
||||
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
|
||||
register(base, register_general_configuration).* |= configuration_enable;
|
||||
|
||||
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
|
||||
_ = runtime.system.write("hpetd: irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("hpetd: bound, sleeping until the hardware speaks\n");
|
||||
|
||||
// --- the driver loop -----------------------------------------------------
|
||||
// Blocked in replyWait. No polling, no spinning: the next line of this function
|
||||
// runs only because an interrupt fired.
|
||||
var receive: [64]u8 = undefined;
|
||||
var count: usize = 0;
|
||||
while (count < target_ticks) {
|
||||
// Blocked here. The task is `.blocked` and off every scheduler queue; the
|
||||
// next line runs only because the HPET raised its line.
|
||||
const r = ipc.replyWait(endpoint, &.{}, &receive);
|
||||
if (!r.isNotification()) continue; // a client request, not our IRQ
|
||||
|
||||
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
|
||||
// line is still asserted and unmasking would refire immediately.
|
||||
register(base, register_int_status).* = 1;
|
||||
count += 1;
|
||||
|
||||
if (count < target_ticks) {
|
||||
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
|
||||
} else {
|
||||
// Last one: stop the source rather than re-arming, so the line is left
|
||||
// both quiet *and* unmasked by the ack below. Re-arming here would leave
|
||||
// a pending interrupt that nobody is waiting for, and the ISR would mask
|
||||
// the line again a moment later.
|
||||
register(base, register_timer0_configuration).* &= ~tn_int_enb;
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpetd: irq\n");
|
||||
if (!device.irqAck(hpet.device_id, hpet.irq)) {
|
||||
_ = runtime.system.write("hpetd: irq_ack failed\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpetd: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
//! The initrd (initial ramdisk) container format — shared by the build-time
|
||||
//! packer (tools/mkinitrd.zig) and the kernel that unpacks it. Deliberately
|
||||
//! trivial: a header, a table of fixed-size entries, then the concatenated file
|
||||
//! blobs. We own both producer and consumer, so it need be no fancier.
|
||||
//!
|
||||
//! Layout:
|
||||
//! Header (magic, count)
|
||||
//! Entry * count (name, offset, len) — offset/len into the image
|
||||
//! blob bytes... (each entry's file, at its offset)
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// "DNRD" — identifies a danos initrd image.
|
||||
pub const magic: u32 = 0x444E5244;
|
||||
|
||||
pub const Header = extern struct {
|
||||
magic: u32,
|
||||
count: u32,
|
||||
};
|
||||
|
||||
pub const Entry = extern struct {
|
||||
name: [32]u8, // NUL-padded file name (basename)
|
||||
offset: u64, // byte offset of the blob within the image
|
||||
len: u64, // blob length in bytes
|
||||
};
|
||||
|
||||
/// A validated view over an initrd image. `init` checks the magic and that the
|
||||
/// entry table fits; `entry` bounds-checks each blob against the image.
|
||||
pub const Reader = struct {
|
||||
image: []const u8,
|
||||
count: u32,
|
||||
|
||||
pub fn init(image: []const u8) ?Reader {
|
||||
if (image.len < @sizeOf(Header)) return null;
|
||||
const h = std.mem.bytesToValue(Header, image[0..@sizeOf(Header)]);
|
||||
if (h.magic != magic) return null;
|
||||
const table_end = @sizeOf(Header) + @as(usize, h.count) * @sizeOf(Entry);
|
||||
if (table_end > image.len) return null;
|
||||
return .{ .image = image, .count = h.count };
|
||||
}
|
||||
|
||||
pub const Item = struct { name: []const u8, blob: []const u8 };
|
||||
|
||||
pub fn entry(self: Reader, i: u32) ?Item {
|
||||
if (i >= self.count) return null;
|
||||
const off = @sizeOf(Header) + @as(usize, i) * @sizeOf(Entry);
|
||||
const e = std.mem.bytesToValue(Entry, self.image[off..][0..@sizeOf(Entry)]);
|
||||
if (e.offset > self.image.len or e.len > self.image.len - e.offset) return null;
|
||||
// The name is stored in the entry's fixed field; return a stable slice
|
||||
// into the image (not the value copy) up to the NUL terminator.
|
||||
const name_field = self.image[off .. off + 32];
|
||||
const nlen = std.mem.indexOfScalar(u8, name_field, 0) orelse name_field.len;
|
||||
return .{
|
||||
.name = name_field[0..nlen],
|
||||
.blob = self.image[@intCast(e.offset)..][0..@intCast(e.len)],
|
||||
};
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,413 @@
|
||||
//! Local APIC and its timer — the source of device interrupts.
|
||||
//!
|
||||
//! Modern x86 routes interrupts through the per-CPU Local APIC (the legacy 8259
|
||||
//! PIC is remapped out of the way and masked). The LAPIC also has a built-in
|
||||
//! timer, which is the simplest device interrupt to bring up: it needs no
|
||||
//! external routing, just a vector and a count. We use it as danos's heartbeat.
|
||||
//!
|
||||
//! The LAPIC is memory-mapped (default physical 0xFEE00000, inside our identity
|
||||
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
|
||||
//! the LAPIC won't deliver the next one.
|
||||
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// The ACPI PM timer, as a calibration reference: an I/O port or MMIO counter.
|
||||
pub const PmTimer = struct { mmio: bool, address: u64, is_32bit: bool };
|
||||
|
||||
// Platform facts from discovery (set by `configure` before bring-up). Defaults are
|
||||
// the legacy-safe assumptions so the code still works if discovery never ran.
|
||||
var configuration_pic_present: bool = true;
|
||||
var configuration_hpet_base: u64 = 0; // 0 = no HPET discovered
|
||||
var configuration_pm_timer: ?PmTimer = null;
|
||||
/// Which reference the last calibration used, for logging.
|
||||
var cal_source: []const u8 = "none";
|
||||
|
||||
/// Hand the LAPIC bring-up the discovered platform facts. Call before `init`.
|
||||
pub fn configure(pic_present: bool, hpet_base: u64, pm_timer: ?PmTimer) void {
|
||||
configuration_pic_present = pic_present;
|
||||
configuration_hpet_base = hpet_base;
|
||||
configuration_pm_timer = pm_timer;
|
||||
}
|
||||
|
||||
/// The calibration reference the timer was measured against ("cpuid"/"hpet"/…).
|
||||
pub fn calibrationSource() []const u8 {
|
||||
return cal_source;
|
||||
}
|
||||
|
||||
/// IDT vector the timer fires on (in the device range, >= 32).
|
||||
pub const timer_vector = 32;
|
||||
/// Spurious-interrupt vector. Low nibble 0xF by convention; also in our gate
|
||||
/// range so a stray spurious interrupt lands on a valid (no-op) handler.
|
||||
const spurious_vector = 47;
|
||||
|
||||
// LAPIC register offsets.
|
||||
const register_spurious = 0x0F0;
|
||||
const register_eoi = 0x0B0;
|
||||
const register_id = 0x020; // this core's LAPIC id, in bits 24-31
|
||||
const register_icr_low = 0x300; // interrupt command register, low dword (writing it sends)
|
||||
const register_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31)
|
||||
const register_lvt_timer = 0x320;
|
||||
const register_timer_initial = 0x380;
|
||||
const register_timer_current = 0x390;
|
||||
const register_timer_divide = 0x3E0;
|
||||
|
||||
const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight
|
||||
|
||||
const lvt_masked = 1 << 16;
|
||||
const lvt_periodic = 1 << 17;
|
||||
const timer_divide_16 = 0x3;
|
||||
|
||||
const ia32_apic_base_msr = 0x1B;
|
||||
|
||||
/// LAPIC MMIO base. A runtime var (not a constant) both because we read it from
|
||||
/// the MSR and so register writes compile to normal stores rather than a
|
||||
/// `mov moffs`, which the self-hosted backend can't encode.
|
||||
var base: usize = 0xFEE00000;
|
||||
|
||||
var tick_count: u64 = 0;
|
||||
|
||||
/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate).
|
||||
/// At divide-by-16, this is the effective counting rate.
|
||||
var ticks_per_ms: u32 = 0;
|
||||
/// The periodic-interrupt frequency the timer is armed at, once initTimer runs.
|
||||
var timer_hz: u32 = 0;
|
||||
|
||||
/// TSC (Time Stamp Counter) calibration: cycles per second, and the count at boot.
|
||||
/// The TSC is a per-core cycle counter, giving a ~nanosecond high-resolution
|
||||
/// monotonic clock — far finer than the millisecond timer tick.
|
||||
var tsc_hz: u64 = 0;
|
||||
var tsc_base: u64 = 0;
|
||||
|
||||
/// Read the 64-bit Time Stamp Counter.
|
||||
fn rdtsc() u64 {
|
||||
var low: u32 = undefined;
|
||||
var high: u32 = undefined;
|
||||
asm volatile ("rdtsc"
|
||||
: [low] "={eax}" (low),
|
||||
[high] "={edx}" (high),
|
||||
);
|
||||
return (@as(u64, high) << 32) | low;
|
||||
}
|
||||
|
||||
fn read(register: u32) u32 {
|
||||
return @as(*volatile u32, @ptrFromInt(base + register)).*;
|
||||
}
|
||||
fn write(register: u32, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(base + register)).* = value;
|
||||
}
|
||||
|
||||
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
|
||||
/// vectors) and mask every line, so it can't deliver interrupts behind the APIC.
|
||||
fn remapAndMaskPic() void {
|
||||
io.outb(0x20, 0x11); // start init (cascade mode)
|
||||
io.outb(0xA0, 0x11);
|
||||
io.outb(0x21, 0x20); // master offset 0x20
|
||||
io.outb(0xA1, 0x28); // slave offset 0x28
|
||||
io.outb(0x21, 0x04); // tell master about slave on IRQ2
|
||||
io.outb(0xA1, 0x02);
|
||||
io.outb(0x21, 0x01); // 8086 mode
|
||||
io.outb(0xA1, 0x01);
|
||||
io.outb(0x21, 0xFF); // mask all
|
||||
io.outb(0xA1, 0xFF);
|
||||
}
|
||||
|
||||
/// Enable the Local APIC: mask the PIC (only if one is present — a legacy-free
|
||||
/// UEFI Class 3 machine may have none), set the global-enable MSR bit, and
|
||||
/// software-enable the APIC via its spurious-vector register.
|
||||
pub fn init() void {
|
||||
if (configuration_pic_present) remapAndMaskPic();
|
||||
|
||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||
// Reach the LAPIC through the physmap (paging.init maps its page there).
|
||||
base = @intCast(danos.physicalToVirtual(msr & 0xFFFFF000)); // physical base is bits 12+
|
||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||
|
||||
write(register_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
||||
}
|
||||
|
||||
/// Software-enable *this* core's Local APIC — the application-processor counterpart
|
||||
/// of `init`, minus the one-time PIC remap (the BSP already masked it) and minus
|
||||
/// calibration (the timer rate is a shared hardware constant, measured once). Each
|
||||
/// core has its own LAPIC at the same MMIO address, so no per-core base is needed.
|
||||
pub fn initSecondary() void {
|
||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||
write(register_spurious, 0x100 | spurious_vector); // software enable
|
||||
}
|
||||
|
||||
// --- application-processor wakeup (INIT–SIPI–SIPI) --------------------------
|
||||
|
||||
/// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of
|
||||
/// the wake sequence. Blocks until the LAPIC reports the IPI was delivered.
|
||||
pub fn sendInit(apic_id: u32) void {
|
||||
write(register_icr_high, apic_id << 24);
|
||||
write(register_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered
|
||||
waitIcrIdle();
|
||||
}
|
||||
|
||||
/// Send a STARTUP IPI (SIPI) telling the target core to begin executing at physical
|
||||
/// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is
|
||||
/// sent twice after the INIT; both calls block until delivery completes.
|
||||
pub fn sendStartup(apic_id: u32, vector: u8) void {
|
||||
write(register_icr_high, apic_id << 24);
|
||||
write(register_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector
|
||||
waitIcrIdle();
|
||||
}
|
||||
|
||||
fn waitIcrIdle() void {
|
||||
while (read(register_icr_low) & icr_delivery_pending != 0) {}
|
||||
}
|
||||
|
||||
/// The calibration window: we time everything against a 10 ms reference interval.
|
||||
const calib_ms = 10;
|
||||
|
||||
/// Measure the LAPIC timer's and the TSC's rates. The PIT (legacy 8254) can be
|
||||
/// absent on UEFI Class 3 firmware — and polling it would hang — so we pick a
|
||||
/// reference clock in order of preference: the CPU's own TSC frequency (CPUID leaf
|
||||
/// 0x15, no external timer needed), then the discovered HPET, then the ACPI PM
|
||||
/// timer, and only the PIT as a last resort. Each path yields the same two rates.
|
||||
pub fn calibrate() void {
|
||||
var done = false;
|
||||
|
||||
// 1. CPUID leaf 0x15 gives the TSC frequency directly — measure the LAPIC
|
||||
// against the TSC itself, needing no external timer at all.
|
||||
if (cpuidTscHz()) |hz| {
|
||||
measure(hz, ~@as(u64, 0), rdtsc);
|
||||
tsc_hz = hz; // keep the exact enumerated value
|
||||
cal_source = "cpuid";
|
||||
done = true;
|
||||
}
|
||||
|
||||
// 2. The discovered HPET.
|
||||
if (!done and configuration_hpet_base != 0) {
|
||||
if (hpetHz()) |hpet_hz| {
|
||||
measure(hpet_hz, hpetMask(), readHpet);
|
||||
cal_source = "hpet";
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 3. The ACPI PM timer (fixed 3.579545 MHz).
|
||||
if (!done) {
|
||||
if (configuration_pm_timer) |pt| {
|
||||
measure(3_579_545, if (pt.is_32bit) 0xFFFF_FFFF else 0xFF_FFFF, readPmTimer);
|
||||
cal_source = "pm-timer";
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 4. The legacy PIT, last resort.
|
||||
if (!done) {
|
||||
calibratePit();
|
||||
cal_source = "pit";
|
||||
}
|
||||
|
||||
// A bad measurement (no reference actually ticked) leaves nonsense; fall back.
|
||||
if (ticks_per_ms == 0 or tsc_hz == 0) {
|
||||
calibratePit();
|
||||
cal_source = "pit";
|
||||
}
|
||||
|
||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||
}
|
||||
|
||||
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
||||
/// clock (frequency `ref_hz`, counter width `ref_mask`) counts out `calib_ms`, and
|
||||
/// snapshot the TSC across the same window. Yields `ticks_per_ms` and `tsc_hz`.
|
||||
fn measure(ref_hz: u64, ref_mask: u64, refNow: *const fn () u64) void {
|
||||
const calib_ticks = ref_hz / (1000 / calib_ms); // reference ticks in calib_ms
|
||||
|
||||
write(register_timer_divide, timer_divide_16);
|
||||
write(register_lvt_timer, lvt_masked);
|
||||
write(register_timer_initial, 0xFFFFFFFF);
|
||||
|
||||
const ref0 = refNow();
|
||||
const tsc0 = rdtsc();
|
||||
while (((refNow() -% ref0) & ref_mask) < calib_ticks) {}
|
||||
const tsc1 = rdtsc();
|
||||
|
||||
const elapsed = 0xFFFFFFFF - read(register_timer_current);
|
||||
write(register_timer_initial, 0);
|
||||
|
||||
ticks_per_ms = elapsed / calib_ms;
|
||||
tsc_hz = (tsc1 -% tsc0) * (1000 / calib_ms);
|
||||
}
|
||||
|
||||
/// The PIT fallback (legacy 8254 channel 2, polled). Only reached when no better
|
||||
/// reference exists — on a legacy-free machine this path isn't taken.
|
||||
fn calibratePit() void {
|
||||
const pit_hz = 1_193_182;
|
||||
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
|
||||
|
||||
write(register_timer_divide, timer_divide_16);
|
||||
write(register_lvt_timer, lvt_masked);
|
||||
write(register_timer_initial, 0xFFFFFFFF);
|
||||
|
||||
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
|
||||
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
|
||||
io.outb(0x42, @truncate(pit_count));
|
||||
io.outb(0x42, @truncate(pit_count >> 8));
|
||||
|
||||
const tsc_start = rdtsc();
|
||||
io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start
|
||||
var guard: u64 = 0;
|
||||
while (io.inb(0x61) & 0x20 == 0 and guard < 100_000_000) : (guard += 1) {} // bounded
|
||||
const tsc_end = rdtsc();
|
||||
|
||||
const elapsed = 0xFFFFFFFF - read(register_timer_current);
|
||||
write(register_timer_initial, 0);
|
||||
|
||||
ticks_per_ms = elapsed / calib_ms;
|
||||
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms);
|
||||
}
|
||||
|
||||
// --- reference clocks ------------------------------------------------------
|
||||
|
||||
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
||||
/// null if the CPU doesn't enumerate it (common under QEMU).
|
||||
fn cpuidTscHz() ?u64 {
|
||||
if (cpuid(0).eax < 0x15) return null;
|
||||
const r = cpuid(0x15);
|
||||
if (r.eax == 0 or r.ebx == 0 or r.ecx == 0) return null; // ratio/crystal not given
|
||||
return @as(u64, r.ecx) * r.ebx / r.eax;
|
||||
}
|
||||
|
||||
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
||||
|
||||
fn cpuid(leaf: u32) CpuidRegs {
|
||||
var a: u32 = undefined;
|
||||
var b: u32 = undefined;
|
||||
var c: u32 = undefined;
|
||||
var d: u32 = undefined;
|
||||
asm volatile ("cpuid"
|
||||
: [a] "={eax}" (a),
|
||||
[b] "={ebx}" (b),
|
||||
[c] "={ecx}" (c),
|
||||
[d] "={edx}" (d),
|
||||
: [leaf] "{eax}" (leaf),
|
||||
[sub] "{ecx}" (@as(u32, 0)),
|
||||
);
|
||||
return .{ .eax = a, .ebx = b, .ecx = c, .edx = d };
|
||||
}
|
||||
|
||||
// HPET registers: capabilities at +0x00 (period in the high dword, in fs; bit 13 =
|
||||
// 64-bit-counter capable), general configuration at +0x10, main counter at +0xF0.
|
||||
fn hpetRead64(off: usize) u64 {
|
||||
return @as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).*;
|
||||
}
|
||||
fn hpetWrite64(off: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
||||
}
|
||||
|
||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
||||
/// address, so the register accessors reach it without the identity map.
|
||||
fn hpetHz() ?u64 {
|
||||
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||
const caps = hpetRead64(0x00);
|
||||
const period_fs = caps >> 32; // femtoseconds per tick
|
||||
if (period_fs == 0) return null;
|
||||
hpetWrite64(0x10, hpetRead64(0x10) | 1); // ENABLE_CNF: start the main counter
|
||||
return 1_000_000_000_000_000 / period_fs; // 1e15 fs/s ÷ fs/tick
|
||||
}
|
||||
|
||||
/// The HPET counter width mask (64- or 32-bit, per caps bit 13).
|
||||
fn hpetMask() u64 {
|
||||
return if (hpetRead64(0x00) & (1 << 13) != 0) ~@as(u64, 0) else 0xFFFF_FFFF;
|
||||
}
|
||||
|
||||
fn readHpet() u64 {
|
||||
return hpetRead64(0xF0);
|
||||
}
|
||||
|
||||
fn readPmTimer() u64 {
|
||||
const pt = configuration_pm_timer.?;
|
||||
// MMIO PM timer via the physmap (mapMmio is idempotent); the common case is
|
||||
// a legacy I/O port.
|
||||
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(paging.mapMmio(pt.address, 4, false))).*;
|
||||
return io.inl(@intCast(pt.address));
|
||||
}
|
||||
|
||||
/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires
|
||||
/// calibrate() to have run.
|
||||
pub fn initTimer(hz: u32) void {
|
||||
timer_hz = hz;
|
||||
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
|
||||
write(register_timer_divide, timer_divide_16);
|
||||
write(register_lvt_timer, timer_vector | lvt_periodic);
|
||||
write(register_timer_initial, @intCast(count));
|
||||
}
|
||||
|
||||
/// Configured periodic-interrupt frequency (Hz).
|
||||
pub fn frequencyHz() u32 {
|
||||
return timer_hz;
|
||||
}
|
||||
|
||||
/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks.
|
||||
pub fn lapicHz() u64 {
|
||||
return @as(u64, ticks_per_ms) * 1000;
|
||||
}
|
||||
|
||||
/// Measured TSC frequency (Hz).
|
||||
pub fn tscHz() u64 {
|
||||
return tsc_hz;
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
||||
// the scheduler uses for sleep deadlines.
|
||||
|
||||
pub fn nanos() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
||||
}
|
||||
|
||||
pub fn micros() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
||||
}
|
||||
|
||||
pub fn millis() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
||||
}
|
||||
|
||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||
pub fn eoi() void {
|
||||
write(register_eoi, 0);
|
||||
}
|
||||
|
||||
/// Optional callback run each tick (the scheduler registers it for preemption).
|
||||
var on_tick: ?*const fn () void = null;
|
||||
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
on_tick = hook;
|
||||
}
|
||||
|
||||
/// The timer interrupt handler: advance the monotonic tick count, then run the
|
||||
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
|
||||
/// the dispatcher before we get here, so a task switch here doesn't stall it.
|
||||
pub fn timerTick() void {
|
||||
// Acknowledge before the tick hook: `on_tick` is the scheduler, which may switch
|
||||
// tasks and not return promptly, and the LAPIC mustn't wait on it to deliver the
|
||||
// next interrupt. (Each device handler now owns its own EOI — see
|
||||
// `idt.interruptDispatch` — because a *routed* interrupt must be masked at the
|
||||
// I/O APIC before it is acknowledged, an ordering the dispatcher can't impose.)
|
||||
eoi();
|
||||
tick_count +%= 1;
|
||||
if (on_tick) |hook| hook();
|
||||
}
|
||||
|
||||
/// This core's Local APIC id — the interrupt destination for `routeGsi`.
|
||||
pub fn localId() u8 {
|
||||
return @truncate(read(register_id) >> 24);
|
||||
}
|
||||
|
||||
/// Number of timer ticks so far. Volatile load: the count is bumped
|
||||
/// asynchronously by the interrupt handler, so callers must re-read memory.
|
||||
pub fn ticks() u64 {
|
||||
return @as(*const volatile u64, &tick_count).*;
|
||||
}
|
||||
@@ -0,0 +1,578 @@
|
||||
//! x86_64 CPU operations. This is the "architecture" module: the generic kernel imports
|
||||
//! it as `@import("architecture")` and never names x86_64 directly, so a second
|
||||
//! architecture is added by pointing that module at a different directory in
|
||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
||||
|
||||
const danos = @import("danos");
|
||||
const parameters = @import("parameters");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const paging = @import("paging.zig");
|
||||
const serial = @import("serial.zig");
|
||||
const apic = @import("apic.zig");
|
||||
const ioapic = @import("ioapic.zig");
|
||||
const io = @import("io.zig");
|
||||
const smp = @import("smp.zig");
|
||||
const pcpu = @import("per-cpu.zig");
|
||||
|
||||
/// The saved register/trap frame passed to a fault handler.
|
||||
pub const CpuState = idt.CpuState;
|
||||
|
||||
// --- trap-frame accessors ---------------------------------------------------
|
||||
// The frame's fields are x86_64 registers; the generic kernel reads it through
|
||||
// these accessors so it never names one.
|
||||
|
||||
/// The interrupted/faulting instruction address (RIP here; ELR_EL1 on aarch64,
|
||||
/// sepc on riscv64).
|
||||
pub fn instructionPointer(state: *const CpuState) u64 {
|
||||
return state.rip;
|
||||
}
|
||||
|
||||
/// The interrupted stack pointer (RSP here).
|
||||
pub fn stackPointer(state: *const CpuState) u64 {
|
||||
return state.rsp;
|
||||
}
|
||||
|
||||
/// Whether the trap came from user mode (CPL 3 here; EL0 on aarch64, U-mode on
|
||||
/// riscv64).
|
||||
pub fn fromUser(state: *const CpuState) bool {
|
||||
return state.cs & 3 == 3;
|
||||
}
|
||||
|
||||
/// The faulting virtual address, if this trap is a page fault (CR2 here;
|
||||
/// FAR_EL1 on aarch64, stval on riscv64). Null for any other exception.
|
||||
pub fn faultAddress(state: *const CpuState) ?u64 {
|
||||
if (state.vector != 14) return null;
|
||||
return asm volatile ("mov %%cr2, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
// --- system_call ABI ------------------------------------------------------------
|
||||
// The System V-style register convention (number in rax, arguments in
|
||||
// rdi/rsi/rdx/r10/r8/r9, result in rax), exposed positionally so the generic
|
||||
// dispatcher never names a register.
|
||||
|
||||
/// The system_call number the user program passed.
|
||||
pub fn systemCallNumber(state: *const CpuState) u64 {
|
||||
return state.rax;
|
||||
}
|
||||
|
||||
/// Positional system_call argument `n`.
|
||||
pub fn systemCallArg(state: *const CpuState, n: u8) u64 {
|
||||
return switch (n) {
|
||||
0 => state.rdi,
|
||||
1 => state.rsi,
|
||||
2 => state.rdx,
|
||||
3 => state.r10,
|
||||
4 => state.r8,
|
||||
5 => state.r9,
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Write the system_call's return value into the frame — the entry paths restore
|
||||
/// user registers from it.
|
||||
pub fn setSystemCallResult(state: *CpuState, value: u64) void {
|
||||
state.rax = value;
|
||||
}
|
||||
|
||||
/// Write a *second* system_call return value (rdx here — restored by both the
|
||||
/// system_call/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by
|
||||
/// sysretq). Used by IPC_ReplyWait to hand back the sender's badge alongside the
|
||||
/// message length in rax.
|
||||
pub fn setSystemCallResult2(state: *CpuState, value: u64) void {
|
||||
state.rdx = value;
|
||||
}
|
||||
|
||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
||||
/// so it can be the very first thing called.
|
||||
pub fn serialInit() void {
|
||||
serial.init();
|
||||
}
|
||||
|
||||
/// Write bytes to the serial port.
|
||||
pub fn serialWrite(bytes: []const u8) void {
|
||||
serial.write(bytes);
|
||||
}
|
||||
|
||||
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||
/// displays. The last-resort progress signal when there's no text output at all.
|
||||
/// Writing 0x80 is universally safe (it's the legacy I/O-delay port).
|
||||
pub fn checkpoint(code: u8) void {
|
||||
io.outb(0x80, code);
|
||||
}
|
||||
|
||||
/// Whether a Bochs/QEMU-style debug console is on port 0xE9 (it returns 0xE9 when
|
||||
/// read). On real hardware the port reads back 0xFF, so this stays false — a safe
|
||||
/// probe before we write to it.
|
||||
pub fn debugconPresent() bool {
|
||||
return io.inb(0xE9) == 0xE9;
|
||||
}
|
||||
|
||||
/// Output sink: write bytes to the 0xE9 debug console (see `debugconPresent`).
|
||||
pub fn debugconWrite(bytes: []const u8) void {
|
||||
for (bytes) |b| io.outb(0xE9, b);
|
||||
}
|
||||
|
||||
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
|
||||
/// stack for double faults), then the IDT with exception handlers. After this a
|
||||
/// CPU fault is reported instead of triple-faulting. Install the fault handler
|
||||
/// (setFaultHandler) first so early faults are caught.
|
||||
pub fn init() void {
|
||||
gdt.init();
|
||||
tss.init();
|
||||
idt.init();
|
||||
pcpu.initSystemCall();
|
||||
}
|
||||
|
||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||
/// kernel's segment layout). Call once the frame allocator is up.
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
|
||||
paging.init(allocFrame, freeFrame, boot_information);
|
||||
}
|
||||
|
||||
/// Create a new address space (returns the physical address of its root table —
|
||||
/// the PML4 here — or null). Shares the kernel's higher half; the user (low)
|
||||
/// half starts empty.
|
||||
pub fn createAddressSpace() ?u64 {
|
||||
return paging.createAddressSpace();
|
||||
}
|
||||
|
||||
/// Free an address space and everything mapped in its user half. Caller must not
|
||||
/// be running on it.
|
||||
pub fn destroyAddressSpace(root: u64) void {
|
||||
paging.destroyAddressSpace(root);
|
||||
}
|
||||
|
||||
/// Map a user page into address space `root` (W^X is the caller's contract).
|
||||
pub fn mapUserPageInto(root: u64, virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||
paging.mapUserInto(root, virtual, physical, writable, executable);
|
||||
}
|
||||
|
||||
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
|
||||
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
|
||||
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
paging.mapUserDeviceInto(root, virtual, physical, len);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
||||
paging.map(virtual, physical, writable);
|
||||
}
|
||||
|
||||
/// Map a device MMIO range and return the virtual address to reach it at. This
|
||||
/// is the device layer's `Hal.mapMmio` — it hands back a physmap pointer and
|
||||
/// never exposes how the mapping is placed.
|
||||
pub fn mapMmio(physical: u64, len: u64, writable: bool) u64 {
|
||||
return paging.mapMmio(physical, len, writable);
|
||||
}
|
||||
|
||||
/// The kernel's page-table root (physical), shared into every address space.
|
||||
pub fn kernelPageTable() u64 {
|
||||
return paging.kernelPml4();
|
||||
}
|
||||
|
||||
/// Switch the active address space (load CR3 with a physical root table).
|
||||
pub fn loadPageTable(root: u64) void {
|
||||
paging.loadCr3(root);
|
||||
}
|
||||
|
||||
/// The physical root of the currently active page tables (CR3 here; TTBR0/satp
|
||||
/// elsewhere).
|
||||
pub fn activePageTable() u64 {
|
||||
return asm volatile ("mov %%cr3, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
|
||||
/// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the
|
||||
/// per-CPU `kernel_rsp` (for the system_call stub, which switches by hand). Updated
|
||||
/// by the scheduler when it switches to a user task.
|
||||
pub fn setKernelStack(cpu: usize, top: usize) void {
|
||||
tss.rsp0Ptr(cpu).* = top;
|
||||
pcpu.setKernelRsp(cpu, top);
|
||||
}
|
||||
|
||||
/// Remove a kernel mapping.
|
||||
pub fn unmapPage(virtual: u64) void {
|
||||
paging.unmap(virtual);
|
||||
}
|
||||
|
||||
/// Remove a page mapping from address space `root` (for munmap of user pages).
|
||||
/// Clears the leaf entry only; freeing the underlying frame is the caller's job.
|
||||
pub fn unmapUserPageInto(root: u64, virtual: u64) void {
|
||||
paging.unmapInto(root, virtual);
|
||||
}
|
||||
|
||||
/// Resolve `virtual` to its physical address in the address space rooted at `root`
|
||||
/// (any address space, not just the live one), or null if unmapped. Used to find
|
||||
/// the frame behind a user page for munmap, and for cross-address-space copies.
|
||||
pub fn translate(root: u64, virtual: u64) ?u64 {
|
||||
return paging.translateIn(root, virtual);
|
||||
}
|
||||
|
||||
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
||||
/// W^X: code read-only + executable, data writable + no-execute.
|
||||
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||
paging.mapUser(virtual, physical, writable, executable);
|
||||
}
|
||||
|
||||
// --- ring 3 entry/exit -----------------------------------------------------
|
||||
|
||||
/// Drop to ring 3 at `rip` on `rsp` (defined in isr.s). Saves the kernel context,
|
||||
/// publishes the kernel stack pointer through `rsp0_slot` (this core's TSS.rsp0,
|
||||
/// so ring-3 interrupts land on a good stack), builds an iretq frame with the
|
||||
/// user selectors, and iretq's. "Returns" only when the user program triggers
|
||||
/// the exit path (user_exit_to_kernel).
|
||||
extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void;
|
||||
|
||||
/// Abandon the in-flight ring-3 trap context and resume the kernel as if
|
||||
/// `enter_user` had returned (defined in isr.s). Called by the exit system_call.
|
||||
extern fn user_exit_to_kernel() callconv(.c) noreturn;
|
||||
|
||||
/// Run user code at `entry` with stack `stack_top` on this core (`cpu` = the
|
||||
/// caller's CPU index; the architecture layer can't ask the scheduler). Returns after the
|
||||
/// user program exits via system_call. Interrupts are disabled on return (the exit
|
||||
/// arrives through an interrupt gate) — the caller re-enables.
|
||||
pub fn enterUser(cpu: usize, entry: u64, stack_top: u64) void {
|
||||
enter_user(entry, stack_top, tss.rsp0Ptr(cpu));
|
||||
}
|
||||
|
||||
/// Never returns to the user program: unwind to the kernel context that called
|
||||
/// `enterUser`. For the exit system_call's handler.
|
||||
pub fn userExit() noreturn {
|
||||
user_exit_to_kernel();
|
||||
}
|
||||
|
||||
/// Register the handler for the user system_call gate (int 0x80, vector 128). The
|
||||
/// handler may write the trap frame (see `setSystemCallResult`).
|
||||
pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void {
|
||||
idt.setSystemCallHandler(handler);
|
||||
}
|
||||
|
||||
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
|
||||
/// core calls this once, after its GDT is in place (a GS *selector* reload would
|
||||
/// clobber the base). See percpu.zig for the swapgs discipline.
|
||||
pub fn setCpuLocal(cpu: usize, ptr: usize) void {
|
||||
pcpu.setLocal(cpu, ptr);
|
||||
}
|
||||
|
||||
/// This core's scheduler pointer (via the GS base) — a per-core register, so each
|
||||
/// core sees its own without locking. Valid in any ring-0 context.
|
||||
pub fn cpuLocal() usize {
|
||||
return pcpu.scheduler();
|
||||
}
|
||||
|
||||
// --- SMP: application-processor bring-up ----------------------------------
|
||||
|
||||
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
|
||||
/// The frame stays inert (zeroed, non-executable) between wakes and is armed only
|
||||
/// while a core is climbing — so a core can be (re)woken at any time (retry, or a
|
||||
/// future power manager) without leaving an executable page resident. See smp.zig.
|
||||
pub fn setTrampolinePage(physical: u64) void {
|
||||
smp.setTrampolinePage(physical);
|
||||
}
|
||||
|
||||
/// Wake the core with hardware id `hw_id` (its Local APIC id here; MPIDR on
|
||||
/// aarch64, hart id on riscv64) as dense CPU `index`, giving it `stack_top` and
|
||||
/// its per-CPU pointer `percpu`; it adopts the kernel page tables. Returns false
|
||||
/// if it doesn't come online within the timeout. Blocks until the core reports in.
|
||||
pub fn startSecondary(hw_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
|
||||
// The AP adopts the kernel page tables explicitly — never the caller's live
|
||||
// CR3, which a future re-wake from a core running a process would make a
|
||||
// process address space.
|
||||
return smp.startAp(hw_id, stack_top, percpu, index, paging.kernelPml4());
|
||||
}
|
||||
|
||||
/// Register the generic entry a woken AP jumps to once its architecture state is up (its own
|
||||
/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here.
|
||||
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
||||
smp.setSecondaryEntry(entry);
|
||||
}
|
||||
|
||||
/// Bytes the kernel should allocate for a secondary core's dedicated fault stack
|
||||
/// (the IST double-fault stack here), and where to record its top before waking
|
||||
/// the core. The stack is heap-allocated per online core (the boot CPU's is
|
||||
/// static — it's needed before the allocator exists). See tss.zig.
|
||||
pub const fault_stack_size = tss.ist_stack_size;
|
||||
pub fn setFaultStack(cpu: usize, top: usize) void {
|
||||
tss.setApIstStack(cpu, top);
|
||||
}
|
||||
|
||||
/// Test hook: force the next `n` AP wake attempts to fail, so the retry path can be
|
||||
/// exercised deterministically (see the smp-retry test). No effect when `n` is 0.
|
||||
pub fn testFailNextWakes(n: u32) void {
|
||||
smp.testFailNextWakes(n);
|
||||
}
|
||||
|
||||
/// The reserved AP-trampoline frame (0 if none). For tests that check it's inert.
|
||||
pub fn trampolinePage() u64 {
|
||||
return smp.trampolinePage();
|
||||
}
|
||||
|
||||
/// Whether the page at `virtual` is currently mapped executable (present, NX clear).
|
||||
pub fn pageExecutable(virtual: u64) bool {
|
||||
return paging.isExecutable(virtual);
|
||||
}
|
||||
|
||||
/// Kernel tick rate (the scheduler's time quantum), from configuration.
|
||||
pub const timer_hz = parameters.timer_hz;
|
||||
|
||||
/// The ACPI PM timer, as a calibration reference (re-exported for the configuration).
|
||||
pub const PmTimer = apic.PmTimer;
|
||||
/// A MADT interrupt-source override (re-exported for the configuration).
|
||||
pub const IsoEntry = ioapic.IsoEntry;
|
||||
|
||||
/// Discovered platform facts the architecture layer needs so it makes no legacy
|
||||
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
||||
pub const PlatformConfiguration = struct {
|
||||
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
||||
pic_present: bool = true,
|
||||
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
||||
hpet_base: u64 = 0,
|
||||
/// The ACPI PM timer, another calibration reference.
|
||||
pm_timer: ?PmTimer = null,
|
||||
/// I/O APIC MMIO base + its first global system interrupt (0 = none).
|
||||
ioapic_base: u64 = 0,
|
||||
ioapic_gsi_base: u32 = 0,
|
||||
/// MADT ISA-IRQ overrides, for I/O APIC routing.
|
||||
overrides: []const IsoEntry = &.{},
|
||||
};
|
||||
|
||||
/// Apply the discovered platform configuration. Must run before `startTimer` (the timer
|
||||
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
||||
/// Maps + masks the I/O APIC immediately.
|
||||
pub fn configurePlatform(configuration: PlatformConfiguration) void {
|
||||
apic.configure(configuration.pic_present, configuration.hpet_base, configuration.pm_timer);
|
||||
ioapic.configure(configuration.ioapic_base, configuration.ioapic_gsi_base, configuration.overrides);
|
||||
ioapic.init();
|
||||
}
|
||||
|
||||
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
||||
pub fn serialReconfigure(is_mmio: bool, address: u64) void {
|
||||
serial.reconfigure(is_mmio, address);
|
||||
}
|
||||
|
||||
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
||||
pub fn timerCalibrationSource() []const u8 {
|
||||
return apic.calibrationSource();
|
||||
}
|
||||
|
||||
/// External-interrupt-router diagnostics, for boot logging / verification (the
|
||||
/// I/O APIC's redirection entries here; a GIC distributor or PLIC elsewhere).
|
||||
pub fn irqRouteCount() u32 {
|
||||
return ioapic.entryCount();
|
||||
}
|
||||
pub fn irqRouteRaw(n: u32) u32 {
|
||||
return ioapic.entryLow(n);
|
||||
}
|
||||
|
||||
// --- device-IRQ plumbing, for system/kernel/irq.zig -----------------------------
|
||||
//
|
||||
// The generic IRQ layer speaks GSIs and vectors; everything below hides the fact
|
||||
// that on x86_64 those mean "I/O APIC redirection entry" and "IDT gate". The
|
||||
// vector window is bounded by the stubs isr.s actually emits: `gate_count` = 48,
|
||||
// vector 32 is the LAPIC timer and 47 is the spurious vector, leaving 33..46.
|
||||
|
||||
pub const irq_vector_base: u8 = 33;
|
||||
pub const irq_vector_count: u8 = 14; // 33..46 inclusive
|
||||
|
||||
/// True if `gsi` is one this machine's interrupt router can deliver.
|
||||
pub fn irqOwnsGsi(gsi: u32) bool {
|
||||
return ioapic.ownsGsi(gsi);
|
||||
}
|
||||
|
||||
/// Install `handler` on `vector` (an absolute IDT gate index).
|
||||
pub fn irqSetHandler(vector: u8, handler: *const fn () void) void {
|
||||
idt.setHandler(vector, handler);
|
||||
}
|
||||
|
||||
/// Route `gsi` to `vector` on *this* core, masked. Unmask with `irqUnmask` once bound.
|
||||
pub fn irqRoute(gsi: u32, vector: u8, level: bool, active_low: bool) void {
|
||||
ioapic.routeGsi(gsi, vector, apic.localId(), level, active_low);
|
||||
}
|
||||
|
||||
pub fn irqMask(gsi: u32) void {
|
||||
ioapic.maskGsi(gsi);
|
||||
}
|
||||
pub fn irqUnmask(gsi: u32) void {
|
||||
ioapic.unmaskGsi(gsi);
|
||||
}
|
||||
|
||||
/// Acknowledge the interrupt currently in service on this core's LAPIC.
|
||||
pub fn irqEoi() void {
|
||||
apic.eoi();
|
||||
}
|
||||
|
||||
/// Enable the Local APIC, calibrate its timer against the best available reference
|
||||
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
||||
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
||||
/// unmasked with enableInterrupts() to be delivered. Run `configurePlatform` first.
|
||||
pub fn startTimer() void {
|
||||
apic.init();
|
||||
apic.calibrate();
|
||||
idt.setHandler(apic.timer_vector, apic.timerTick);
|
||||
apic.initTimer(timer_hz);
|
||||
}
|
||||
|
||||
/// Number of timer ticks since startTimer().
|
||||
pub fn ticks() u64 {
|
||||
return apic.ticks();
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock (from the TSC), one function per resolution.
|
||||
pub fn nanos() u64 {
|
||||
return apic.nanos();
|
||||
}
|
||||
pub fn micros() u64 {
|
||||
return apic.micros();
|
||||
}
|
||||
pub fn millis() u64 {
|
||||
return apic.millis();
|
||||
}
|
||||
|
||||
/// Measured frequency of the tick timer's input clock (the LAPIC timer here), in
|
||||
/// Hz, from calibration.
|
||||
pub fn timerClockHz() u64 {
|
||||
return apic.lapicHz();
|
||||
}
|
||||
|
||||
/// Measured frequency of the monotonic clock's underlying counter (the TSC here;
|
||||
/// CNTVCT on aarch64, `time` on riscv64), in Hz.
|
||||
pub fn clockHz() u64 {
|
||||
return apic.tscHz();
|
||||
}
|
||||
|
||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||
pub fn enableInterrupts() void {
|
||||
asm volatile ("sti");
|
||||
}
|
||||
|
||||
/// Mask maskable interrupts (`cli`).
|
||||
pub fn disableInterrupts() void {
|
||||
asm volatile ("cli");
|
||||
}
|
||||
|
||||
/// Disable interrupts and return the previous flags, so a nested critical section
|
||||
/// can restore the caller's state rather than blindly re-enabling. Pairs with
|
||||
/// restoreInterrupts.
|
||||
pub fn saveInterrupts() u64 {
|
||||
var flags: u64 = undefined;
|
||||
asm volatile (
|
||||
\\pushfq
|
||||
\\pop %[f]
|
||||
\\cli
|
||||
: [f] "=r" (flags),
|
||||
:
|
||||
: .{ .memory = true }
|
||||
);
|
||||
return flags;
|
||||
}
|
||||
|
||||
/// Re-enable interrupts only if they were enabled when `flags` was captured.
|
||||
pub fn restoreInterrupts(flags: u64) void {
|
||||
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
|
||||
}
|
||||
|
||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
apic.setTickHook(hook);
|
||||
}
|
||||
|
||||
// --- context switching (for the scheduler) -------------------------------
|
||||
|
||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
||||
|
||||
pub fn switchContext(old_sp: *usize, new_sp: usize) void {
|
||||
switch_context(old_sp, new_sp);
|
||||
}
|
||||
|
||||
/// Build the initial stack for a new task so that switching to it lands in
|
||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
||||
/// The layout must match switch_context's push order (callee-saved, then the
|
||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
||||
var sp = stack_top;
|
||||
const push = struct {
|
||||
fn f(p: *usize, value: usize) void {
|
||||
p.* -= @sizeOf(usize);
|
||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
||||
}
|
||||
}.f;
|
||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
||||
push(&sp, 0); // rbx
|
||||
push(&sp, 0); // rbp
|
||||
push(&sp, 0); // r12
|
||||
push(&sp, 0); // r13
|
||||
push(&sp, 0); // r14
|
||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
||||
return sp;
|
||||
}
|
||||
|
||||
/// Drop the current (kernel-context) task to ring 3 at `rip` on `rsp`, never
|
||||
/// returning (defined in isr.s). Used by the scheduler's user-task trampoline
|
||||
/// once it has switched onto the task and read its entry/stack. Interrupts are
|
||||
/// disabled across the swapgs+iretq so no interrupt observes the user GS base in
|
||||
/// ring 0; the pushed RFLAGS re-enables them in ring 3.
|
||||
extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn;
|
||||
|
||||
pub fn jumpToUser(entry: u64, stack_top: u64) noreturn {
|
||||
jump_to_user(entry, stack_top);
|
||||
}
|
||||
|
||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||
/// return. Until set, faults just halt the core.
|
||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||
idt.on_fault = handler;
|
||||
}
|
||||
|
||||
/// A human-readable name for a CPU exception vector.
|
||||
pub fn exceptionName(vector: u64) []const u8 {
|
||||
return idt.vectorName(vector);
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1/2/4) from an I/O port. The generic device layer drives
|
||||
/// ACPI registers through this rather than naming x86 port instructions; on an
|
||||
/// MMIO-only architecture this would be implemented differently.
|
||||
pub fn pioRead(width: u8, port: u16) u32 {
|
||||
return switch (width) {
|
||||
1 => io.inb(port),
|
||||
2 => io.inw(port),
|
||||
4 => io.inl(port),
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Write `width` bytes (1/2/4) to an I/O port.
|
||||
pub fn pioWrite(width: u8, port: u16, value: u32) void {
|
||||
switch (width) {
|
||||
1 => io.outb(port, @truncate(value)),
|
||||
2 => io.outw(port, @truncate(value)),
|
||||
4 => io.outl(port, value),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
||||
/// docs/halting.md for the full reasoning.
|
||||
pub fn halt() noreturn {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
/// Spin-wait hint (`pause`). Emitted in the body of a spinlock's busy-wait: it
|
||||
/// relaxes the core while it polls a contended lock — yielding pipeline resources
|
||||
/// to a hyperthread sibling and easing the cache-coherency traffic on the lock
|
||||
/// line. Purely a performance/power hint; correct to omit, but kinder on the bus.
|
||||
pub fn cpuRelax() void {
|
||||
asm volatile ("pause");
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
|
||||
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
|
||||
//! reference a code selector — so we install our own flat GDT with known
|
||||
//! selectors (0x08/0x10 kernel code/data, 0x18/0x20 user data/code for ring 3)
|
||||
//! rather than trusting whatever the firmware left in place.
|
||||
//!
|
||||
//! The code/data descriptors are identical on every core, but the **TSS descriptor
|
||||
//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see
|
||||
//! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its
|
||||
//! own copy of the table with its own TSS descriptor. Slot 0 is the BSP.
|
||||
|
||||
const parameters = @import("parameters");
|
||||
|
||||
/// Selectors into the table (index * 8). Same on every core's GDT.
|
||||
pub const kernel_code = 0x08;
|
||||
pub const kernel_data = 0x10;
|
||||
pub const user_data = 0x18;
|
||||
pub const user_code = 0x20;
|
||||
pub const tss_selector = 0x28;
|
||||
|
||||
/// Ring-3 selectors as loaded from user mode: RPL 3 or'd in. The user *data*
|
||||
/// descriptor is load-bearing even in long mode — iretq to CPL 3 with a null SS
|
||||
/// raises #GP(0).
|
||||
pub const user_code_rpl3 = user_code | 3;
|
||||
pub const user_data_rpl3 = user_data | 3;
|
||||
|
||||
const maximum_cpus = parameters.maximum_cpus;
|
||||
const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high
|
||||
|
||||
/// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor,
|
||||
/// filled in per core by `setTssFor`.
|
||||
/// kernel code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
||||
/// kernel data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
||||
/// user data: present, ring 3, writable -> 0x00CFF200_0000FFFF
|
||||
/// user code: present, ring 3, executable, readable, L=1 -> 0x00AFFA00_0000FFFF
|
||||
/// User data sits below user code so a future SYSRET works unchanged: it loads
|
||||
/// CS = STAR.SYSRET_CS + 16 and SS = STAR.SYSRET_CS + 8, so with SYSRET_CS = 0x10
|
||||
/// those land on 0x20 (user code) and 0x18 (user data).
|
||||
const template = [entries]u64{
|
||||
0, // null descriptor (required)
|
||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
||||
0x00CF92000000FFFF, // kernel data (0x10)
|
||||
0x00CFF2000000FFFF, // user data (0x18)
|
||||
0x00AFFA000000FFFF, // user code (0x20)
|
||||
0, // TSS descriptor low (0x28)
|
||||
0, // TSS descriptor high
|
||||
};
|
||||
|
||||
/// One GDT per core (each a copy of the template, differing only in its TSS slot).
|
||||
var gdts = [_][entries]u64{template} ** maximum_cpus;
|
||||
|
||||
/// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task
|
||||
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
|
||||
/// TSS. Write it into that core's GDT before it loads the TSS selector.
|
||||
pub fn setTssFor(cpu: usize, base: u64, limit: u64) void {
|
||||
gdts[cpu][5] = (limit & 0xFFFF) |
|
||||
((base & 0xFFFF) << 16) |
|
||||
(((base >> 16) & 0xFF) << 32) |
|
||||
(@as(u64, 0x89) << 40) |
|
||||
(((limit >> 16) & 0xF) << 48) |
|
||||
(((base >> 24) & 0xFF) << 56);
|
||||
gdts[cpu][6] = (base >> 32) & 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
||||
const Descriptor = packed struct {
|
||||
limit: u16,
|
||||
base: u64,
|
||||
};
|
||||
|
||||
/// Loads the GDT and reloads the segment registers (including CS). Defined in
|
||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match the table.
|
||||
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
||||
|
||||
/// Load core `cpu`'s GDT and switch onto its segments. Note this reloads the segment
|
||||
/// registers, which zeroes the GS base — so a core must publish its per-CPU pointer
|
||||
/// (setCpuLocal) *after* calling this.
|
||||
pub fn loadOnThisCpu(cpu: usize) void {
|
||||
const descriptor = Descriptor{
|
||||
.limit = @sizeOf([entries]u64) - 1,
|
||||
.base = @intFromPtr(&gdts[cpu]),
|
||||
};
|
||||
gdt_flush(&descriptor);
|
||||
}
|
||||
|
||||
/// Install the bootstrap processor's GDT (slot 0) and switch onto its segments.
|
||||
pub fn init() void {
|
||||
loadOnThisCpu(0);
|
||||
}
|
||||
@@ -0,0 +1,186 @@
|
||||
//! Interrupt Descriptor Table, CPU-exception handlers, and device-interrupt
|
||||
//! dispatch. Without this, any fault (a stray pointer, a bad page-table entry)
|
||||
//! triple-faults and silently resets the machine. With it, the CPU vectors into
|
||||
//! our stubs, which capture the register state and hand it to a dispatcher.
|
||||
//!
|
||||
//! Vectors split in two: 0-31 are CPU exceptions (terminal — reported and
|
||||
//! halted); 32+ are device interrupts (a registered handler runs, the APIC is
|
||||
//! acknowledged, and we return to the interrupted code).
|
||||
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
|
||||
/// Highest vector we install a gate/stub for (exceptions 0-31 plus the device
|
||||
/// range 32-47, which covers the timer and the spurious vector).
|
||||
const gate_count = 48;
|
||||
|
||||
/// A device-interrupt handler. It doesn't get the trap frame (a timer or keyboard
|
||||
/// handler doesn't need the interrupted registers); add that if one ever does.
|
||||
pub const Handler = *const fn () void;
|
||||
|
||||
var handlers = [_]?Handler{null} ** 256;
|
||||
|
||||
/// Register `handler` for a device-interrupt `vector` (>= 32).
|
||||
pub fn setHandler(vector: usize, handler: Handler) void {
|
||||
handlers[vector] = handler;
|
||||
}
|
||||
|
||||
/// The ring-3 system_call gate's vector (`int $0x80`, the classic choice — well away
|
||||
/// from the device range) and its handler. Unlike device handlers, a system_call
|
||||
/// handler gets the (mutable) trap frame: it reads its arguments from the saved
|
||||
/// user registers and writes rax as the return value, which isr_common then
|
||||
/// restores into the user context.
|
||||
pub const system_call_vector = 128;
|
||||
|
||||
var system_call_handler: ?*const fn (*CpuState) void = null;
|
||||
|
||||
pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void {
|
||||
system_call_handler = handler;
|
||||
}
|
||||
|
||||
/// The register + trap frame the ISR stubs build on the stack, laid out so the
|
||||
/// lowest address (where RSP points when we call the handler) is the first field.
|
||||
/// See the push order in `isrCommon` below.
|
||||
pub const CpuState = extern struct {
|
||||
r15: u64,
|
||||
r14: u64,
|
||||
r13: u64,
|
||||
r12: u64,
|
||||
r11: u64,
|
||||
r10: u64,
|
||||
r9: u64,
|
||||
r8: u64,
|
||||
rbp: u64,
|
||||
rdi: u64,
|
||||
rsi: u64,
|
||||
rdx: u64,
|
||||
rcx: u64,
|
||||
rbx: u64,
|
||||
rax: u64,
|
||||
vector: u64, // pushed by the per-vector stub
|
||||
error_code: u64, // real one from the CPU, or 0 pushed by the stub
|
||||
rip: u64, // from here down: pushed by the CPU on entry
|
||||
cs: u64,
|
||||
rflags: u64,
|
||||
rsp: u64,
|
||||
ss: u64,
|
||||
};
|
||||
|
||||
/// Where a fault is reported. The kernel overrides this (see setFaultHandler) with
|
||||
/// something that prints to the console; until then, just stop.
|
||||
pub var on_fault: *const fn (*const CpuState) noreturn = defaultFault;
|
||||
|
||||
fn defaultFault(_: *const CpuState) noreturn {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
/// Names for the 32 defined exception vectors, for readable output.
|
||||
const names = [_][]const u8{
|
||||
"divide error", "debug",
|
||||
"NMI", "breakpoint",
|
||||
"overflow", "bound range exceeded",
|
||||
"invalid opcode", "device not available",
|
||||
"double fault", "coprocessor segment overrun",
|
||||
"invalid TSS", "segment not present",
|
||||
"stack-segment fault", "general protection fault",
|
||||
"page fault", "reserved (15)",
|
||||
"x87 floating-point", "alignment check",
|
||||
"machine check", "SIMD floating-point",
|
||||
"virtualization", "control protection",
|
||||
"reserved (22)", "reserved (23)",
|
||||
"reserved (24)", "reserved (25)",
|
||||
"reserved (26)", "reserved (27)",
|
||||
"hypervisor injection", "VMM communication",
|
||||
"security exception", "reserved (31)",
|
||||
};
|
||||
|
||||
pub fn vectorName(vector: u64) []const u8 {
|
||||
return if (vector < names.len) names[vector] else "unknown";
|
||||
}
|
||||
|
||||
/// A 64-bit IDT gate descriptor (16 bytes).
|
||||
const Gate = packed struct {
|
||||
offset_low: u16,
|
||||
selector: u16,
|
||||
ist: u8, // interrupt-stack-table index; 0 = use the current stack
|
||||
flags: u8, // present, DPL, gate type
|
||||
offset_mid: u16,
|
||||
offset_high: u32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
var idt = [_]Gate{std.mem.zeroes(Gate)} ** 256;
|
||||
|
||||
const Descriptor = packed struct {
|
||||
limit: u16,
|
||||
base: u64,
|
||||
};
|
||||
|
||||
/// Loads the IDT (`lidt`). Defined in isr.s.
|
||||
extern fn idt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
||||
|
||||
fn setGate(vector: usize, handler: u64) void {
|
||||
idt[vector] = .{
|
||||
.offset_low = @truncate(handler),
|
||||
.selector = gdt.kernel_code,
|
||||
.ist = 0,
|
||||
.flags = 0x8E, // present, ring 0, 64-bit interrupt gate
|
||||
.offset_mid = @truncate(handler >> 16),
|
||||
.offset_high = @truncate(handler >> 32),
|
||||
};
|
||||
}
|
||||
|
||||
/// Point every installed vector at its stub (isr.s) and load the IDT.
|
||||
pub fn init() void {
|
||||
@setEvalBranchQuota(20000); // comptimePrint across all the gates adds up
|
||||
inline for (0..gate_count) |vector| {
|
||||
const stub = @extern(*const anyopaque, .{ .name = std.fmt.comptimePrint("isr{d}", .{vector}) });
|
||||
setGate(vector, @intFromPtr(stub));
|
||||
}
|
||||
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
|
||||
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
|
||||
idt[8].ist = tss.double_fault_ist;
|
||||
// The system_call gate. Installed outside the 0..gate_count loop (stubs 48-127
|
||||
// don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a
|
||||
// #GP. An interrupt gate (not trap): IF is cleared for the handler, which
|
||||
// the ring-3 exit path relies on.
|
||||
const system_call_stub = @extern(*const anyopaque, .{ .name = "isr128" });
|
||||
setGate(system_call_vector, @intFromPtr(system_call_stub));
|
||||
idt[system_call_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate
|
||||
loadOnThisCpu();
|
||||
}
|
||||
|
||||
/// Load the (shared, already-populated) IDT on the current core. The gate table is
|
||||
/// read-only after `init`, so every core points its IDTR at the same one. Called by
|
||||
/// the BSP via `init` and by each AP during bring-up.
|
||||
pub fn loadOnThisCpu() void {
|
||||
const descriptor = Descriptor{
|
||||
.limit = @sizeOf(@TypeOf(idt)) - 1,
|
||||
.base = @intFromPtr(&idt),
|
||||
};
|
||||
idt_flush(&descriptor);
|
||||
}
|
||||
|
||||
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
|
||||
/// assembly stubs can `call` it by name. Exceptions are terminal; device
|
||||
/// interrupts run their handler, get acknowledged, and return.
|
||||
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
|
||||
if (state.vector < 32) {
|
||||
on_fault(state); // CPU exception — never returns
|
||||
} else if (state.vector == system_call_vector) {
|
||||
// Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI.
|
||||
if (system_call_handler) |handler| handler(state);
|
||||
} else if (handlers[state.vector]) |handler| {
|
||||
// The handler owns its EOI. It used to be issued here, before the call —
|
||||
// correct for the LAPIC timer, but impossible to reconcile with a
|
||||
// level-triggered device line, which must be **masked at the I/O APIC
|
||||
// before** it is acknowledged or it redelivers instantly and storms
|
||||
// (the driver that would quiet it lives in ring 3 and hasn't run yet).
|
||||
// Only the handler knows which discipline its source needs, so only the
|
||||
// handler can sequence it. See apic.timerTick and irq.dispatch.
|
||||
handler();
|
||||
}
|
||||
// else: spurious/unhandled device interrupt — don't acknowledge it
|
||||
}
|
||||
|
||||
const std = @import("std");
|
||||
@@ -0,0 +1,68 @@
|
||||
//! x86 port I/O and model-specific registers — the low-level primitives the
|
||||
//! serial port and the APIC talk to hardware through.
|
||||
|
||||
pub fn outb(port: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[port]"
|
||||
:
|
||||
: [value] "{al}" (value),
|
||||
[port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn inb(port: u16) u8 {
|
||||
return asm volatile ("inb %[port], %[value]"
|
||||
: [value] "={al}" (-> u8),
|
||||
: [port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn outw(port: u16, value: u16) void {
|
||||
asm volatile ("outw %[value], %[port]"
|
||||
:
|
||||
: [value] "{ax}" (value),
|
||||
[port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn inw(port: u16) u16 {
|
||||
return asm volatile ("inw %[port], %[value]"
|
||||
: [value] "={ax}" (-> u16),
|
||||
: [port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn outl(port: u16, value: u32) void {
|
||||
asm volatile ("outl %[value], %[port]"
|
||||
:
|
||||
: [value] "{eax}" (value),
|
||||
[port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn inl(port: u16) u32 {
|
||||
return asm volatile ("inl %[port], %[value]"
|
||||
: [value] "={eax}" (-> u32),
|
||||
: [port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
/// Read a model-specific register (returns edx:eax combined).
|
||||
pub fn rdmsr(msr: u32) u64 {
|
||||
var low: u32 = undefined;
|
||||
var high: u32 = undefined;
|
||||
asm volatile ("rdmsr"
|
||||
: [low] "={eax}" (low),
|
||||
[high] "={edx}" (high),
|
||||
: [msr] "{ecx}" (msr),
|
||||
);
|
||||
return (@as(u64, high) << 32) | low;
|
||||
}
|
||||
|
||||
pub fn wrmsr(msr: u32, value: u64) void {
|
||||
asm volatile ("wrmsr"
|
||||
:
|
||||
: [msr] "{ecx}" (msr),
|
||||
[low] "{eax}" (@as(u32, @truncate(value))),
|
||||
[high] "{edx}" (@as(u32, @truncate(value >> 32))),
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,152 @@
|
||||
//! I/O APIC — routes external device interrupts (a device's line) to a LAPIC
|
||||
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
||||
//! ACPI's MADT (via discovery), never assumed.
|
||||
//!
|
||||
//! `init` maps the I/O APIC and **masks every input** — the correct quiescent state
|
||||
//! on a legacy-free machine. Lines are then unmasked one at a time, as user-space
|
||||
//! drivers bind them (`routeGsi`/`unmaskGsi`, driven by system/kernel/irq.zig).
|
||||
//!
|
||||
//! Two entry points, for two kinds of caller. `routeIrq` takes a legacy **ISA IRQ**
|
||||
//! and resolves it through the MADT overrides — for in-kernel use, and still without
|
||||
//! a caller. `routeGsi` takes a **GSI** directly, which is what a device's own
|
||||
//! routing capability names (e.g. the HPET's `Tn_INT_ROUTE_CAP`), and is the path a
|
||||
//! bound driver interrupt takes.
|
||||
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// A MADT Interrupt Source Override: an ISA IRQ that appears at a different global
|
||||
/// system interrupt, with its own polarity/trigger (MPS INTI `flags`).
|
||||
pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
|
||||
|
||||
var base: u64 = 0; // 0 = no I/O APIC discovered
|
||||
var gsi_base: u32 = 0;
|
||||
var maximum_entries: u32 = 0;
|
||||
var overrides: [16]IsoEntry = undefined;
|
||||
var override_count: usize = 0;
|
||||
|
||||
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
||||
const register_ioregsel = 0x00;
|
||||
const register_iowin = 0x10;
|
||||
const register_version = 0x01;
|
||||
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
||||
const redir_mask = 1 << 16; // mask bit in the low dword
|
||||
|
||||
/// Supply the discovered I/O APIC location + the MADT IRQ overrides. Call before `init`.
|
||||
pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) void {
|
||||
base = ioapic_base;
|
||||
gsi_base = ioapic_gsi_base;
|
||||
override_count = @min(isos.len, overrides.len);
|
||||
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
||||
}
|
||||
|
||||
fn registerRead(index: u32) u32 {
|
||||
@as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index;
|
||||
return @as(*volatile u32, @ptrFromInt(base + register_iowin)).*;
|
||||
}
|
||||
fn registerWrite(index: u32, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index;
|
||||
@as(*volatile u32, @ptrFromInt(base + register_iowin)).* = value;
|
||||
}
|
||||
|
||||
fn writeEntry(n: u32, low: u32, high: u32) void {
|
||||
registerWrite(redir_base + 2 * n, low);
|
||||
registerWrite(redir_base + 2 * n + 1, high);
|
||||
}
|
||||
|
||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
||||
pub fn init() void {
|
||||
if (base == 0) return;
|
||||
// Reach the I/O APIC through the physmap; switch `base` to that virtual
|
||||
// address so the register accessors work without the identity map.
|
||||
base = paging.mapMmio(base, 0x1000, true);
|
||||
maximum_entries = ((registerRead(register_version) >> 16) & 0xFF) + 1;
|
||||
var n: u32 = 0;
|
||||
while (n < maximum_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
||||
}
|
||||
|
||||
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
||||
/// for its GSI/polarity/trigger, and unmask it. No caller yet — groundwork for the
|
||||
/// first device driver.
|
||||
pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
||||
if (base == 0) return;
|
||||
|
||||
var gsi: u32 = irq;
|
||||
var flags: u16 = 0;
|
||||
for (overrides[0..override_count]) |o| {
|
||||
if (o.source == irq) {
|
||||
gsi = o.gsi;
|
||||
flags = o.flags;
|
||||
}
|
||||
}
|
||||
if (gsi < gsi_base) return;
|
||||
const n = gsi - gsi_base;
|
||||
if (n >= maximum_entries) return;
|
||||
|
||||
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
||||
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
||||
var low: u32 = vector;
|
||||
if (flags & 0x3 == 3) low |= (1 << 13);
|
||||
if ((flags >> 2) & 0x3 == 3) low |= (1 << 15);
|
||||
const high: u32 = @as(u32, apic_id) << 24; // destination APIC ID
|
||||
writeEntry(n, low, high);
|
||||
}
|
||||
|
||||
// --- GSI-level control (the user-space driver path) --------------------------
|
||||
//
|
||||
// `routeIrq` above takes an *ISA IRQ* and resolves it through the MADT overrides.
|
||||
// A driver-bound interrupt is already a **GSI** (the device told us so, e.g. the
|
||||
// HPET's `Tn_INT_ROUTE_CAP`), so it needs no override lookup — just the redirection
|
||||
// entry. These three are what `system/kernel/irq.zig` drives.
|
||||
//
|
||||
// Callers must serialise: the I/O APIC is reached through an index/data register
|
||||
// pair, so two cores interleaving `registerWrite` would corrupt each other. The kernel
|
||||
// holds the big lock across these.
|
||||
|
||||
/// Redirection-entry index for `gsi`, or null if this I/O APIC doesn't own it.
|
||||
fn entryFor(gsi: u32) ?u32 {
|
||||
if (base == 0 or gsi < gsi_base) return null;
|
||||
const n = gsi - gsi_base;
|
||||
return if (n < maximum_entries) n else null;
|
||||
}
|
||||
|
||||
/// True if `gsi` lands on this I/O APIC — the kernel's validity check before binding.
|
||||
pub fn ownsGsi(gsi: u32) bool {
|
||||
return entryFor(gsi) != null;
|
||||
}
|
||||
|
||||
/// Point `gsi` at `vector` on the LAPIC `apic_id`, with explicit polarity/trigger,
|
||||
/// and leave it **masked**. The caller unmasks once a handler is bound — otherwise a
|
||||
/// device asserting between route and bind would fire into a null handler.
|
||||
pub fn routeGsi(gsi: u32, vector: u8, apic_id: u8, level: bool, active_low: bool) void {
|
||||
const n = entryFor(gsi) orelse return;
|
||||
var low: u32 = @as(u32, vector) | redir_mask; // masked until bound
|
||||
if (active_low) low |= (1 << 13);
|
||||
if (level) low |= (1 << 15);
|
||||
writeEntry(n, low, @as(u32, apic_id) << 24);
|
||||
}
|
||||
|
||||
/// Stop `gsi` reaching any CPU. Called from the ISR *before* the LAPIC EOI: a
|
||||
/// level-triggered line is still asserted at that point, so an unmasked entry would
|
||||
/// redeliver immediately and storm before the user-space driver ever runs.
|
||||
pub fn maskGsi(gsi: u32) void {
|
||||
const n = entryFor(gsi) orelse return;
|
||||
registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) | redir_mask);
|
||||
}
|
||||
|
||||
/// Let `gsi` through again — the tail of `irq_ack`, once the driver has quieted the
|
||||
/// device (so the line is deasserted and this can't immediately refire).
|
||||
pub fn unmaskGsi(gsi: u32) void {
|
||||
const n = entryFor(gsi) orelse return;
|
||||
registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) & ~@as(u32, redir_mask));
|
||||
}
|
||||
|
||||
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
||||
pub fn entryCount() u32 {
|
||||
return maximum_entries;
|
||||
}
|
||||
|
||||
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
||||
pub fn entryLow(n: u32) u32 {
|
||||
if (base == 0) return 0;
|
||||
return registerRead(redir_base + 2 * n);
|
||||
}
|
||||
@@ -0,0 +1,382 @@
|
||||
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
||||
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
||||
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
|
||||
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
||||
#
|
||||
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
||||
# error code where the CPU pushes none, then the vector number — and jumps to the
|
||||
# shared tail, which saves the general registers and calls the Zig handler with a
|
||||
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
||||
|
||||
.text
|
||||
|
||||
# _start: the kernel entry. The loader jumps here (higher-half address) with
|
||||
# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned
|
||||
# stack in .bss (the loader stack is a low address that goes away once the low
|
||||
# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never
|
||||
# returns; the hlt loop is a belt-and-braces backstop.
|
||||
.global _start
|
||||
_start:
|
||||
leaq bootstrap_stack_top(%rip), %rsp
|
||||
call kmainEntry
|
||||
1: hlt
|
||||
jmp 1b
|
||||
|
||||
# The kernel's initial stack (used until the scheduler hands each task its own).
|
||||
# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it
|
||||
# needs more than a token stack. Lives in .bss (zeroed, higher-half).
|
||||
.section .bss
|
||||
.balign 16
|
||||
bootstrap_stack:
|
||||
.skip 65536
|
||||
bootstrap_stack_top:
|
||||
.text
|
||||
|
||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
||||
# set with mov, so we far-return through the caller's own return address.
|
||||
.global gdt_flush
|
||||
gdt_flush:
|
||||
lgdt (%rdi)
|
||||
mov $0x10, %ax # kernel data selector
|
||||
mov %ax, %ds
|
||||
mov %ax, %es
|
||||
mov %ax, %ss
|
||||
mov %ax, %fs
|
||||
mov %ax, %gs
|
||||
pop %rax # caller's return address
|
||||
push $0x08 # kernel code selector (new CS)
|
||||
push %rax # return address (new RIP)
|
||||
lretq
|
||||
|
||||
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
||||
.global idt_flush
|
||||
idt_flush:
|
||||
lidt (%rdi)
|
||||
ret
|
||||
|
||||
# load_tr(di = TSS selector): load the task register.
|
||||
.global load_tr
|
||||
load_tr:
|
||||
ltr %di
|
||||
ret
|
||||
|
||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
||||
# Cooperative context switch: save the callee-saved registers on the current
|
||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
||||
# registers are the compiler's responsibility (this looks like a normal call).
|
||||
.global switch_context
|
||||
switch_context:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
||||
mov %rsi, %rsp # switch to the new task's stack
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret # return into the new task's saved instruction pointer
|
||||
|
||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
||||
# leaves its entry function in r15. A fresh task is switched to with the big kernel
|
||||
# lock held (the hand-off rule in sync.zig) but has no enter/leave frame of its own,
|
||||
# so it releases the lock here before running its body. r15 survives the call (it's
|
||||
# callee-saved). New tasks then start with interrupts enabled.
|
||||
.extern releaseForFreshTask
|
||||
.global task_trampoline
|
||||
task_trampoline:
|
||||
call releaseForFreshTask # drop the kernel lock we inherited across the switch
|
||||
sti
|
||||
call *%r15 # call the task entry (fn() void)
|
||||
1: hlt # if the entry returns, idle (still preemptible)
|
||||
jmp 1b
|
||||
|
||||
# user_task_trampoline: the first thing a freshly-spawned *user* task runs.
|
||||
# init_user_task_stack leaves the user entry in r15 and the user stack in r14
|
||||
# (both callee-saved, so they survive the lock-release call). Like task_trampoline
|
||||
# it drops the inherited kernel lock, then — instead of calling a kernel fn — it
|
||||
# builds an iretq frame and drops to ring 3. The scheduler's switchTo already
|
||||
# loaded this task's address space (CR3) and published its kernel stack
|
||||
# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls
|
||||
# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed).
|
||||
# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context
|
||||
# to ring 3, never returning. The scheduler calls this from a fresh user task's
|
||||
# trampoline (after the lock is released and the entry/stack read from the Task).
|
||||
# cli guards the swapgs..iretq window: an interrupt there would run in ring 0
|
||||
# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set)
|
||||
# re-enables interrupts on the drop to ring 3.
|
||||
.global jump_to_user
|
||||
jump_to_user:
|
||||
cli
|
||||
push $0x1B # user SS (0x18 | RPL 3)
|
||||
push %rsi # user RSP
|
||||
push $0x202 # RFLAGS: IF | reserved-1
|
||||
push $0x23 # user CS (0x20 | RPL 3)
|
||||
push %rdi # user RIP
|
||||
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||
iretq
|
||||
|
||||
# --- ring 3 entry/exit ------------------------------------------------------
|
||||
|
||||
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
||||
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
|
||||
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
|
||||
# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames
|
||||
# grow safely into it — then builds the 5-word iretq frame with the user
|
||||
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
|
||||
# timer keeps running in user mode.
|
||||
.global enter_user
|
||||
enter_user:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
|
||||
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
|
||||
mov %rsp, %gs:0 # kernel_rsp: the syscall stub stacks here too
|
||||
push $0x1B # user SS (0x18 | RPL 3)
|
||||
push %rsi # user RSP
|
||||
push $0x202 # RFLAGS: IF | reserved-1
|
||||
push $0x23 # user CS (0x20 | RPL 3)
|
||||
push %rdi # user RIP
|
||||
swapgs # user GS base for ring 3 (isr_common swaps back)
|
||||
iretq
|
||||
|
||||
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
|
||||
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
|
||||
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
|
||||
# and stay off — the Zig caller re-enables.
|
||||
.global user_exit_to_kernel
|
||||
user_exit_to_kernel:
|
||||
mov user_saved_rsp(%rip), %rsp
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret
|
||||
|
||||
# syscall_entry: the target of the SYSCALL instruction (LSTAR). The CPU does NOT
|
||||
# switch stacks — it puts the return RIP in RCX, the saved RFLAGS in R11, loads
|
||||
# CS/SS from STAR, masks RFLAGS with SFMASK (so IF is already clear), and jumps
|
||||
# here with RSP still the *user* stack. We swap in the kernel GS, switch to the
|
||||
# task's kernel stack via the per-CPU block, build a CpuState frame identical to
|
||||
# the interrupt path's, and reuse interruptDispatch (vector 128) — then SYSRET.
|
||||
#
|
||||
# Hazard (acceptable while init is the only, trusted, user program): SYSRETQ #GPs
|
||||
# in ring 0 if the return RIP (RCX) is non-canonical. A hostile user could arrange
|
||||
# that; hardening (canonical check / iretq fallback) is a later security-track item.
|
||||
.global syscall_entry
|
||||
syscall_entry:
|
||||
swapgs # kernel GS base
|
||||
movq %rsp, %gs:8 # stash user rsp in the scratch slot
|
||||
movq %gs:0, %rsp # switch to this task's kernel stack
|
||||
# Build the trap frame (same field order as isr_common), highest field first.
|
||||
pushq $0x1B # ss (user data | 3)
|
||||
pushq %gs:8 # rsp (user, from scratch)
|
||||
pushq %r11 # rflags (saved by syscall)
|
||||
pushq $0x23 # cs (user code | 3)
|
||||
pushq %rcx # rip (saved by syscall)
|
||||
pushq $0 # error_code (none for a syscall)
|
||||
pushq $128 # vector (same as the int 0x80 gate)
|
||||
push %rax
|
||||
push %rbx
|
||||
push %rcx
|
||||
push %rdx
|
||||
push %rsi
|
||||
push %rdi
|
||||
push %rbp
|
||||
push %r8
|
||||
push %r9
|
||||
push %r10
|
||||
push %r11
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # trap-frame pointer
|
||||
call interruptDispatch
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %r11
|
||||
pop %r10
|
||||
pop %r9
|
||||
pop %r8
|
||||
pop %rbp
|
||||
pop %rdi
|
||||
pop %rsi
|
||||
pop %rdx
|
||||
pop %rcx
|
||||
pop %rbx
|
||||
pop %rax
|
||||
add $16, %rsp # drop vector + error_code -> rsp at rip
|
||||
popq %rcx # rip -> RCX (SYSRETQ restores RIP from RCX)
|
||||
addq $8, %rsp # skip the cs slot (SYSRETQ loads CS from STAR)
|
||||
popq %r11 # rflags -> R11 (SYSRETQ restores RFLAGS from R11)
|
||||
popq %rsp # user rsp (the ss slot below is abandoned)
|
||||
swapgs # user GS base
|
||||
sysretq # -> ring 3: RIP=RCX, RFLAGS=R11, CS/SS from STAR
|
||||
|
||||
.section .bss
|
||||
.balign 8
|
||||
user_saved_rsp:
|
||||
.skip 8
|
||||
.text
|
||||
|
||||
# --- user-mode test program --------------------------------------------------
|
||||
# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and
|
||||
# entered via enter_user. Position-independent (immediates and short jumps only).
|
||||
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
|
||||
# from the user mapping. (The old hello/ping blob was retired once /sbin/init
|
||||
# became the real ring-3 exerciser; only the isolation proof remains.)
|
||||
.section .rodata
|
||||
|
||||
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
|
||||
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
|
||||
# page, so this must take a #PF with error code 0x5 (present | user) before any
|
||||
# access happens. movabs loads the full 64-bit higher-half address (a disp32
|
||||
# would sign-extend and miss).
|
||||
.global user_pf_start
|
||||
.global user_pf_end
|
||||
user_pf_start:
|
||||
movabs $0xFFFF8800FEE00000, %rcx
|
||||
mov (%rcx), %rax
|
||||
1: jmp 1b
|
||||
user_pf_end:
|
||||
|
||||
.text
|
||||
|
||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
||||
.macro STUB_NOERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $0
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
||||
.macro STUB_ERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
STUB_NOERR 0
|
||||
STUB_NOERR 1
|
||||
STUB_NOERR 2
|
||||
STUB_NOERR 3
|
||||
STUB_NOERR 4
|
||||
STUB_NOERR 5
|
||||
STUB_NOERR 6
|
||||
STUB_NOERR 7
|
||||
STUB_ERR 8
|
||||
STUB_NOERR 9
|
||||
STUB_ERR 10
|
||||
STUB_ERR 11
|
||||
STUB_ERR 12
|
||||
STUB_ERR 13
|
||||
STUB_ERR 14
|
||||
STUB_NOERR 15
|
||||
STUB_NOERR 16
|
||||
STUB_ERR 17
|
||||
STUB_NOERR 18
|
||||
STUB_NOERR 19
|
||||
STUB_NOERR 20
|
||||
STUB_ERR 21
|
||||
STUB_NOERR 22
|
||||
STUB_NOERR 23
|
||||
STUB_NOERR 24
|
||||
STUB_NOERR 25
|
||||
STUB_NOERR 26
|
||||
STUB_NOERR 27
|
||||
STUB_NOERR 28
|
||||
STUB_NOERR 29
|
||||
STUB_NOERR 30
|
||||
STUB_NOERR 31
|
||||
|
||||
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
||||
# code, so they all use the dummy-zero form.
|
||||
STUB_NOERR 32
|
||||
STUB_NOERR 33
|
||||
STUB_NOERR 34
|
||||
STUB_NOERR 35
|
||||
STUB_NOERR 36
|
||||
STUB_NOERR 37
|
||||
STUB_NOERR 38
|
||||
STUB_NOERR 39
|
||||
STUB_NOERR 40
|
||||
STUB_NOERR 41
|
||||
STUB_NOERR 42
|
||||
STUB_NOERR 43
|
||||
STUB_NOERR 44
|
||||
STUB_NOERR 45
|
||||
STUB_NOERR 46
|
||||
STUB_NOERR 47
|
||||
|
||||
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
|
||||
# vector; dispatched specially in interruptDispatch.
|
||||
STUB_NOERR 128
|
||||
|
||||
.extern interruptDispatch
|
||||
|
||||
# Shared tail. Register push order here defines the CpuState field order.
|
||||
# If the interrupt came from ring 3 the GS base holds the user's value, so swap
|
||||
# in the kernel's before anything reads per-CPU data (swapgs discipline; see
|
||||
# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24).
|
||||
isr_common:
|
||||
testb $3, 24(%rsp)
|
||||
jz 1f
|
||||
swapgs
|
||||
1: push %rax
|
||||
push %rbx
|
||||
push %rcx
|
||||
push %rdx
|
||||
push %rsi
|
||||
push %rdi
|
||||
push %rbp
|
||||
push %r8
|
||||
push %r9
|
||||
push %r10
|
||||
push %r11
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
||||
call interruptDispatch
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %r11
|
||||
pop %r10
|
||||
pop %r9
|
||||
pop %r8
|
||||
pop %rbp
|
||||
pop %rdi
|
||||
pop %rsi
|
||||
pop %rdx
|
||||
pop %rcx
|
||||
pop %rbx
|
||||
pop %rax
|
||||
add $16, %rsp # drop the vector and error code
|
||||
# Symmetric to entry: if returning to ring 3, restore the user GS base. CS is
|
||||
# now at offset 8 (RIP@0, CS@8).
|
||||
testb $3, 8(%rsp)
|
||||
jz 1f
|
||||
swapgs
|
||||
1: iretq
|
||||
@@ -0,0 +1,52 @@
|
||||
/* Kernel link layout — higher half.
|
||||
*
|
||||
* The kernel is linked to *run* in the higher half (virtual base
|
||||
* 0xFFFF_FFFF_8000_0000, matching danos.kernel_virt_base and build.zig's
|
||||
* image_base) but is *loaded* low. Each section's load address (LMA) is its
|
||||
* virtual address minus KERNEL_VIRT_BASE via AT(), so the ELF's p_paddr lands
|
||||
* at a low physical address (.text at 1 MiB) that the loader can allocate and
|
||||
* copy into. The loader maps p_vaddr (high) -> p_paddr (low) in its bootstrap
|
||||
* tables and jumps to the high entry; the kernel then builds its own tables
|
||||
* with the physmap and abandons the identity map. Requires LLD (build.zig pins
|
||||
* it) — the self-hosted linker ignores PHDRS/AT()/section order.
|
||||
*/
|
||||
|
||||
KERNEL_VIRT_BASE = 0xFFFFFFFF80000000;
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
||||
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
.text ALIGN(4K) : AT(ADDR(.text) - KERNEL_VIRT_BASE) {
|
||||
*(.text .text.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : AT(ADDR(.rodata) - KERNEL_VIRT_BASE) {
|
||||
*(.rodata .rodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : AT(ADDR(.data) - KERNEL_VIRT_BASE) {
|
||||
*(.data .data.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space. The loader zeroes it via the
|
||||
* gap between each PT_LOAD segment's file size and memory size, so no
|
||||
* boundary symbols are needed here. */
|
||||
.bss ALIGN(4K) : AT(ADDR(.bss) - KERNEL_VIRT_BASE) {
|
||||
*(.bss .bss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,389 @@
|
||||
//! The kernel's page tables and virtual memory manager.
|
||||
//!
|
||||
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
|
||||
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
|
||||
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
|
||||
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
|
||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||
//! which the kernel heap will build on.
|
||||
//!
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
// Page-table entry bits.
|
||||
const present: u64 = 1 << 0;
|
||||
const writable: u64 = 1 << 1;
|
||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||
const pwt: u64 = 1 << 3; // page write-through
|
||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// ELF segment flags (p_flags).
|
||||
const pf_x: u32 = 1;
|
||||
const pf_w: u32 = 2;
|
||||
|
||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
||||
var kernel_pml4: u64 = 0;
|
||||
var alloc_frame: *const fn () ?u64 = undefined;
|
||||
var free_frame: *const fn (u64) void = undefined; // for tearing down address spaces
|
||||
|
||||
/// Set once the kernel is running on its own tables (past the CR3 load in
|
||||
/// `init`). Before that, the kernel reaches page-table frames through the
|
||||
/// *loader's* bootstrap physmap, which only covers the low 4 GiB — so every
|
||||
/// frame allocated for a table during that window must be below 4 GiB. Both the
|
||||
/// frame allocator and this code scan from low addresses up, so it holds
|
||||
/// naturally; the assertion in `allocTable` makes a violation loud rather than
|
||||
/// a silent fault. After the switch the kernel's own physmap covers all RAM.
|
||||
var on_own_tables = false;
|
||||
|
||||
/// Set at the end of `init`. Guards against a new *higher-half* PML4 entry being
|
||||
/// created afterward: the kernel half is pre-populated at init and then shared
|
||||
/// by copying PML4[256..512) into every process address space (M3), so a late
|
||||
/// top-half entry would be invisible to already-created address spaces.
|
||||
var init_done = false;
|
||||
|
||||
const bootstrap_physmap_limit: u64 = 4 << 30;
|
||||
|
||||
/// Dereference a page-table frame by its physical address, via the physmap.
|
||||
/// This is the single hinge for the higher-half move: page tables hold physical
|
||||
/// frame addresses (pmm gives out physical frames, and CR3/PTEs must be
|
||||
/// physical), but the kernel reaches them at `physmap_base + physical`. Valid under
|
||||
/// both the loader's bootstrap tables and the kernel's own, which share the
|
||||
/// physmap base.
|
||||
fn tableAt(physical: u64) *[512]u64 {
|
||||
return @ptrFromInt(danos.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
fn allocTable() u64 {
|
||||
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
|
||||
if (!on_own_tables and frame >= bootstrap_physmap_limit)
|
||||
@panic("paging: table frame above the 4 GiB bootstrap physmap");
|
||||
@memset(tableAt(frame)[0..], 0);
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Return the table an entry points at, creating it if empty. Intermediate
|
||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||
/// writable only if every level is; non-executable if any level is).
|
||||
fn descend(entry: *u64) u64 {
|
||||
if (entry.* & present != 0) return entry.* & address_mask;
|
||||
const frame = allocTable();
|
||||
entry.* = frame | present | writable;
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Map one 4 KiB page `virtual` -> `physical` with `flags` (present is added).
|
||||
fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
// The kernel half is fixed after init: every top-half PML4 entry is
|
||||
// pre-created so address spaces can share it by copying these slots. A new
|
||||
// one here would be invisible to address spaces already made.
|
||||
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||
@panic("paging: new higher-half PML4 entry after init");
|
||||
const pdpt = descend(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte);
|
||||
const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF];
|
||||
const pt = descend(pde);
|
||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||
}
|
||||
|
||||
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||
/// window onto physical memory once the low identity map goes away.
|
||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
||||
var address = physical_base & ~@as(u64, page_size - 1);
|
||||
const end = physical_base + len;
|
||||
while (address < end) : (address += page_size) {
|
||||
mapPage(pml4, danos.physicalToVirtual(address), address, flags);
|
||||
}
|
||||
}
|
||||
|
||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(mm.regions)))[0..mm.len];
|
||||
}
|
||||
|
||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
||||
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
|
||||
fn enableNx() void {
|
||||
const efer_msr = 0xC0000080;
|
||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
|
||||
alloc_frame = allocFrame;
|
||||
free_frame = freeFrame;
|
||||
enableNx();
|
||||
const pml4 = allocTable();
|
||||
|
||||
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||
// mapping: the low half belongs to user space. MMIO is skipped here and
|
||||
// mapped on demand (mapMmio) or explicitly below.
|
||||
for (regions(boot_information.memory_map)) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||
}
|
||||
|
||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||
// the kernel touches directly), RW + NX.
|
||||
const fb = boot_information.framebuffer;
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapPage(pml4, danos.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||
// to their low physical load addresses with real ELF permissions: code
|
||||
// R+X, rodata R, data R+W+NX. This is the W^X guarantee.
|
||||
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||
var flags: u64 = present;
|
||||
if (seg.flags & pf_w != 0) flags |= writable;
|
||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
||||
var off: u64 = 0;
|
||||
while (off < seg.pages * page_size) : (off += page_size) {
|
||||
mapPage(pml4, seg.virtual + off, seg.physical + off, flags);
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Pre-create every higher-half PML4 entry (an empty PDPT where none
|
||||
// exists yet), so the whole kernel half is a fixed set of top-level
|
||||
// slots. A process address space (M3) then shares the kernel half simply
|
||||
// by copying PML4[256..512) — growth beneath these slots (heap, on-demand
|
||||
// MMIO) propagates to every address space because they share the PDPTs.
|
||||
for (256..512) |i| {
|
||||
const e = &tableAt(pml4)[i];
|
||||
if (e.* & present == 0) e.* = allocTable() | present | writable;
|
||||
}
|
||||
|
||||
kernel_pml4 = pml4;
|
||||
asm volatile ("mov %[pml4], %%cr3"
|
||||
:
|
||||
: [pml4] "r" (pml4),
|
||||
: .{ .memory = true }
|
||||
);
|
||||
on_own_tables = true; // now on the kernel's physmap (covers all RAM)
|
||||
init_done = true; // the kernel half is fixed from here
|
||||
}
|
||||
|
||||
/// The kernel's own top-level page table (physical). Every kernel task and every
|
||||
/// per-process address space shares this table's higher half.
|
||||
pub fn kernelPml4() u64 {
|
||||
return kernel_pml4;
|
||||
}
|
||||
|
||||
/// Load CR3 (switch the active address space). `pml4` is a physical frame.
|
||||
pub fn loadCr3(pml4: u64) void {
|
||||
asm volatile ("mov %[pml4], %%cr3"
|
||||
:
|
||||
: [pml4] "r" (pml4),
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
||||
pub fn map(virtual: u64, physical: u64, writable_page: bool) void {
|
||||
var flags: u64 = present | no_execute;
|
||||
if (writable_page) flags |= writable;
|
||||
mapPage(kernel_pml4, virtual, physical, flags);
|
||||
invalidate(virtual);
|
||||
}
|
||||
|
||||
/// Map a device MMIO range into the physmap and return the virtual address to
|
||||
/// use for it (physicalToVirtual(physical)). The single way the kernel (and the device
|
||||
/// layer, via the HAL) reaches memory-mapped registers once the identity map is
|
||||
/// gone: physmap pages are RW + NX, so a driver never executes device memory.
|
||||
/// Idempotent for already-mapped ranges. `len` 0 maps one page.
|
||||
pub fn mapMmio(physical: u64, len: u64, writable_page: bool) u64 {
|
||||
var flags: u64 = present | no_execute;
|
||||
if (writable_page) flags |= writable;
|
||||
const first = physical & ~@as(u64, page_size - 1);
|
||||
const last = physical + (if (len == 0) 1 else len) - 1;
|
||||
var address = first;
|
||||
while (address <= (last & ~@as(u64, page_size - 1))) : (address += page_size) {
|
||||
const virtual = danos.physicalToVirtual(address);
|
||||
mapPage(kernel_pml4, virtual, address, flags);
|
||||
invalidate(virtual);
|
||||
}
|
||||
return danos.physicalToVirtual(physical);
|
||||
}
|
||||
|
||||
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
|
||||
/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves
|
||||
/// existing entries untouched. Only used under user-exclusive virtual ranges, so
|
||||
/// no kernel mapping's protection is widened (the leaf still governs).
|
||||
fn descendUser(entry: *u64) u64 {
|
||||
const table = descend(entry);
|
||||
entry.* |= user;
|
||||
return table;
|
||||
}
|
||||
|
||||
/// Map one 4 KiB page `virtual` -> `physical` accessible from ring 3. W^X is the
|
||||
/// caller's contract: code pages are read-only + executable, data pages are
|
||||
/// writable + no-execute. `virtual` must lie in a user-exclusive region (see
|
||||
/// `descendUser`).
|
||||
pub fn mapUser(virtual: u64, physical: u64, writable_page: bool, executable: bool) void {
|
||||
mapUserInto(kernel_pml4, virtual, physical, writable_page, executable);
|
||||
}
|
||||
|
||||
/// Map a ring-3-accessible page into the address space rooted at `pml4` (which
|
||||
/// may be a process's own table or the kernel's). W^X is the caller's contract.
|
||||
pub fn mapUserInto(pml4: u64, virtual: u64, physical: u64, writable_page: bool, executable: bool) void {
|
||||
var flags: u64 = present | user;
|
||||
if (writable_page) flags |= writable;
|
||||
if (!executable) flags |= no_execute;
|
||||
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
const pdpt = descendUser(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = descendUser(pdpte);
|
||||
const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF];
|
||||
const pt = descendUser(pde);
|
||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags;
|
||||
invalidate(virtual);
|
||||
}
|
||||
|
||||
/// Map a device MMIO window `[physical, physical+len)` into the user (low) half of the
|
||||
/// address space rooted at `pml4`, page by page. Unlike `mapUserInto` these pages
|
||||
/// are **strong-uncacheable** (PCD|PWT — device registers must not be cached) and
|
||||
/// carry the `device_grant` bit so teardown does not return the MMIO frames to the
|
||||
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
|
||||
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
|
||||
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
|
||||
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
|
||||
const first = physical & ~@as(u64, page_size - 1);
|
||||
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||
var off: u64 = 0;
|
||||
while (first + off <= last) : (off += page_size) {
|
||||
const v = virtual + off;
|
||||
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
|
||||
const pdpt = descendUser(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
|
||||
const pd = descendUser(pdpte);
|
||||
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
|
||||
const pt = descendUser(pde);
|
||||
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
|
||||
invalidate(v);
|
||||
}
|
||||
}
|
||||
|
||||
/// Create a new address space: a fresh PML4 with an empty user half and the
|
||||
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
||||
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
||||
/// the kernel half propagates to every address space). Returns the physical
|
||||
/// PML4, or null if out of frames.
|
||||
pub fn createAddressSpace() ?u64 {
|
||||
const pml4 = alloc_frame() orelse return null;
|
||||
const t = tableAt(pml4);
|
||||
@memset(t[0..256], 0); // empty user half
|
||||
@memcpy(t[256..512], tableAt(kernel_pml4)[256..512]); // shared kernel half
|
||||
return pml4;
|
||||
}
|
||||
|
||||
/// Tear down an address space created by `createAddressSpace`: free every frame
|
||||
/// and table in the user half [0..256), then the PML4 itself. The shared kernel
|
||||
/// half [256..512) is never touched. The caller must not be running on `pml4`.
|
||||
pub fn destroyAddressSpace(pml4: u64) void {
|
||||
const t = tableAt(pml4);
|
||||
for (0..256) |i| {
|
||||
if (t[i] & present != 0) freeSubtree(t[i] & address_mask, 3); // PDPT level
|
||||
}
|
||||
free_frame(pml4);
|
||||
}
|
||||
|
||||
/// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At
|
||||
/// level 1 the entries are leaf data frames; above, they are child tables.
|
||||
fn freeSubtree(physical: u64, level: u32) void {
|
||||
const t = tableAt(physical);
|
||||
for (t) |e| {
|
||||
if (e & present == 0) continue;
|
||||
if (level > 1) {
|
||||
freeSubtree(e & address_mask, level - 1);
|
||||
} else if (e & device_grant == 0) {
|
||||
// A device-grant leaf points at MMIO, not RAM — returning it to the
|
||||
// frame allocator would corrupt the pool. Only reclaim real RAM.
|
||||
free_frame(e & address_mask);
|
||||
}
|
||||
}
|
||||
free_frame(physical); // page-table frames are always real RAM
|
||||
}
|
||||
|
||||
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
||||
pub fn isExecutable(virtual: u64) bool {
|
||||
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return false;
|
||||
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return false;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return false;
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return false;
|
||||
return pte & no_execute == 0;
|
||||
}
|
||||
|
||||
/// Make an already-identity-mapped RAM page **executable** (clear its NX bit),
|
||||
/// leaving it present and writable. The blanket RAM mapping is NX for W^X, but the
|
||||
/// application processors fetch the AP trampoline from a low RAM page under paging —
|
||||
/// so that one page must be executable. A deliberate, temporary W^X exception for a
|
||||
/// single bring-up page; the caller frees it once every AP is up.
|
||||
pub fn setExecutable(physical: u64) void {
|
||||
mapPage(kernel_pml4, physical, physical, present | writable); // note: no no_execute
|
||||
invalidate(physical);
|
||||
}
|
||||
|
||||
/// Remove a mapping and flush it from the TLB.
|
||||
pub fn unmap(virtual: u64) void {
|
||||
unmapInto(kernel_pml4, virtual);
|
||||
}
|
||||
|
||||
/// Remove a mapping from the address space rooted at `pml4` (a process's own
|
||||
/// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE —
|
||||
/// the intermediate tables and any frame the PTE pointed at are left to the
|
||||
/// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables).
|
||||
pub fn unmapInto(pml4: u64, virtual: u64) void {
|
||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return;
|
||||
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return;
|
||||
tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF] = 0;
|
||||
invalidate(virtual);
|
||||
}
|
||||
|
||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
||||
/// needs the frame behind a user vaddr to free it).
|
||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return null;
|
||||
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return null;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return null;
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return null;
|
||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||
}
|
||||
|
||||
fn invalidate(virtual: u64) void {
|
||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||
// inline asm won't form directly, so stage the address in a register first.
|
||||
asm volatile (
|
||||
\\mov %[v], %%rax
|
||||
\\invlpg (%%rax)
|
||||
:
|
||||
: [v] "r" (virtual),
|
||||
: .{ .rax = true, .memory = true }
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
//! Per-CPU data reached through the GS segment base. The GS base holds a pointer
|
||||
//! to this core's `ArchitecturePerCpu`, so kernel code gets the running core's block with
|
||||
//! a single MSR read (`scheduler()`) and the system_call entry stub gets its kernel stack
|
||||
//! with a `%gs`-relative load (no usable stack yet at that point).
|
||||
//!
|
||||
//! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds
|
||||
//! the user's own GS (which ring 3 may set freely), and this pointer lives in the
|
||||
//! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the
|
||||
//! system_call stub and the conditional swapgs in isr_common) brings it back, and
|
||||
//! every ring-0 -> ring-3 exit swaps it away. Because the very first ring
|
||||
//! transition is always an exit (the kernel starts in ring 0), the swap pairs
|
||||
//! keep the invariant without seeding KERNEL_GS_BASE. `scheduler()` is therefore
|
||||
//! valid in any ring-0 context and never sees a user-controlled base.
|
||||
|
||||
const std = @import("std");
|
||||
const io = @import("io.zig");
|
||||
const parameters = @import("parameters");
|
||||
|
||||
const ia32_gs_base = 0xC000_0101;
|
||||
|
||||
/// Layout is load-bearing: the system_call entry stub in isr.s reaches `kernel_rsp`
|
||||
/// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below
|
||||
/// pin the offsets.
|
||||
pub const ArchitecturePerCpu = extern struct {
|
||||
kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for system_call entry (== TSS.rsp0)
|
||||
scratch: u64 = 0, // %gs:8 — stashes the user rsp during system_call entry
|
||||
scheduler: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns)
|
||||
};
|
||||
|
||||
comptime {
|
||||
std.debug.assert(@offsetOf(ArchitecturePerCpu, "kernel_rsp") == 0);
|
||||
std.debug.assert(@offsetOf(ArchitecturePerCpu, "scratch") == 8);
|
||||
}
|
||||
|
||||
var blocks = [_]ArchitecturePerCpu{.{}} ** parameters.maximum_cpus;
|
||||
|
||||
/// Publish core `index`'s per-CPU block: record the scheduler pointer and point
|
||||
/// the GS base at the block. Called once per core during bring-up, after the GDT
|
||||
/// is loaded (a GS *selector* reload would clobber the base).
|
||||
pub fn setLocal(index: usize, scheduler_ptr: usize) void {
|
||||
blocks[index].scheduler = scheduler_ptr;
|
||||
io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index]));
|
||||
}
|
||||
|
||||
/// The scheduler pointer for the running core (via the GS base). Valid in any
|
||||
/// ring-0 context under the swapgs discipline.
|
||||
pub fn scheduler() usize {
|
||||
return @as(*const ArchitecturePerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).scheduler;
|
||||
}
|
||||
|
||||
/// Record core `index`'s kernel stack top, used by the system_call entry stub to
|
||||
/// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it
|
||||
/// switches to a user task.
|
||||
pub fn setKernelRsp(index: usize, top: usize) void {
|
||||
blocks[index].kernel_rsp = top;
|
||||
}
|
||||
|
||||
// Fast-system_call MSRs.
|
||||
const ia32_efer = 0xC000_0080;
|
||||
const ia32_star = 0xC000_0081;
|
||||
const ia32_lstar = 0xC000_0082;
|
||||
const ia32_sfmask = 0xC000_0084;
|
||||
|
||||
/// Enable the `system_call`/`sysret` fast path on this core (BSP and each AP). EFER.SCE
|
||||
/// turns the instructions on; STAR sets the selectors system_call/sysret load; LSTAR
|
||||
/// is the entry stub (isr.s); SFMASK clears RFLAGS bits on entry (notably IF —
|
||||
/// the handler runs with interrupts off, like the int-gate path). The GDT is laid
|
||||
/// out (kernel code 0x08, then user data 0x18 / code 0x20) precisely so these line
|
||||
/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8
|
||||
/// with RPL forced to 3, so base 0x10 gives CS 0x23 (user code|3) and SS 0x1B.
|
||||
pub fn initSystemCall() void {
|
||||
io.wrmsr(ia32_efer, io.rdmsr(ia32_efer) | 1); // SCE
|
||||
io.wrmsr(ia32_star, (@as(u64, 0x08) << 32) | (@as(u64, 0x10) << 48));
|
||||
const entry = @extern(*const anyopaque, .{ .name = "syscall_entry" });
|
||||
io.wrmsr(ia32_lstar, @intFromPtr(entry));
|
||||
io.wrmsr(ia32_sfmask, 0x4_0700); // clear IF, TF, DF, AC on entry
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
//! Serial console (16550-compatible UART) — the kernel's machine-readable output
|
||||
//! channel. Unlike the framebuffer console, serial text can be captured to a file
|
||||
//! by QEMU (`-serial file:...`), which is what the test harness asserts on.
|
||||
//!
|
||||
//! The UART defaults to the legacy PC COM1 at I/O port `0x3F8`, but a UEFI Class 3
|
||||
//! (legacy-free) machine may have no COM1 — or its debug UART somewhere else, and
|
||||
//! reachable via MMIO rather than port I/O. So the location is a runtime value:
|
||||
//! `reconfigure` repoints it once ACPI's SPCR table has been read. Early boot logs
|
||||
//! optimistically to COM1 (harmless if absent); the framebuffer console is the
|
||||
//! always-present log.
|
||||
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// How the UART registers are reached: legacy I/O ports or memory-mapped.
|
||||
const Access = enum { port, mmio };
|
||||
|
||||
var access: Access = .port;
|
||||
var base: u64 = 0x3F8; // COM1
|
||||
|
||||
fn portOut(p: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[p]"
|
||||
:
|
||||
: [value] "{al}" (value),
|
||||
[p] "{dx}" (p),
|
||||
);
|
||||
}
|
||||
|
||||
fn portIn(p: u16) u8 {
|
||||
return asm volatile ("inb %[p], %[value]"
|
||||
: [value] "={al}" (-> u8),
|
||||
: [p] "{dx}" (p),
|
||||
);
|
||||
}
|
||||
|
||||
/// Read UART register `off` through the active access method.
|
||||
fn register(off: u64) u8 {
|
||||
if (access == .mmio) return @as(*volatile u8, @ptrFromInt(base + off)).*;
|
||||
return portIn(@intCast(base + off));
|
||||
}
|
||||
|
||||
/// Write UART register `off` through the active access method.
|
||||
fn setRegister(off: u64, value: u8) void {
|
||||
if (access == .mmio) {
|
||||
@as(*volatile u8, @ptrFromInt(base + off)).* = value;
|
||||
} else {
|
||||
portOut(@intCast(base + off), value);
|
||||
}
|
||||
}
|
||||
|
||||
/// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything
|
||||
/// else; it has no dependencies, and is a harmless no-op if the port is absent.
|
||||
pub fn init() void {
|
||||
setRegister(1, 0x00); // disable interrupts
|
||||
setRegister(3, 0x80); // enable DLAB (set baud divisor)
|
||||
setRegister(0, 0x03); // divisor low: 38400 baud
|
||||
setRegister(1, 0x00); // divisor high
|
||||
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||
setRegister(4, 0x0B); // RTS/DSR set
|
||||
}
|
||||
|
||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||
/// re-run the UART setup there. Called after discovery when an SPCR entry exists.
|
||||
pub fn reconfigure(is_mmio: bool, address: u64) void {
|
||||
access = if (is_mmio) .mmio else .port;
|
||||
// An MMIO UART is reached through the physmap; an I/O-port UART keeps its
|
||||
// port number unchanged.
|
||||
base = if (is_mmio) paging.mapMmio(address, 0x100, true) else address;
|
||||
init();
|
||||
}
|
||||
|
||||
fn writeByte(c: u8) void {
|
||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
||||
var guard: u32 = 0;
|
||||
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
||||
setRegister(0, c);
|
||||
}
|
||||
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
for (bytes) |c| {
|
||||
if (c == '\n') writeByte('\r');
|
||||
writeByte(c);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
//! Application-processor (AP) bring-up: waking the cores the firmware left parked.
|
||||
//!
|
||||
//! The firmware starts only the bootstrap processor (BSP); the others sit idle until
|
||||
//! the kernel wakes them with an INIT–SIPI–SIPI sequence (Intel SDM Vol.3, "MP
|
||||
//! Initialization"). A woken core begins in 16-bit real mode at a low physical page,
|
||||
//! runs the [trampoline](trampoline.s) up into 64-bit long mode, and lands in
|
||||
//! `apEntry` here. This module copies the trampoline into place, patches its
|
||||
//! per-AP parameters, drives the wake IPIs, and waits for each core to report in.
|
||||
//!
|
||||
//! Cores are brought up **one at a time**: a single trampoline page and parameter
|
||||
//! block are reused, so the BSP patches, wakes, and waits for one AP before the
|
||||
//! next. That also lets `apEntry` pick up its dense CPU index from a plain global.
|
||||
//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic
|
||||
//! scheduler entry and joins the run loop — mechanism here, policy there.
|
||||
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const apic = @import("apic.zig");
|
||||
const paging = @import("paging.zig");
|
||||
const pcpu = @import("per-cpu.zig");
|
||||
|
||||
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
|
||||
/// path doesn't depend on cpu.zig and risk an import cycle).
|
||||
const ia32_gs_base = 0xC000_0101;
|
||||
const page_size = 0x1000;
|
||||
|
||||
/// Physical address of the low (<1 MiB) frame reserved for the trampoline. Held for
|
||||
/// the life of the system so any core can be (re)woken on demand — a retry, or a
|
||||
/// future power manager bringing a core back online. The frame is kept **inert**
|
||||
/// between wakes (zeroed and non-executable) and only armed for the brief moment a
|
||||
/// core is actually climbing. Its low 20 bits are zero, so `physical >> 12` is the SIPI
|
||||
/// vector.
|
||||
var tramp_physical: u64 = 0;
|
||||
|
||||
/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own
|
||||
/// bring-up. The BSP clears it before each wake and polls it afterwards — a simple
|
||||
/// one-at-a-time handshake (only one AP is being started at any moment).
|
||||
var ap_alive: u32 = 0;
|
||||
|
||||
/// The dense CPU index of the AP currently being started. Set by the BSP before the
|
||||
/// wake, read by `apEntry` (safe because bring-up is strictly one core at a time).
|
||||
var boot_index: usize = 0;
|
||||
|
||||
/// The generic scheduler entry a woken core jumps to once its architecture state is up. Set
|
||||
/// by the kernel via `setSecondaryEntry`; never returns.
|
||||
var secondary_entry: ?*const fn () callconv(.c) noreturn = null;
|
||||
|
||||
/// Register the generic entry an AP calls once its per-CPU tables/LAPIC/timer are up.
|
||||
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
||||
secondary_entry = entry;
|
||||
}
|
||||
|
||||
/// Test hook: force the next `n` wake attempts to fail (skipping the actual
|
||||
/// INIT-SIPI-SIPI), so the retry path can be exercised deterministically. Zero in
|
||||
/// normal operation — the smp-retry test arms it via `architecture.testFailNextWakes`.
|
||||
var fail_next_wakes: u32 = 0;
|
||||
pub fn testFailNextWakes(n: u32) void {
|
||||
fail_next_wakes = n;
|
||||
}
|
||||
|
||||
/// Record the reserved low frame the trampoline uses. Call once at boot. The frame
|
||||
/// starts inert (identity-mapped RW+NX like all RAM); each wake arms it and disarms
|
||||
/// it again, so it's only ever executable while a core is climbing.
|
||||
pub fn setTrampolinePage(physical: u64) void {
|
||||
tramp_physical = physical;
|
||||
}
|
||||
|
||||
/// The reserved trampoline frame (0 if SMP bring-up never ran). Exposed so a test
|
||||
/// can verify it's inert — zeroed and non-executable — when dormant.
|
||||
pub fn trampolinePage() u64 {
|
||||
return tramp_physical;
|
||||
}
|
||||
|
||||
/// Arm the trampoline for a wake: make its page executable (W^X exception for the
|
||||
/// duration of the climb) and copy the blob in.
|
||||
fn arm() void {
|
||||
// The AP executes this page at its physical address (identity) while it
|
||||
// climbs from real to long mode, so it needs a low identity mapping that is
|
||||
// executable — the one deliberate, transient W^X exception. The BSP writes
|
||||
// the blob into the frame through the physmap.
|
||||
paging.setExecutable(tramp_physical);
|
||||
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
|
||||
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
|
||||
const len = @intFromPtr(end) - @intFromPtr(start);
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
|
||||
@memcpy(destination[0..len], start[0..len]);
|
||||
}
|
||||
|
||||
/// Disarm after a wake: wipe the page through the physmap and remove its low
|
||||
/// identity mapping, so no executable code (nor any stale bytes, nor any
|
||||
/// low-half mapping) lingers between wakes. Safe once the woken core has
|
||||
/// reported in — it's long past the trampoline by then, in the kernel image; a
|
||||
/// core that never answered is dead and can't be mid-climb.
|
||||
fn disarm() void {
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
|
||||
@memset(destination[0..page_size], 0);
|
||||
paging.unmap(tramp_physical); // drop the transient low identity mapping
|
||||
}
|
||||
|
||||
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
|
||||
/// base plus the field's offset within it (a same-section symbol difference). The
|
||||
/// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86
|
||||
/// tolerates unaligned stores, so we don't force layout constraints on the asm.
|
||||
fn param(comptime name: []const u8) *align(1) volatile u64 {
|
||||
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
|
||||
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
|
||||
return @ptrFromInt(danos.physicalToVirtual(tramp_physical + (sym - start)));
|
||||
}
|
||||
|
||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
|
||||
/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive. This
|
||||
/// is one self-contained attempt: it arms the trampoline, drives INIT–SIPI–SIPI, and
|
||||
/// disarms again before returning — so it's safe to call repeatedly (a retry, or a
|
||||
/// power manager re-waking a core; the INIT resets a core that was wedged). Returns
|
||||
/// false if the core doesn't report in within the timeout (left parked, no harm to
|
||||
/// the running system). `cr3` is the kernel page tables the AP adopts. Precondition:
|
||||
/// `setTrampolinePage` has run.
|
||||
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool {
|
||||
// The trampoline loads CR3 with a 32-bit `movl` before it reaches long mode,
|
||||
// so the page-table root must be addressable in 32 bits.
|
||||
if (cr3 >= (1 << 32)) @panic("smp: kernel page tables above 4 GiB");
|
||||
arm();
|
||||
defer disarm();
|
||||
|
||||
if (fail_next_wakes > 0) { // test hook: simulate a core missing this attempt
|
||||
fail_next_wakes -= 1;
|
||||
return false;
|
||||
}
|
||||
|
||||
boot_index = index;
|
||||
param("ap_tramp_cr3").* = cr3;
|
||||
param("ap_tramp_stack").* = stack_top;
|
||||
param("ap_tramp_entry").* = @intFromPtr(&apEntry);
|
||||
param("ap_tramp_percpu").* = percpu;
|
||||
|
||||
@atomicStore(u32, &ap_alive, 0, .seq_cst);
|
||||
|
||||
const vector: u8 = @intCast(tramp_physical >> 12);
|
||||
apic.sendInit(apic_id);
|
||||
delayMicros(10_000); // 10 ms INIT settle
|
||||
apic.sendStartup(apic_id, vector);
|
||||
delayMicros(200);
|
||||
apic.sendStartup(apic_id, vector);
|
||||
|
||||
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
|
||||
const deadline = apic.millis() + 100;
|
||||
while (apic.millis() < deadline) {
|
||||
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
|
||||
asm volatile ("pause");
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Busy-wait `us` microseconds against the calibrated TSC clock (the AP wake happens
|
||||
/// after the timer is up, so the clock is available).
|
||||
fn delayMicros(us: u64) void {
|
||||
const start = apic.micros();
|
||||
while (apic.micros() - start < us) asm volatile ("pause");
|
||||
}
|
||||
|
||||
/// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU
|
||||
/// pointer in RDI. Brings up this core's own descriptor tables, LAPIC and timer,
|
||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
const cpu = boot_index;
|
||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||
idt.loadOnThisCpu(); // the shared IDT
|
||||
pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload
|
||||
pcpu.initSystemCall(); // enable system_call/sysret on this core
|
||||
|
||||
apic.initSecondary(); // software-enable this core's LAPIC
|
||||
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
|
||||
|
||||
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
||||
|
||||
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
||||
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
# AP trampoline: brings a waking application processor from the 16-bit real mode it
|
||||
# starts in (after INIT-SIPI-SIPI) up through protected mode into 64-bit long mode,
|
||||
# then jumps to the Zig AP entry (arch/x86_64/smp.zig:apEntry).
|
||||
#
|
||||
# A STARTUP IPI vectors a core to physical address `vector << 12` in real mode, so
|
||||
# this blob is copied to a low (<1 MiB) page and started there; at entry CS = that
|
||||
# page >> 4 and IP = 0. It is fully **position-independent**: it derives its own
|
||||
# linear base (CS << 4) into EBX and addresses every internal datum as
|
||||
# `(label - ap_trampoline_start)(%ebx)` — a difference of two symbols in the same
|
||||
# section, which the assembler folds to a constant page offset no matter where the
|
||||
# blob was linked or copied to. The BSP patches the parameter block (CR3, stack,
|
||||
# entry, per-CPU pointer) before each wake; see arch/x86_64/smp.zig.
|
||||
#
|
||||
# It lives in .rodata (not .text): it is data to be copied out and executed
|
||||
# elsewhere, never run at its link address, so it must not be a normal code segment.
|
||||
|
||||
.section .rodata
|
||||
.balign 16
|
||||
.code16
|
||||
.global ap_trampoline_start
|
||||
ap_trampoline_start:
|
||||
cli
|
||||
cld
|
||||
|
||||
# Linear base of this page (CS << 4) into EBX; all data is addressed off it.
|
||||
xorl %eax, %eax
|
||||
mov %cs, %ax
|
||||
shll $4, %eax
|
||||
movl %eax, %ebx
|
||||
|
||||
mov %cs, %ax # DS = CS, so we address our data as DS:(label - start):
|
||||
mov %ax, %ds # the segment base (CS<<4) already supplies the page base,
|
||||
# so data operands use the page *offset*, not EBX.
|
||||
|
||||
# Relocate the pointers whose absolute (linear) targets depend on where we were
|
||||
# copied: the GDT base and the two far-jump targets = EBX + their page offsets.
|
||||
# EBX supplies the base for the *value* (via leal); the store address is DS-rel.
|
||||
leal (gdt32 - ap_trampoline_start)(%ebx), %eax
|
||||
movl %eax, gdtr32_base - ap_trampoline_start
|
||||
leal (prot_entry - ap_trampoline_start)(%ebx), %eax
|
||||
movl %eax, jmp32_off - ap_trampoline_start
|
||||
leal (long_entry - ap_trampoline_start)(%ebx), %eax
|
||||
movl %eax, jmp64_off - ap_trampoline_start
|
||||
|
||||
lgdtl gdtr32 - ap_trampoline_start
|
||||
|
||||
movl %cr0, %eax # enter protected mode (CR0.PE)
|
||||
orl $1, %eax
|
||||
movl %eax, %cr0
|
||||
|
||||
ljmpl *(jmp32_ptr - ap_trampoline_start) # -> prot_entry, CS = 0x08
|
||||
|
||||
.code32
|
||||
prot_entry:
|
||||
movw $0x10, %ax # flat 32-bit data segments
|
||||
movw %ax, %ds
|
||||
movw %ax, %es
|
||||
movw %ax, %ss
|
||||
movw %ax, %fs
|
||||
movw %ax, %gs
|
||||
|
||||
# CR4: PAE (required for long mode) + OSFXSR/OSXMMEXCPT. The kernel is built with
|
||||
# SSE (part of the x86_64 baseline), and the compiler emits SSE for things as
|
||||
# ordinary as a struct copy — without OSFXSR those instructions #UD. The BSP got
|
||||
# these bits from UEFI; an AP starts fresh, so we must set them ourselves.
|
||||
movl %cr4, %eax
|
||||
orl $((1 << 5) | (1 << 9) | (1 << 10)), %eax
|
||||
movl %eax, %cr4
|
||||
|
||||
# CR0: clear EM (no x87 emulation) and set MP, so SSE/x87 don't fault.
|
||||
movl %cr0, %eax
|
||||
andl $~(1 << 2), %eax # ~EM
|
||||
orl $(1 << 1), %eax # MP
|
||||
movl %eax, %cr0
|
||||
|
||||
movl (param_cr3 - ap_trampoline_start)(%ebx), %eax # kernel page tables
|
||||
movl %eax, %cr3
|
||||
|
||||
movl $0xC0000080, %ecx # EFER: long mode enable (LME) + NX enable (NXE, since
|
||||
rdmsr # the kernel's PTEs set the NX bit)
|
||||
orl $((1 << 8) | (1 << 11)), %eax
|
||||
wrmsr
|
||||
|
||||
movl %cr0, %eax # paging on (CR0.PG) — now in long mode (compat sub-mode)
|
||||
orl $(1 << 31), %eax
|
||||
movl %eax, %cr0
|
||||
|
||||
ljmpl *(jmp64_ptr - ap_trampoline_start)(%ebx) # -> long_entry, CS = 0x18 (L=1)
|
||||
|
||||
.code64
|
||||
long_entry:
|
||||
movw $0x10, %ax # sane flat data segments
|
||||
movw %ax, %ds
|
||||
movw %ax, %es
|
||||
movw %ax, %ss
|
||||
|
||||
# RBX = EBX (zero-extended) = page base. Load our stack and per-CPU pointer, then
|
||||
# call the Zig entry — which runs from the kernel image and never returns.
|
||||
movq (param_stack - ap_trampoline_start)(%rbx), %rsp
|
||||
movq (param_percpu - ap_trampoline_start)(%rbx), %rdi # SysV arg 0
|
||||
movq (param_entry - ap_trampoline_start)(%rbx), %rax
|
||||
callq *%rax
|
||||
1: hlt # unreachable; guard against a stray return
|
||||
jmp 1b
|
||||
|
||||
# --- data: GDT, far pointers, and the BSP-patched parameter block -----------
|
||||
.balign 8
|
||||
gdt32:
|
||||
.quad 0x0000000000000000 # 0x00 null
|
||||
.quad 0x00CF9A000000FFFF # 0x08 32-bit code (G, D, present, exec/read)
|
||||
.quad 0x00CF92000000FFFF # 0x10 data (valid in 32- and 64-bit)
|
||||
.quad 0x00AF9A000000FFFF # 0x18 64-bit code (L=1)
|
||||
gdt32_end:
|
||||
|
||||
gdtr32:
|
||||
.word gdt32_end - gdt32 - 1
|
||||
gdtr32_base:
|
||||
.long 0 # patched (16-bit code): linear base of gdt32
|
||||
|
||||
jmp32_ptr: # indirect far-jump operand: offset then selector
|
||||
jmp32_off:
|
||||
.long 0 # patched: linear address of prot_entry
|
||||
.word 0x08 # 32-bit code selector
|
||||
|
||||
jmp64_ptr:
|
||||
jmp64_off:
|
||||
.long 0 # patched: linear address of long_entry
|
||||
.word 0x18 # 64-bit code selector
|
||||
|
||||
# The parameter block, filled in by the BSP (smp.zig) before each STARTUP IPI. Global
|
||||
# so the Zig side can locate each field as (symbol - ap_trampoline_start).
|
||||
.global ap_tramp_cr3
|
||||
.global ap_tramp_stack
|
||||
.global ap_tramp_entry
|
||||
.global ap_tramp_percpu
|
||||
param_cr3:
|
||||
ap_tramp_cr3:
|
||||
.quad 0 # kernel PML4 physical address (CR3)
|
||||
param_stack:
|
||||
ap_tramp_stack:
|
||||
.quad 0 # top of this AP's kernel stack
|
||||
param_entry:
|
||||
ap_tramp_entry:
|
||||
.quad 0 # address of apEntry (the Zig AP entry)
|
||||
param_percpu:
|
||||
ap_tramp_percpu:
|
||||
.quad 0 # this AP's per-CPU pointer (goes in GS base)
|
||||
|
||||
.global ap_trampoline_end
|
||||
ap_trampoline_end:
|
||||
@@ -0,0 +1,87 @@
|
||||
//! Task State Segment and its interrupt stacks. In long mode the TSS has two
|
||||
//! jobs. First, the Interrupt Stack Table: an IDT gate can name an IST entry,
|
||||
//! and the CPU switches to that stack when the exception fires — no matter how
|
||||
//! broken the interrupted stack was. We use IST1 for the double-fault handler,
|
||||
//! so a fault that happens *because* the current stack is unusable still lands
|
||||
//! on solid ground instead of triple-faulting. Second, rsp0: the kernel stack
|
||||
//! the CPU switches to when an interrupt arrives from ring 3 (published by the
|
||||
//! user-mode entry path via `rsp0Ptr`).
|
||||
//!
|
||||
//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at
|
||||
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
|
||||
//! indexed by CPU number; slot 0 is the BSP.
|
||||
|
||||
const parameters = @import("parameters");
|
||||
const gdt = @import("gdt.zig");
|
||||
|
||||
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
||||
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
|
||||
const Tss = packed struct {
|
||||
reserved0: u32 = 0,
|
||||
rsp0: u64 = 0,
|
||||
rsp1: u64 = 0,
|
||||
rsp2: u64 = 0,
|
||||
reserved1: u64 = 0,
|
||||
ist1: u64 = 0,
|
||||
ist2: u64 = 0,
|
||||
ist3: u64 = 0,
|
||||
ist4: u64 = 0,
|
||||
ist5: u64 = 0,
|
||||
ist6: u64 = 0,
|
||||
ist7: u64 = 0,
|
||||
reserved2: u64 = 0,
|
||||
reserved3: u16 = 0,
|
||||
iomap_base: u16 = 0,
|
||||
};
|
||||
|
||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
||||
pub const double_fault_ist = 1;
|
||||
|
||||
const maximum_cpus = parameters.maximum_cpus;
|
||||
pub const ist_stack_size = parameters.ist_stack_size;
|
||||
|
||||
/// One TSS per core (small — kept static). The IST stacks are 16 KiB each, so only
|
||||
/// the **BSP's** is static: it must exist before the frame allocator does, to catch a
|
||||
/// fault during early boot. Each **AP** gets a heap-allocated IST stack at bring-up
|
||||
/// (after the heap is up), the top of which the BSP records here before waking it —
|
||||
/// so we reserve big stacks only for cores that actually come online.
|
||||
var tss_table = [_]Tss{.{}} ** maximum_cpus;
|
||||
var bsp_ist_stack: [ist_stack_size]u8 align(16) = undefined;
|
||||
var ap_ist_top = [_]usize{0} ** maximum_cpus; // per-AP IST stack top (0 = BSP / not set)
|
||||
|
||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
||||
|
||||
/// Address of core `cpu`'s rsp0 slot — the kernel stack the CPU switches to on a
|
||||
/// ring-3 -> ring-0 interrupt. Computed as base + 4 (rsp0's architectural offset,
|
||||
/// which is why the pointer is only 4-aligned) rather than `&t.rsp0`, which on a
|
||||
/// packed struct would be an unaligned bit-pointer type. The ring-3 entry path
|
||||
/// (enter_user in isr.s) writes the current kernel stack pointer through this
|
||||
/// before dropping to user mode.
|
||||
pub fn rsp0Ptr(cpu: usize) *align(4) u64 {
|
||||
return @ptrFromInt(@intFromPtr(&tss_table[cpu]) + 4);
|
||||
}
|
||||
|
||||
/// Record the top of the IST stack the kernel allocated for AP `cpu`. Called on the
|
||||
/// BSP before waking that core; read by the core's own `setupThisCpu`.
|
||||
pub fn setApIstStack(cpu: usize, top: usize) void {
|
||||
ap_ist_top[cpu] = top;
|
||||
}
|
||||
|
||||
/// Set up core `cpu`'s TSS: point IST1 at its stack (the BSP's static one for core 0,
|
||||
/// the allocated one recorded via `setApIstStack` for an AP), install the TSS
|
||||
/// descriptor into that core's GDT, and load it into the task register. Requires the
|
||||
/// core's GDT to already be loaded (gdt.loadOnThisCpu first).
|
||||
pub fn setupThisCpu(cpu: usize) void {
|
||||
const t = &tss_table[cpu];
|
||||
t.* = .{};
|
||||
t.ist1 = if (cpu == 0) @intFromPtr(&bsp_ist_stack) + ist_stack_size else ap_ist_top[cpu];
|
||||
t.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
||||
gdt.setTssFor(cpu, @intFromPtr(t), @sizeOf(Tss) - 1);
|
||||
load_tr(gdt.tss_selector);
|
||||
}
|
||||
|
||||
/// Set up the bootstrap processor's TSS (slot 0). Requires gdt.init first.
|
||||
pub fn init() void {
|
||||
setupThisCpu(0);
|
||||
}
|
||||
@@ -0,0 +1,158 @@
|
||||
//! A framebuffer text console: draws glyphs from an embedded PSF2 font directly
|
||||
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
||||
//! — just pixels.
|
||||
//!
|
||||
//! This is a **bootstrap** console — a stop-gap so early boot has something on
|
||||
//! screen. The framebuffer is a general graphics surface, *not* inherently a text
|
||||
//! terminal; once the driver machinery exists it becomes a proper graphics device
|
||||
//! driver and this text-grid crutch goes away. It is therefore kept **separate
|
||||
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file,
|
||||
//! while this only paints the handful of user-facing status lines and panics.
|
||||
//!
|
||||
//! The module owns a single console and a `present` flag; `write` is a no-op when
|
||||
//! the firmware handed over no framebuffer (a headless machine), so the kernel
|
||||
//! never assumes a display exists.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
|
||||
/// The one framebuffer console, valid only when `con_present`.
|
||||
var con: Console = undefined;
|
||||
var con_present: bool = false;
|
||||
|
||||
/// Set up the console over `fb`, or mark it absent if there's no usable
|
||||
/// framebuffer. Clears the screen when present.
|
||||
pub fn init(fb: danos.Framebuffer) void {
|
||||
if (!fb.present()) {
|
||||
con_present = false;
|
||||
return;
|
||||
}
|
||||
con = Console.init(fb);
|
||||
if (con.cols == 0 or con.rows == 0) {
|
||||
con_present = false;
|
||||
return;
|
||||
}
|
||||
con.clear();
|
||||
con_present = true;
|
||||
}
|
||||
|
||||
/// Whether an on-screen console is available.
|
||||
pub fn present() bool {
|
||||
return con_present;
|
||||
}
|
||||
|
||||
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present,
|
||||
/// so it's always safe to call.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
if (!con_present) return;
|
||||
for (bytes) |c| con.putChar(c);
|
||||
}
|
||||
|
||||
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
|
||||
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
|
||||
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
|
||||
const font = @embedFile("font.psf");
|
||||
const glyph_w = 8;
|
||||
const glyph_h = 16;
|
||||
const glyph_bytes = glyph_h; // 8 pixels wide => 1 byte per row
|
||||
const glyph_data = 32; // PSF2 header size
|
||||
|
||||
pub const Console = struct {
|
||||
fb: danos.Framebuffer,
|
||||
cols: u32,
|
||||
rows: u32,
|
||||
col: u32 = 0,
|
||||
row: u32 = 0,
|
||||
fg: u32 = 0x00c8_c8c8, // light grey
|
||||
bg: u32 = 0x0000_0000, // black
|
||||
|
||||
pub fn init(fb: danos.Framebuffer) Console {
|
||||
// Reach the framebuffer through the physmap, so the pointer stays valid
|
||||
// once the low identity map is gone. The base is mapped by both the
|
||||
// loader's bootstrap tables and paging.init.
|
||||
var mapped = fb;
|
||||
if (fb.base != 0) mapped.base = danos.physicalToVirtual(fb.base);
|
||||
return .{
|
||||
.fb = mapped,
|
||||
.cols = fb.width / glyph_w,
|
||||
.rows = fb.height / glyph_h,
|
||||
};
|
||||
}
|
||||
|
||||
/// Fill the whole screen with the background colour and home the cursor.
|
||||
pub fn clear(self: *Console) void {
|
||||
var y: u32 = 0;
|
||||
while (y < self.fb.height) : (y += 1) self.fillRow(y, self.bg);
|
||||
self.col = 0;
|
||||
self.row = 0;
|
||||
}
|
||||
|
||||
pub fn putChar(self: *Console, ch: u8) void {
|
||||
switch (ch) {
|
||||
'\n' => self.newline(),
|
||||
'\r' => self.col = 0,
|
||||
else => {
|
||||
if (self.col >= self.cols) self.newline();
|
||||
self.drawGlyph(ch, self.col * glyph_w, self.row * glyph_h);
|
||||
self.col += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn newline(self: *Console) void {
|
||||
self.col = 0;
|
||||
if (self.row + 1 >= self.rows) {
|
||||
self.scroll();
|
||||
} else {
|
||||
self.row += 1;
|
||||
}
|
||||
}
|
||||
|
||||
fn drawGlyph(self: *Console, ch: u8, px: u32, py: u32) void {
|
||||
const rows = font[glyph_data + @as(usize, ch) * glyph_bytes ..][0..glyph_bytes];
|
||||
var gy: u32 = 0;
|
||||
while (gy < glyph_h) : (gy += 1) {
|
||||
const bits = rows[gy];
|
||||
var gx: u32 = 0;
|
||||
while (gx < glyph_w) : (gx += 1) {
|
||||
// Leftmost pixel is the high bit.
|
||||
const on = (bits >> @as(u3, @intCast(7 - gx))) & 1 != 0;
|
||||
self.pixel(px + gx, py + gy, if (on) self.fg else self.bg);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Shift the visible text up one glyph row and clear the freed bottom row,
|
||||
/// leaving the cursor on that now-blank last line.
|
||||
fn scroll(self: *Console) void {
|
||||
const visible = self.rows * glyph_h;
|
||||
var y: u32 = 0;
|
||||
while (y + glyph_h < visible) : (y += 1) self.copyRow(y, y + glyph_h);
|
||||
while (y < visible) : (y += 1) self.fillRow(y, self.bg);
|
||||
self.row = self.rows - 1;
|
||||
}
|
||||
|
||||
inline fn rowPtr(self: *Console, y: u32) [*]volatile u32 {
|
||||
const base: [*]volatile u8 = @ptrFromInt(self.fb.base);
|
||||
return @ptrCast(@alignCast(base + y * self.fb.pitch));
|
||||
}
|
||||
|
||||
inline fn pixel(self: *Console, x: u32, y: u32, color: u32) void {
|
||||
self.rowPtr(y)[x] = color;
|
||||
}
|
||||
|
||||
fn fillRow(self: *Console, y: u32, color: u32) void {
|
||||
const row = self.rowPtr(y);
|
||||
var x: u32 = 0;
|
||||
while (x < self.fb.width) : (x += 1) row[x] = color;
|
||||
}
|
||||
|
||||
fn copyRow(self: *Console, destination_y: u32, source_y: u32) void {
|
||||
const destination = self.rowPtr(destination_y);
|
||||
const source = self.rowPtr(source_y);
|
||||
var x: u32 = 0;
|
||||
while (x < self.fb.width) : (x += 1) destination[x] = source[x];
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
//! Device service: the kernel side of user-space driver access. At boot it
|
||||
//! flattens the discovered device tree (src/device) into a stable, id-indexed
|
||||
//! snapshot and a per-device claim table. User drivers enumerate the snapshot,
|
||||
//! claim the device they own, and map its MMIO — the claim is the capability that
|
||||
//! gates `mmio_map`/`irq_bind`, so a process can only ever touch hardware the
|
||||
//! firmware-neutral device tree says it owns.
|
||||
//!
|
||||
//! The table is a **tree**: each entry carries its parent's id. Firmware discovery
|
||||
//! seeds it, and a **bus driver** grows it — a process that has claimed a bus can
|
||||
//! `register` children below it as it enumerates them (USB devices behind a hub, PCI
|
||||
//! functions behind a bridge, comparators inside a timer block).
|
||||
//!
|
||||
//! Registration is where the capability model earns its keep. A `DeviceDescriptor` is, in
|
||||
//! effect, a licence to map physical memory: whoever claims it may `mmio_map` its
|
||||
//! `.memory` resources and `irq_bind` its `.irq` resources. If a bus driver could
|
||||
//! invent arbitrary resources, it would invent one covering the kernel's RAM, claim
|
||||
//! it, and map it. So `register` enforces **containment**: every resource of a child
|
||||
//! must lie inside a resource of the same kind on its parent. A bus driver can only
|
||||
//! ever subdivide what it was already given.
|
||||
|
||||
const std = @import("std");
|
||||
const platform = @import("platform");
|
||||
const danos = @import("danos");
|
||||
|
||||
const maximum_devices = 64;
|
||||
|
||||
/// Cap on children a single parent may have. A zero-resource child (legal — a USB
|
||||
/// device is addressed through its controller, not by MMIO) sidesteps the containment
|
||||
/// check, so without a bound a process that claimed one device could loop
|
||||
/// `device_register` and exhaust the whole table, permanently denying it to every other
|
||||
/// driver. This bounds the blast radius of one claim; a real quota (and a
|
||||
/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md.
|
||||
const maximum_children_per_parent = 16;
|
||||
|
||||
var devices: [maximum_devices]danos.DeviceDescriptor = undefined;
|
||||
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
||||
var count: usize = 0;
|
||||
|
||||
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
||||
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
||||
/// which would otherwise be an entirely silent failure. Logged at boot.
|
||||
pub var dropped: usize = 0;
|
||||
|
||||
/// Snapshot the device tree into the flat table. Run once, right after discovery.
|
||||
pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||
count = 0;
|
||||
dropped = 0;
|
||||
for (&claimed) |*c| c.* = null;
|
||||
walk(device_tree.root, danos.no_parent);
|
||||
}
|
||||
|
||||
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
||||
/// assigned it down to its children as their parent.
|
||||
fn walk(node: *platform.Device, parent_id: u64) void {
|
||||
const id = if (node.class == .root) danos.no_parent else record(node, parent_id);
|
||||
var child = node.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) walk(c, id);
|
||||
}
|
||||
|
||||
fn record(node: *platform.Device, parent_id: u64) u64 {
|
||||
if (count >= maximum_devices) {
|
||||
dropped += 1;
|
||||
return danos.no_parent; // children of a dropped node become roots, not orphans
|
||||
}
|
||||
var d = std.mem.zeroes(danos.DeviceDescriptor);
|
||||
d.id = count;
|
||||
d.parent = parent_id;
|
||||
d.class = @intFromEnum(node.class);
|
||||
const h = node.hid();
|
||||
d.hid_len = @min(h.len, d.hid.len);
|
||||
@memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]);
|
||||
const rc = @min(node.resource_count, danos.maximum_device_resources);
|
||||
d.resource_count = rc;
|
||||
for (0..rc) |i| {
|
||||
const r = node.resources[i];
|
||||
d.resources[i] = .{ .kind = @intFromEnum(r.kind), .start = r.start, .len = r.len };
|
||||
}
|
||||
devices[count] = d;
|
||||
count += 1;
|
||||
return d.id;
|
||||
}
|
||||
|
||||
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
||||
/// available (which may exceed `out.len`).
|
||||
pub fn enumerate(out: []danos.DeviceDescriptor) usize {
|
||||
const n = @min(count, out.len);
|
||||
@memcpy(out[0..n], devices[0..n]);
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
||||
/// out of range or already claimed.
|
||||
pub fn claim(id: u64, owner: u32) bool {
|
||||
if (id >= count) return false;
|
||||
if (claimed[@intCast(id)] != null) return false;
|
||||
claimed[@intCast(id)] = owner;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The task that owns device `id`, or null.
|
||||
pub fn ownerOf(id: u64) ?u32 {
|
||||
if (id >= count) return null;
|
||||
return claimed[@intCast(id)];
|
||||
}
|
||||
|
||||
/// Resource `index` of device `id`, or null if out of range.
|
||||
pub fn resourceOf(id: u64, index: u64) ?danos.ResourceDescriptor {
|
||||
if (id >= count) return null;
|
||||
const d = &devices[@intCast(id)];
|
||||
if (index >= d.resource_count) return null;
|
||||
return d.resources[@intCast(index)];
|
||||
}
|
||||
|
||||
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
|
||||
/// interval containment; for an irq it's equality, since an interrupt line is not
|
||||
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
||||
/// and would otherwise vacuously "fit" anywhere.
|
||||
fn contains(parent: danos.ResourceDescriptor, child: danos.ResourceDescriptor) bool {
|
||||
if (parent.kind != child.kind) return false;
|
||||
if (child.kind == @intFromEnum(danos.ResourceKind.irq)) return parent.start == child.start;
|
||||
if (child.len == 0 or parent.len == 0) return false;
|
||||
// No overflow: a resource that wraps the address space is not containable.
|
||||
const child_end = std.math.add(u64, child.start, child.len) catch return false;
|
||||
const parent_end = std.math.add(u64, parent.start, parent.len) catch return false;
|
||||
return child.start >= parent.start and child_end <= parent_end;
|
||||
}
|
||||
|
||||
pub const RegisterError = error{
|
||||
NoSpace, // the device table is full
|
||||
BadParent, // no such device, or not claimed by this task
|
||||
TooManyResources,
|
||||
TooManyChildren, // this parent is at maximum_children_per_parent
|
||||
NotContained, // a child resource escapes its parent's window
|
||||
};
|
||||
|
||||
/// Number of devices currently recorded with `parent_id` as their parent.
|
||||
fn childCount(parent_id: u64) usize {
|
||||
var n: usize = 0;
|
||||
for (devices[0..count]) |d| {
|
||||
if (d.parent == parent_id) n += 1;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new
|
||||
/// device id. The child is left **unclaimed**, so another process (a class driver)
|
||||
/// can claim it — that is how a bus hands a device to its driver.
|
||||
///
|
||||
/// `owner` must have claimed `parent_id`, and every resource in `descriptor` must be
|
||||
/// contained in a parent resource of the same kind. A device with no resources is
|
||||
/// fine and common: a USB device is addressed through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, owner: u32, descriptor: *const danos.DeviceDescriptor) RegisterError!u64 {
|
||||
const parent_owner = ownerOf(parent_id) orelse return error.BadParent;
|
||||
if (parent_owner != owner) return error.BadParent;
|
||||
if (descriptor.resource_count > danos.maximum_device_resources) return error.TooManyResources;
|
||||
if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;
|
||||
if (count >= maximum_devices) return error.NoSpace;
|
||||
|
||||
const parent = &devices[@intCast(parent_id)];
|
||||
for (0..@intCast(descriptor.resource_count)) |i| {
|
||||
const r = descriptor.resources[i];
|
||||
var ok = false;
|
||||
for (0..@intCast(parent.resource_count)) |j| {
|
||||
if (contains(parent.resources[j], r)) ok = true;
|
||||
}
|
||||
if (!ok) return error.NotContained;
|
||||
}
|
||||
|
||||
var d = std.mem.zeroes(danos.DeviceDescriptor);
|
||||
d.id = count;
|
||||
d.parent = parent_id;
|
||||
d.class = descriptor.class;
|
||||
d.hid_len = @min(descriptor.hid_len, d.hid.len);
|
||||
@memcpy(d.hid[0..@intCast(d.hid_len)], descriptor.hid[0..@intCast(d.hid_len)]);
|
||||
d.resource_count = descriptor.resource_count;
|
||||
for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i];
|
||||
|
||||
devices[count] = d;
|
||||
count += 1;
|
||||
return d.id;
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,174 @@
|
||||
//! The kernel heap: dynamic allocation for the kernel.
|
||||
//!
|
||||
//! Where the frame allocator ([pmm]) hands out fixed 4 KiB physical frames, the
|
||||
//! heap hands out arbitrary byte-sized blocks from a virtual region, growing on
|
||||
//! demand by mapping fresh frames into it (architecture.mapPage) — the first real user of
|
||||
//! the VMM (see docs/paging.md).
|
||||
//!
|
||||
//! The algorithm is a first-fit free list: an address-ordered singly linked list
|
||||
//! of free blocks, split on allocation and coalesced with neighbours on free. It
|
||||
//! is exposed as a std.mem.Allocator, so the kernel can use std containers.
|
||||
//!
|
||||
//! Not yet concurrency-safe: it assumes a single caller and no allocation from
|
||||
//! interrupt handlers (ours don't). A lock comes with threads/SMP.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const architecture = @import("architecture");
|
||||
const pmm = @import("pmm.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
/// Virtual base of the heap: the start of the higher half, which is unmapped and
|
||||
/// well clear of the identity-mapped low half. (Canonical on x86_64; an architecture that
|
||||
/// splits the address space differently would choose its own.)
|
||||
const heap_base: usize = 0xFFFF_8000_0000_0000;
|
||||
/// Cap on heap growth for now.
|
||||
const heap_maximum: usize = 64 * 1024 * 1024;
|
||||
|
||||
/// A block header, placed at the start of every block. While the block is free it
|
||||
/// also links into the free list via `next`.
|
||||
const Block = extern struct {
|
||||
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||
next: ?*Block, // free-list link (only meaningful while free)
|
||||
};
|
||||
|
||||
const header_size = @sizeOf(Block); // 16
|
||||
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||
|
||||
var free_list: ?*Block = null;
|
||||
var heap_end: usize = heap_base; // [heap_base, heap_end) is currently mapped
|
||||
|
||||
fn alignUp(value: usize, alignment: usize) usize {
|
||||
return (value + alignment - 1) & ~(alignment - 1);
|
||||
}
|
||||
|
||||
fn payloadOf(block: *Block) [*]u8 {
|
||||
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||
}
|
||||
|
||||
/// Bring the heap up with an initial mapped region.
|
||||
pub fn init() void {
|
||||
free_list = null;
|
||||
heap_end = heap_base;
|
||||
_ = grow(page_size);
|
||||
}
|
||||
|
||||
/// Map more pages onto the end of the heap and add them as a free block. Returns
|
||||
/// false if out of heap virtual space or out of physical frames.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const start = heap_end;
|
||||
const bytes = alignUp(minimum_bytes, page_size);
|
||||
if (start + bytes > heap_base + heap_maximum) return false;
|
||||
|
||||
var virtual = start;
|
||||
while (virtual < start + bytes) : (virtual += page_size) {
|
||||
const frame = pmm.alloc() orelse return false;
|
||||
architecture.mapPage(virtual, frame, true);
|
||||
}
|
||||
heap_end = start + bytes;
|
||||
|
||||
const block: *Block = @ptrFromInt(start);
|
||||
block.size = bytes;
|
||||
insertFree(block); // coalesces with the previous tail block if adjacent
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Insert a block into the address-ordered free list, coalescing with the
|
||||
/// physically adjacent free blocks on either side.
|
||||
fn insertFree(block: *Block) void {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |c| : (current = c.next) {
|
||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||
previous = c;
|
||||
}
|
||||
|
||||
block.next = current;
|
||||
if (previous) |p| p.next = block else free_list = block;
|
||||
|
||||
// Merge forward into `current` if they're contiguous.
|
||||
if (current) |c| {
|
||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||
block.size += c.size;
|
||||
block.next = c.next;
|
||||
}
|
||||
}
|
||||
// Merge `previous` forward into `block` if they're contiguous.
|
||||
if (previous) |p| {
|
||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||
p.size += block.size;
|
||||
p.next = block.next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < 2) : (attempts += 1) {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |block| : ({
|
||||
previous = block;
|
||||
current = block.next;
|
||||
}) {
|
||||
if (block.size < need) continue;
|
||||
|
||||
if (block.size >= need + minimum_block) {
|
||||
// Split: carve `need` off the front, leave the rest free.
|
||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||
rest.size = block.size - need;
|
||||
rest.next = block.next;
|
||||
if (previous) |p| p.next = rest else free_list = rest;
|
||||
block.size = need;
|
||||
} else {
|
||||
// Take the whole block.
|
||||
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||
}
|
||||
return payloadOf(block);
|
||||
}
|
||||
|
||||
// Nothing fit: grow and try once more.
|
||||
if (!grow(need)) return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
// --- std.mem.Allocator interface -----------------------------------------
|
||||
|
||||
pub fn allocator() std.mem.Allocator {
|
||||
return .{ .ptr = undefined, .vtable = &vtable };
|
||||
}
|
||||
|
||||
const vtable = std.mem.Allocator.VTable{
|
||||
.alloc = allocImpl,
|
||||
.resize = resizeImpl,
|
||||
.remap = remapImpl,
|
||||
.free = freeImpl,
|
||||
};
|
||||
|
||||
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||
// Blocks are 16-byte aligned; larger alignments aren't supported yet.
|
||||
if (alignment.toByteUnits() > 16) return null;
|
||||
return rawAlloc(len);
|
||||
}
|
||||
|
||||
fn resizeImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) bool {
|
||||
return false; // no in-place resize; the caller reallocates
|
||||
}
|
||||
|
||||
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||
rawFree(memory.ptr);
|
||||
}
|
||||
@@ -0,0 +1,314 @@
|
||||
//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a
|
||||
//! rendezvous point; a client `call`s it (send a message, block for a reply) and
|
||||
//! a server `replyWait`s on it (reply to the last client, then block for the next
|
||||
//! request). This is the substrate the user-space VFS server and device drivers
|
||||
//! are reached through — `open`/`read`/`write` become user-space wrappers that
|
||||
//! marshal a request into a `call`.
|
||||
//!
|
||||
//! Design (see docs/syscall.md, the plan):
|
||||
//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame
|
||||
//! through the physmap (`copyAcross`), which is mapped in every address space's
|
||||
//! shared kernel half — so the kernel reads/writes either process's user memory
|
||||
//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing.
|
||||
//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply
|
||||
//! to exactly one client at a time; that caller is held in `Task.ipc_client`.
|
||||
//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without
|
||||
//! becoming runnable*, which a WaitQueue can't express, so callers queue on the
|
||||
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
||||
//! waiting for work use a normal WaitQueue.
|
||||
//!
|
||||
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
|
||||
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
|
||||
//! copy is a later security-track item, matching the existing debug_write gap.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const architecture = @import("architecture");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const Task = scheduler.Task;
|
||||
|
||||
/// Largest message a single call/reply may carry. Bumping it is trivial; kept
|
||||
/// small because the copy runs under the big kernel lock.
|
||||
pub const MESSAGE_MAXIMUM: usize = 256;
|
||||
|
||||
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
||||
pub const maximum_services = 8;
|
||||
|
||||
/// Errno-style failures, returned as `-value` in the system_call result register.
|
||||
pub const EBADF: i64 = 1; // bad handle
|
||||
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
|
||||
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
||||
pub const ENOENT: i64 = 4; // no such registered service
|
||||
pub const ENOSPC: i64 = 5; // handle table or registry full
|
||||
pub const ENOMEM: i64 = 6; // out of memory
|
||||
|
||||
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
|
||||
/// message from a client — there is no reply owed. The low bits carry the source
|
||||
/// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in system/kernel/irq.zig;
|
||||
/// the message path uses a plain task-id badge with this bit clear. Defined in the
|
||||
/// shared contract (system/danos.zig), because ring 3 has to test the same bit.
|
||||
pub const notify_badge_bit: u64 = danos.notify_badge_bit;
|
||||
|
||||
/// End of the user (low) canonical half — user buffers must lie below it.
|
||||
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
|
||||
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||
pub const Endpoint = struct {
|
||||
refcount: u32 = 1,
|
||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||
sender_head: ?*Task = null,
|
||||
sender_tail: ?*Task = null,
|
||||
// Servers blocked in `replyWait` awaiting a request.
|
||||
receive_wait_queue: scheduler.WaitQueue = .{},
|
||||
// Pending asynchronous notifications (badges), a small coalescing ring.
|
||||
notify_buffer: [8]u64 = undefined,
|
||||
notify_head: u8 = 0,
|
||||
notify_tail: u8 = 0,
|
||||
};
|
||||
|
||||
pub fn createEndpoint() ?*Endpoint {
|
||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||
endpoint.* = .{};
|
||||
return endpoint;
|
||||
}
|
||||
|
||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||
pub fn dropRef(endpoint: *Endpoint) void {
|
||||
if (endpoint.refcount > 1) {
|
||||
endpoint.refcount -= 1;
|
||||
} else {
|
||||
heap.allocator().destroy(endpoint);
|
||||
}
|
||||
}
|
||||
|
||||
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
||||
|
||||
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
||||
t.next = null;
|
||||
if (endpoint.sender_tail) |tail| tail.next = t else endpoint.sender_head = t;
|
||||
endpoint.sender_tail = t;
|
||||
}
|
||||
|
||||
fn dequeueSender(endpoint: *Endpoint) ?*Task {
|
||||
const t = endpoint.sender_head orelse return null;
|
||||
endpoint.sender_head = t.next;
|
||||
if (endpoint.sender_head == null) endpoint.sender_tail = null;
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
// --- cross-address-space copy ----------------------------------------------
|
||||
|
||||
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
|
||||
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
|
||||
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
||||
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
|
||||
/// unmapped or out of range. Handles page-straddling buffers.
|
||||
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
|
||||
const source_root = if (source_as != 0) source_as else architecture.kernelPageTable();
|
||||
const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable();
|
||||
if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false;
|
||||
if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false;
|
||||
|
||||
var off: usize = 0;
|
||||
while (off < len) {
|
||||
const s = architecture.translate(source_root, source_va + off) orelse return false;
|
||||
const d = architecture.translate(destination_root, destination_va + off) orelse return false;
|
||||
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
||||
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
||||
const n = @min(@min(s_left, d_left), len - off);
|
||||
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(d));
|
||||
@memcpy(destination[0..n], source[0..n]);
|
||||
off += n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel
|
||||
/// buffer `destination`, walking the user page tables through the physmap. Returns false if
|
||||
/// the range escapes the user half or any source page is unmapped — so a bad user
|
||||
/// pointer *fails the system_call* rather than faulting the kernel (danos has no
|
||||
/// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt
|
||||
/// the machine). The correct way to pull a fixed-size struct in from user space, and
|
||||
/// a single fetch: no TOCTOU against a hostile pointer.
|
||||
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||
if (user_as == 0) return false; // not a user address space
|
||||
if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false;
|
||||
var off: usize = 0;
|
||||
while (off < destination.len) {
|
||||
const s = architecture.translate(user_as, user_va + off) orelse return false;
|
||||
const s_left = page_size - ((user_va + off) & (page_size - 1));
|
||||
const n = @min(s_left, destination.len - off);
|
||||
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
|
||||
@memcpy(destination[off..][0..n], source[0..n]);
|
||||
off += n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- the two IPC operations -------------------------------------------------
|
||||
|
||||
/// Client side of IPC_Call: send `[message_ptr, message_len)` to `endpoint` and block until a
|
||||
/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a
|
||||
/// negative errno. Runs as the current task.
|
||||
pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64) i64 {
|
||||
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const me = scheduler.current();
|
||||
me.ipc_send_ptr = message_ptr;
|
||||
me.ipc_send_len = message_len;
|
||||
me.ipc_reply_ptr = reply_ptr;
|
||||
me.ipc_reply_cap = reply_cap;
|
||||
me.ipc_status = 0;
|
||||
|
||||
enqueueSender(endpoint, me); // join the FIFO, then...
|
||||
scheduler.wakeLocked(&endpoint.receive_wait_queue); // ...wake a waiting server (no-op if none)
|
||||
scheduler.blockCurrentLocked(); // block until the reply readies us again
|
||||
|
||||
return me.ipc_status; // reply length or -errno, written by the replier
|
||||
}
|
||||
|
||||
/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client
|
||||
/// we currently owe (if any), then receive the next request into
|
||||
/// `[receive_ptr, receive_cap)`, blocking until one arrives. Writes the sender's badge
|
||||
/// to `out_badge` and returns the request length, or a negative errno. A pending
|
||||
/// notification is delivered ahead of client requests (length 0, badge with
|
||||
/// `notify_badge_bit` set, no reply owed).
|
||||
pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, out_badge: *u64) i64 {
|
||||
if (reply_len > MESSAGE_MAXIMUM or receive_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const me = scheduler.current();
|
||||
|
||||
// (1) Reply to the client we're still holding, if any.
|
||||
if (me.ipc_client) |client| {
|
||||
me.ipc_client = null;
|
||||
const n = @min(reply_len, client.ipc_reply_cap);
|
||||
if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
|
||||
client.ipc_status = @intCast(n);
|
||||
} else {
|
||||
client.ipc_status = -EFAULT;
|
||||
}
|
||||
scheduler.readyLocked(client); // its `call` now returns
|
||||
}
|
||||
|
||||
// (2) Receive the next request (or notification), blocking until one is ready.
|
||||
while (true) {
|
||||
if (popNotify(endpoint)) |badge| {
|
||||
out_badge.* = badge | notify_badge_bit;
|
||||
return 0; // notification: no payload, no reply owed
|
||||
}
|
||||
if (dequeueSender(endpoint)) |caller| {
|
||||
const n = @min(caller.ipc_send_len, receive_cap);
|
||||
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) {
|
||||
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
||||
scheduler.readyLocked(caller);
|
||||
continue;
|
||||
}
|
||||
me.ipc_client = caller; // remember who to reply to
|
||||
out_badge.* = caller.id;
|
||||
return @intCast(n);
|
||||
}
|
||||
scheduler.waitLocked(&endpoint.receive_wait_queue); // nothing yet — sleep until woken, then retry
|
||||
}
|
||||
}
|
||||
|
||||
// --- asynchronous notification (for IRQ-as-message, M10) --------------------
|
||||
|
||||
fn popNotify(endpoint: *Endpoint) ?u64 {
|
||||
if (endpoint.notify_head == endpoint.notify_tail) return null;
|
||||
const badge = endpoint.notify_buffer[endpoint.notify_head % endpoint.notify_buffer.len];
|
||||
endpoint.notify_head +%= 1;
|
||||
return badge;
|
||||
}
|
||||
|
||||
/// Post an asynchronous notification carrying `badge` to `endpoint` and wake a waiting
|
||||
/// receiver. Precondition: the big kernel lock is held.
|
||||
///
|
||||
/// The lock must already cover whatever produced `endpoint` — an ISR that looked the
|
||||
/// endpoint up in a table and *then* took the lock could be racing a process exit
|
||||
/// that unbinds and frees it in between. See irq.dispatch, which holds one lock
|
||||
/// region across the table read and this call.
|
||||
///
|
||||
/// A full ring drops the notification. That is the correct semantics, not a
|
||||
/// concession: a notification is a *level* ("this device wants attention"), and the
|
||||
/// driver re-reads device state on wake. It is never a count of events.
|
||||
pub fn notifyLocked(endpoint: *Endpoint, badge: u64) void {
|
||||
if (endpoint.notify_tail -% endpoint.notify_head < endpoint.notify_buffer.len) {
|
||||
endpoint.notify_buffer[endpoint.notify_tail % endpoint.notify_buffer.len] = badge;
|
||||
endpoint.notify_tail +%= 1;
|
||||
}
|
||||
scheduler.wakeLocked(&endpoint.receive_wait_queue);
|
||||
}
|
||||
|
||||
/// `notifyLocked` as a self-contained ISR critical section, for a caller that holds
|
||||
/// `endpoint` by some means other than a table the lock protects. Releases the lock without
|
||||
/// touching the interrupt flag (the ISR's iretq restores it), like the timer tick.
|
||||
pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
|
||||
_ = sync.enter();
|
||||
notifyLocked(endpoint, badge);
|
||||
sync.leaveIsr();
|
||||
}
|
||||
|
||||
// --- per-process handle table + name registry -------------------------------
|
||||
|
||||
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
|
||||
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
||||
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||
for (&t.handles, 0..) |*slot, i| {
|
||||
if (slot.* == null) {
|
||||
slot.* = @ptrCast(endpoint);
|
||||
return @intCast(i);
|
||||
}
|
||||
}
|
||||
return -ENOSPC;
|
||||
}
|
||||
|
||||
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
if (h >= t.handles.len) return null;
|
||||
const slot = t.handles[@intCast(h)] orelse return null;
|
||||
return @ptrCast(@alignCast(slot));
|
||||
}
|
||||
|
||||
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
||||
/// exit path so a dead server's endpoints don't linger referenced.
|
||||
pub fn closeHandles(t: *Task) void {
|
||||
for (&t.handles) |*slot| {
|
||||
if (slot.*) |p| {
|
||||
dropRef(@ptrCast(@alignCast(p)));
|
||||
slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
||||
|
||||
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||
pub fn register(id: u32, endpoint: *Endpoint) i64 {
|
||||
if (id >= maximum_services) return -ENOENT;
|
||||
if (registry[id]) |old| dropRef(old);
|
||||
endpoint.refcount += 1;
|
||||
registry[id] = endpoint;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, taking a reference for the caller to
|
||||
/// install in its handle table. Null if nothing is registered there.
|
||||
pub fn lookup(id: u32) ?*Endpoint {
|
||||
if (id >= maximum_services) return null;
|
||||
const endpoint = registry[id] orelse return null;
|
||||
endpoint.refcount += 1;
|
||||
return endpoint;
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
//! Inter-process communication: message-passing channels.
|
||||
//!
|
||||
//! IPC is the backbone of a microkernel ([vision](../docs/vision.md)): once
|
||||
//! drivers and services live in separate address spaces, a message is how they
|
||||
//! talk. This first form is a **bounded blocking channel** — a ring buffer of
|
||||
//! messages with a producer/consumer rendezvous, built on the scheduler's
|
||||
//! [wait queues](scheduling.md). `send` blocks when the channel is full, `receive`
|
||||
//! blocks when it's empty; neither busy-waits.
|
||||
//!
|
||||
//! For now both endpoints are kernel threads sharing the kernel address space.
|
||||
//! When user mode arrives, the same primitive carries messages across the
|
||||
//! isolation boundary (with the payload copied between address spaces).
|
||||
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
|
||||
/// A bounded blocking channel of `capacity` messages of type `T`.
|
||||
pub fn Channel(comptime T: type, comptime capacity: usize) type {
|
||||
return struct {
|
||||
const Self = @This();
|
||||
|
||||
buffer: [capacity]T = undefined,
|
||||
head: usize = 0, // next slot to read
|
||||
tail: usize = 0, // next slot to write
|
||||
count: usize = 0,
|
||||
not_full: scheduler.WaitQueue = .{}, // senders wait here
|
||||
not_empty: scheduler.WaitQueue = .{}, // receivers wait here
|
||||
|
||||
/// Send a message, blocking while the channel is full.
|
||||
pub fn send(self: *Self, message: T) void {
|
||||
const flags = sync.enter();
|
||||
// Recheck the condition in a loop: a wakeup only means "try again"
|
||||
// (another waiter may have taken the slot first).
|
||||
while (self.count == capacity) scheduler.waitLocked(&self.not_full);
|
||||
self.buffer[self.tail] = message;
|
||||
self.tail = (self.tail + 1) % capacity;
|
||||
self.count += 1;
|
||||
scheduler.wakeLocked(&self.not_empty); // a receiver can now proceed
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// Receive a message, blocking while the channel is empty.
|
||||
pub fn receive(self: *Self) T {
|
||||
const flags = sync.enter();
|
||||
while (self.count == 0) scheduler.waitLocked(&self.not_empty);
|
||||
const message = self.buffer[self.head];
|
||||
self.head = (self.head + 1) % capacity;
|
||||
self.count -= 1;
|
||||
scheduler.wakeLocked(&self.not_full); // a sender can now proceed
|
||||
sync.leave(flags);
|
||||
return message;
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,190 @@
|
||||
//! IRQ-as-IPC: delivering a hardware interrupt to a user-space driver.
|
||||
//!
|
||||
//! A microkernel can't run driver code in the ISR — the driver is a ring-3 process
|
||||
//! in another address space. So the kernel's ISR does the least it can: quiet the
|
||||
//! line, acknowledge the CPU, and post an asynchronous notification to the endpoint
|
||||
//! the driver is blocked on (`ipc_sync.notifyFromIsr`). The driver wakes out of
|
||||
//! `IPC_ReplyWait`, services the device, and calls `irq_ack` to re-arm.
|
||||
//!
|
||||
//! The full cycle, and why each step is where it is:
|
||||
//!
|
||||
//! ISR irqMask(gsi) -- the line is still asserted; stop it reaching a CPU
|
||||
//! irqEoi() -- now safe to tell the LAPIC we're done
|
||||
//! notifyFromIsr() -- wake the driver (it runs much later)
|
||||
//! driver <services device> -- reads/clears the device's status register
|
||||
//! driver irq_ack(device,resource) -- irqUnmask(gsi): the line is quiet, let it through
|
||||
//!
|
||||
//! Mask-before-EOI is the load-bearing part. A level-triggered line stays asserted
|
||||
//! until the *device* is quieted, which only the ring-3 driver can do. EOI with the
|
||||
//! entry unmasked and the I/O APIC redelivers immediately, forever, before the
|
||||
//! driver is ever scheduled. Masking converts "level" into something a deferred
|
||||
//! handler can cope with; `irq_ack` is what closes the loop.
|
||||
//!
|
||||
//! Binding is capability-gated exactly like `mmio_map`: the caller must have
|
||||
//! `device_claim`ed the device, and the GSI must come from one of that device's `irq`
|
||||
//! resources in the discovered device table (system/kernel/device-service.zig). A driver can
|
||||
//! therefore never bind an interrupt it doesn't own — a raw-GSI system_call would let
|
||||
//! any process steal the keyboard's line.
|
||||
//!
|
||||
//! KNOWN ISSUE (real hardware, not QEMU). Masking a level-triggered redirection entry
|
||||
//! while its remote-IRR bit is set does not clear remote-IRR on some chipsets, and the
|
||||
//! line then never fires again — a driver would take exactly one interrupt and block
|
||||
//! forever. QEMU's I/O APIC clears remote-IRR on EOI regardless of the mask, so the
|
||||
//! `hpet` test cannot see this. Linux's workaround is to flush remote-IRR by briefly
|
||||
//! flipping the entry to edge-trigger and back. Revisit when danos first boots on
|
||||
//! metal; MSI (no mask cycle at all) sidesteps it entirely.
|
||||
|
||||
const architecture = @import("architecture");
|
||||
const sync = @import("sync.zig");
|
||||
const ipc_sync = @import("ipc-synchronous.zig");
|
||||
|
||||
/// GSIs a single I/O APIC covers. 24 is the standard redirection-table size; a
|
||||
/// second I/O APIC (none on QEMU's q35) would extend this.
|
||||
pub const maximum_gsi = 24;
|
||||
|
||||
/// The endpoint to notify for each bound GSI, or null if unbound. Read from the ISR
|
||||
/// and written from syscalls, always under the big kernel lock.
|
||||
var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi;
|
||||
|
||||
/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint
|
||||
/// pointer: an endpoint can be shared between processes (ipc_register/ipc_lookup hand
|
||||
/// out extra references), so "every GSI pointing at this endpoint" is not the same
|
||||
/// set as "every GSI this process bound", and releasing the former on exit would mask
|
||||
/// a live sibling's device line.
|
||||
var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi;
|
||||
|
||||
/// No GSI assigned to this vector.
|
||||
const no_gsi: u32 = 0xFFFF_FFFF;
|
||||
|
||||
/// Reverse map for the ISR: which GSI does this vector carry? Populated at bind.
|
||||
/// `interruptDispatch` hands a handler no arguments, so the vector→GSI edge has to
|
||||
/// be recovered from somewhere — the trampolines below capture the vector at
|
||||
/// comptime, and this turns it back into a GSI. Trampolines are installed on every
|
||||
/// vector in the window at boot, so an unbound one must be distinguishable from
|
||||
/// GSI 0 — hence the sentinel rather than a zero default.
|
||||
var vector_gsi: [256]u32 = .{no_gsi} ** 256;
|
||||
|
||||
/// Vector currently assigned to each GSI (0 = none), so a rebind reuses it.
|
||||
var gsi_vector: [maximum_gsi]u8 = .{0} ** maximum_gsi;
|
||||
|
||||
/// Set once the trampolines are installed.
|
||||
var installed = false;
|
||||
|
||||
/// The ISR body for a bound device line. Runs with interrupts off, on the
|
||||
/// interrupted task's kernel stack, on whichever core the I/O APIC picked.
|
||||
fn dispatch(vector: u8) void {
|
||||
const gsi = vector_gsi[vector];
|
||||
if (gsi == no_gsi) {
|
||||
// Nothing is routed here. Acknowledge so the LAPIC doesn't wedge on an
|
||||
// in-service bit that never clears, but touch no redirection entry.
|
||||
architecture.irqEoi();
|
||||
return;
|
||||
}
|
||||
|
||||
// One lock region for the whole cycle. Two reasons, and the second is subtle:
|
||||
//
|
||||
// - The I/O APIC is an index/data register pair, so two cores interleaving a
|
||||
// read-modify-write of a redirection entry would corrupt it.
|
||||
// - `bound[gsi]` must be *read and used* under the same acquisition that
|
||||
// `unbind` writes it under. Dropping the lock between the load and
|
||||
// `notifyLocked` would let a driver exiting on another core free the endpoint
|
||||
// in the gap, and we would post a notification into freed memory. The GSI is
|
||||
// routed to the core that bound it, but a driver may migrate and exit
|
||||
// elsewhere, so this is reachable on SMP.
|
||||
//
|
||||
// No deadlock: the lock is non-recursive, but a core holding it runs with
|
||||
// interrupts disabled and so cannot interrupt itself into here.
|
||||
_ = sync.enter();
|
||||
defer sync.leaveIsr();
|
||||
|
||||
architecture.irqMask(gsi); // the line is still asserted; stop it reaching a CPU
|
||||
architecture.irqEoi(); // now safe to release the LAPIC's in-service bit
|
||||
|
||||
// Wakes the driver if it's blocked in ReplyWait; otherwise queues the badge on
|
||||
// the endpoint's notify ring, so an interrupt taken while the driver is off
|
||||
// doing something else is not lost.
|
||||
if (bound[gsi]) |endpoint| ipc_sync.notifyLocked(endpoint, gsi);
|
||||
}
|
||||
|
||||
/// Install one no-argument trampoline per usable vector. Each closes over its own
|
||||
/// `vector` as a comptime constant — that's the trick that gets an argument into
|
||||
/// `idt.Handler` (`*const fn () void`) without a per-vector hand-written stub.
|
||||
pub fn init() void {
|
||||
if (installed) return;
|
||||
inline for (0..architecture.irq_vector_count) |i| {
|
||||
const vector: u8 = @intCast(@as(usize, architecture.irq_vector_base) + i);
|
||||
architecture.irqSetHandler(vector, &struct {
|
||||
fn trampoline() void {
|
||||
dispatch(vector);
|
||||
}
|
||||
}.trampoline);
|
||||
}
|
||||
installed = true;
|
||||
}
|
||||
|
||||
/// Lowest unused vector in the device window, or null if they're all spoken for.
|
||||
fn allocVector() ?u8 {
|
||||
var v: u8 = architecture.irq_vector_base;
|
||||
while (v < architecture.irq_vector_base + architecture.irq_vector_count) : (v += 1) {
|
||||
var used = false;
|
||||
for (gsi_vector) |gv| {
|
||||
if (gv == v) used = true;
|
||||
}
|
||||
if (!used) return v;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub const BindError = error{ BadGsi, InUse, NoVector };
|
||||
|
||||
/// Deliver `gsi` to `endpoint` as an IPC notification, on behalf of task `owner`. Routes the
|
||||
/// line to this core, installs the binding, and unmasks. Caller must hold the big
|
||||
/// kernel lock, and must already have checked that `owner` claimed the device this GSI
|
||||
/// belongs to.
|
||||
pub fn bind(gsi: u32, endpoint: *ipc_sync.Endpoint, owner: u32) BindError!void {
|
||||
if (gsi >= maximum_gsi or !architecture.irqOwnsGsi(gsi)) return error.BadGsi;
|
||||
if (bound[gsi] != null) return error.InUse;
|
||||
const vector = allocVector() orelse return error.NoVector;
|
||||
|
||||
vector_gsi[vector] = gsi;
|
||||
gsi_vector[gsi] = vector;
|
||||
bound[gsi] = endpoint;
|
||||
bound_owner[gsi] = owner;
|
||||
|
||||
// Level-triggered, active-high. Level is the general case a driver must survive
|
||||
// (and what hpetd configures its comparator for); an edge source simply never
|
||||
// leaves the line asserted, so the mask/ack cycle is harmless there.
|
||||
//
|
||||
// Hardcoded for now: a device whose MADT interrupt-source override declares the
|
||||
// line active-*low* (most legacy PCI INTx) will need the polarity threaded
|
||||
// through from discovery. Nothing danos binds today is such a device.
|
||||
architecture.irqRoute(gsi, vector, true, false);
|
||||
architecture.irqUnmask(gsi);
|
||||
}
|
||||
|
||||
/// Re-arm `gsi` after the driver has quieted the device. Caller holds the big lock
|
||||
/// and has verified ownership.
|
||||
pub fn ack(gsi: u32) bool {
|
||||
if (gsi >= maximum_gsi or bound[gsi] == null) return false;
|
||||
architecture.irqUnmask(gsi);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Drop every binding made by task `owner` — called as that task exits, *before* its
|
||||
/// endpoints are freed. Each line is left masked, so a dead driver's device goes quiet
|
||||
/// rather than interrupting into a freed endpoint. Caller holds the big kernel lock,
|
||||
/// which is what makes this safe against a concurrent `dispatch` on another core.
|
||||
///
|
||||
/// Keyed on the owner, not the endpoint: endpoints are shared (a registered service's
|
||||
/// endpoint has references in several processes), so releasing "everything pointing at
|
||||
/// this endpoint" would tear down bindings this task never made.
|
||||
pub fn releaseOwner(owner: u32) void {
|
||||
for (&bound, 0..) |*slot, gsi| {
|
||||
if (slot.* != null and bound_owner[gsi] == owner) {
|
||||
architecture.irqMask(@intCast(gsi));
|
||||
slot.* = null;
|
||||
bound_owner[gsi] = 0;
|
||||
gsi_vector[gsi] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
//! The kernel's multi-sink **diagnostic** log — the machine-readable stream of
|
||||
//! what the kernel is doing, separate from any user-facing display.
|
||||
//!
|
||||
//! Output is a *diagnostic convenience, never a correctness dependency* — the
|
||||
//! kernel must boot and run correctly with zero output channels. So logging fans
|
||||
//! out to a set of registered **sinks**, each best-effort and self-guarding: the
|
||||
//! serial UART, the 0xE9 debug console, and — later — a file on a ramdisk/USB/SSD.
|
||||
//! A message reaches whatever channels exist; if none do, the kernel runs on,
|
||||
//! silent but correct.
|
||||
//!
|
||||
//! The **framebuffer is deliberately not a sink here.** It's a separate output
|
||||
//! surface (a bootstrap text console today, a graphics device driver later), so
|
||||
//! the log never assumes the machine is text-based. `main.zig` mirrors a few
|
||||
//! user-facing status lines and panics to it explicitly; the verbose log does not.
|
||||
//!
|
||||
//! No allocation: the sink table is fixed, so the log works before the heap is up
|
||||
//! and inside a panic. Two channels don't go through the sink list because they
|
||||
//! must survive even a total-output failure: `checkpoint` (a one-byte POST code)
|
||||
//! and `recordPanic` (a breadcrumb in a fixed record).
|
||||
|
||||
const std = @import("std");
|
||||
const architecture = @import("architecture");
|
||||
|
||||
pub const SinkFn = *const fn ([]const u8) void;
|
||||
|
||||
const maximum_sinks = 8;
|
||||
var sinks: [maximum_sinks]SinkFn = undefined;
|
||||
var sink_count: usize = 0;
|
||||
|
||||
/// Register an output sink. Every registered sink receives every message; sinks
|
||||
/// must be self-guarding (safe to call when their device is absent).
|
||||
pub fn addSink(sink: SinkFn) void {
|
||||
if (sink_count < maximum_sinks) {
|
||||
sinks[sink_count] = sink;
|
||||
sink_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Fan `bytes` out to every registered sink.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
}
|
||||
|
||||
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
||||
/// this is safe to call from interrupt context and from a panic.
|
||||
pub fn print(comptime fmt: []const u8, args: anytype) void {
|
||||
var buffer: [256]u8 = undefined;
|
||||
write(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// Emit a one-byte checkpoint/POST code (I/O port 0x80) — the always-available
|
||||
/// progress channel for when there is no text output at all. Independent of the
|
||||
/// sink list, so it works even before any sink is registered.
|
||||
pub fn checkpoint(code: u8) void {
|
||||
architecture.checkpoint(code);
|
||||
}
|
||||
|
||||
// --- persistent panic breadcrumb -------------------------------------------
|
||||
//
|
||||
// A fixed record in the kernel image that a panic fills in, so a post-mortem — an
|
||||
// attached debugger, a RAM dump, or (later) a file/pstore reader — can recover
|
||||
// what killed the kernel even when there was no live console. `magic` is written
|
||||
// *last*, so a reader only trusts a fully-written record.
|
||||
|
||||
pub const panic_magic: u64 = 0xD1ED_B00B_5EED_F00D;
|
||||
|
||||
pub const PanicRecord = extern struct {
|
||||
magic: u64 = 0,
|
||||
len: u32 = 0,
|
||||
_pad: u32 = 0,
|
||||
message: [512]u8 = undefined,
|
||||
};
|
||||
|
||||
/// Findable by symbol (`log.panic_record`) for a debugger or RAM dump.
|
||||
pub var panic_record: PanicRecord = .{};
|
||||
|
||||
/// Stamp the panic message into the breadcrumb record.
|
||||
pub fn recordPanic(message: []const u8) void {
|
||||
const n: u32 = @intCast(@min(message.len, panic_record.message.len));
|
||||
@memcpy(panic_record.message[0..n], message[0..n]);
|
||||
panic_record.len = n;
|
||||
panic_record.magic = panic_magic; // set last: a reader sees a complete record
|
||||
}
|
||||
@@ -0,0 +1,431 @@
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const parameters = @import("parameters");
|
||||
const architecture = @import("architecture");
|
||||
const console = @import("console.zig");
|
||||
const log = @import("log.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const process = @import("process.zig");
|
||||
const device_service = @import("device-service.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const initrd = @import("initrd");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = danos.BootInformation;
|
||||
|
||||
/// The calling convention used to enter the kernel. Pinned to SystemV explicitly:
|
||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
||||
/// x64 (first argument in RCX), while the kernel is SystemV (first argument in
|
||||
/// RDI). Both sides reference this so the `boot_information` pointer lands in the
|
||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
||||
pub const kernel_abi = danos.kernel_abi;
|
||||
|
||||
// POST/checkpoint codes emitted to I/O port 0x80 at boot milestones — the
|
||||
// last-resort progress signal on a machine with no text output at all.
|
||||
const cp_entry = 0x10;
|
||||
const cp_paging = 0x20;
|
||||
const cp_heap = 0x30;
|
||||
const cp_discovery = 0x40;
|
||||
const cp_scheduler = 0x50;
|
||||
const cp_timer = 0x60;
|
||||
const cp_running = 0x70;
|
||||
const cp_exception = 0xE0;
|
||||
const cp_panic = 0xEE;
|
||||
|
||||
/// Physical address of the low page reserved at boot for the AP trampoline (0 = none
|
||||
/// was available). Claimed right after the frame allocator comes up, before paging
|
||||
/// and the heap consume the scarce sub-1 MiB frames.
|
||||
var ap_trampoline_page: u64 = 0;
|
||||
|
||||
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
||||
/// caller to return to, so this never returns.
|
||||
/// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss
|
||||
/// then calls this with the loader's `boot_information` pointer in RDI. We can't keep
|
||||
/// running on the loader's stack: it's a low physical address that the identity
|
||||
/// map covers only transitionally, and vanishes once the kernel drops the low
|
||||
/// half. `boot_information` (also low) is reached through the physmap — its base is the
|
||||
/// same under the loader's bootstrap tables and the kernel's own.
|
||||
export fn kmainEntry(boot_information: *const BootInformation) callconv(kernel_abi) noreturn {
|
||||
kmain(@ptrFromInt(danos.physicalToVirtual(@intFromPtr(boot_information))));
|
||||
}
|
||||
|
||||
fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// The **log** is the machine-readable diagnostic stream: it fans out to every
|
||||
// *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a
|
||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||
// with port-0x80 checkpoints as the only progress signal.
|
||||
architecture.serialInit();
|
||||
log.addSink(architecture.serialWrite);
|
||||
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||
|
||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||
const fb = boot_information.framebuffer;
|
||||
console.init(fb);
|
||||
|
||||
log.checkpoint(cp_entry);
|
||||
|
||||
// Catch CPU exceptions before doing anything that might fault: install our
|
||||
// reporter, then bring up the GDT + IDT.
|
||||
architecture.setFaultHandler(onException);
|
||||
architecture.init();
|
||||
|
||||
status("danos: initialising kernel...\n");
|
||||
log.write(if (console.present())
|
||||
"danos: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
else
|
||||
"danos: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
log.write("danos: cpu tables online (GDT, IDT, TSS)\n");
|
||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||
log.print(" format : {s}\n", .{@tagName(fb.format)});
|
||||
log.print(" framebuffer: 0x{x:0>16}\n", .{fb.base});
|
||||
log.print (" footprint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
|
||||
|
||||
// Summarise the physical memory the loader handed us. The array is danos's
|
||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
|
||||
var usable_pages: u64 = 0;
|
||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
||||
for (regions) |r| {
|
||||
switch (r.kind) {
|
||||
.usable => usable_pages += r.pages,
|
||||
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
|
||||
.mmio => {},
|
||||
}
|
||||
}
|
||||
const total_pages = usable_pages + reserved_pages;
|
||||
const total_bytes = total_pages * danos.page_size;
|
||||
const gib = 1 << 30;
|
||||
|
||||
log.write("\ndanos: physical memory\n");
|
||||
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
||||
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
||||
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
||||
log.print(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
|
||||
|
||||
// Bring up the physical frame allocator over that map, and prove it works:
|
||||
// allocate three frames, then hand them back.
|
||||
pmm.init(boot_information.memory_map);
|
||||
// Claim the AP trampoline's low (<1 MiB) page *now*, before paging and the heap
|
||||
// draw down sub-1 MiB frames (the allocator scans upward from frame 0). Held
|
||||
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
|
||||
ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0;
|
||||
const s1 = pmm.stats();
|
||||
log.print("\ndanos: frame allocator online\n", .{});
|
||||
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
||||
const f0 = pmm.alloc();
|
||||
const f1 = pmm.alloc();
|
||||
const f2 = pmm.alloc();
|
||||
log.print(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
|
||||
if (f0) |p| pmm.free(p);
|
||||
if (f1) |p| pmm.free(p);
|
||||
if (f2) |p| pmm.free(p);
|
||||
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
||||
|
||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||
architecture.enablePaging(pmm.alloc, pmm.free, boot_information);
|
||||
log.checkpoint(cp_paging);
|
||||
log.print("\ndanos: paging enabled\n", .{});
|
||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||
|
||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||
heap.init();
|
||||
log.checkpoint(cp_heap);
|
||||
log.write("\ndanos: kernel heap online\n");
|
||||
// Measure the amount of resources the kernel is actually using
|
||||
const s2 = pmm.stats();
|
||||
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
||||
|
||||
// Enumerate hardware from the firmware tables (ACPI here) into a generic
|
||||
// device tree, then list it. Discovery walks ACPI memory directly (identity-
|
||||
// mapped) and maps PCIe configuration space on demand via the VMM. A failure here is
|
||||
// not fatal yet — log it and carry on.
|
||||
const hal = platform.Hal{
|
||||
.mapMmio = architecture.mapMmio,
|
||||
.pioRead = architecture.pioRead,
|
||||
.pioWrite = architecture.pioWrite,
|
||||
};
|
||||
if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| {
|
||||
var device_tree = devtree;
|
||||
log.write("\ndanos: device discovery online\n");
|
||||
device_tree.dump(log.write);
|
||||
|
||||
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
|
||||
// mmio_map operate on this flat, id-indexed table + claim map).
|
||||
device_service.init(&device_tree);
|
||||
if (device_service.dropped > 0) {
|
||||
// Otherwise entirely silent: drivers would just never see that hardware.
|
||||
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{device_service.dropped});
|
||||
}
|
||||
|
||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||
irq.init();
|
||||
|
||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
||||
const pw = platform.powerInformation();
|
||||
log.write("danos: power\n");
|
||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||
if (pw.s5) |s| {
|
||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
||||
} else {
|
||||
log.write(" S5 slp_typ : (not found)\n");
|
||||
}
|
||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||
|
||||
// AML namespace parse integrity: consumed should equal total.
|
||||
const am = platform.amlStats();
|
||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
||||
|
||||
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||
const pinfo = platform.platformInformation();
|
||||
const hpet_base: u64 = if (device_tree.firstOfClass(.timer)) |t|
|
||||
(if (t.firstResource(.memory)) |r| r.start else 0)
|
||||
else
|
||||
0;
|
||||
var ioapic_base: u64 = 0;
|
||||
var ioapic_gsi: u32 = 0;
|
||||
if (device_tree.firstOfClass(.interrupt_controller)) |ic| {
|
||||
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
|
||||
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
|
||||
}
|
||||
var isos: [16]architecture.IsoEntry = undefined;
|
||||
const iso_n = @min(pinfo.override_count, isos.len);
|
||||
for (0..iso_n) |i| isos[i] = .{
|
||||
.source = pinfo.overrides[i].source,
|
||||
.gsi = pinfo.overrides[i].gsi,
|
||||
.flags = pinfo.overrides[i].flags,
|
||||
};
|
||||
const pm_timer: ?architecture.PmTimer = if (pinfo.pm_timer.present())
|
||||
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
|
||||
else
|
||||
null;
|
||||
architecture.configurePlatform(.{
|
||||
.pic_present = pinfo.pic_present,
|
||||
.hpet_base = hpet_base,
|
||||
.pm_timer = pm_timer,
|
||||
.ioapic_base = ioapic_base,
|
||||
.ioapic_gsi_base = ioapic_gsi,
|
||||
.overrides = isos[0..iso_n],
|
||||
});
|
||||
if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address);
|
||||
|
||||
log.write("danos: platform\n");
|
||||
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
||||
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
||||
log.print(" hpet base : 0x{x}\n", .{hpet_base});
|
||||
log.print(" pm timer : {s} 0x{x} ({s})\n", .{ if (pinfo.pm_timer.mmio) "mmio" else "io", pinfo.pm_timer.address, if (pinfo.pm_timer_32bit) "32-bit" else "24-bit" });
|
||||
if (pinfo.spcr_uart) |u| {
|
||||
log.print(" console UART: {s} 0x{x} (SPCR type {d})\n", .{ if (u.mmio) "mmio" else "io", u.address, pinfo.spcr_kind });
|
||||
} else {
|
||||
log.write(" console UART: none in SPCR -> legacy COM1\n");
|
||||
}
|
||||
log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, architecture.irqRouteCount(), architecture.irqRouteRaw(0) });
|
||||
const cores = platform.cpus();
|
||||
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
||||
if (platform.cpusDropped() > 0)
|
||||
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
||||
} else |err| {
|
||||
log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
log.checkpoint(cp_discovery);
|
||||
|
||||
// Install the system_call handler (int 0x80 gate + system_call stub) once, before any
|
||||
// user code runs.
|
||||
process.init();
|
||||
|
||||
// Register the current context as the first task before enabling preemption.
|
||||
scheduler.init(4);
|
||||
log.checkpoint(cp_scheduler);
|
||||
log.write("\ndanos: scheduler online\n");
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||
// the timer preempts among tasks.
|
||||
architecture.startTimer();
|
||||
architecture.enableInterrupts();
|
||||
log.checkpoint(cp_timer);
|
||||
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||
|
||||
// Wake the other cores (application processors). A no-op on a single-core
|
||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
||||
bringUpSecondaries();
|
||||
|
||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||
// Normal builds fall through to the idle halt.
|
||||
if (build_options.test_case) |case| {
|
||||
tests.run(case, boot_information);
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
log.checkpoint(cp_running);
|
||||
status("kernel initialised.\n");
|
||||
|
||||
// Hand over to user space: load /sbin/init (read off the boot volume by the
|
||||
// loader) and spawn it as a real ring-3 process, PID 1. It runs on its own
|
||||
// address space, preemptively, alongside the kernel — no cooperative
|
||||
// borrowing. This boot context then becomes the BSP's idle loop.
|
||||
if (boot_information.init_len != 0) {
|
||||
status("starting /sbin/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
process.spawnProcess(image, 4) catch |err| {
|
||||
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
} else {
|
||||
status("no /sbin/init on the boot volume.\n");
|
||||
}
|
||||
|
||||
// Spawn the extra user binaries the loader ferried in the initrd (the VFS
|
||||
// server, and later device drivers). For now the kernel launches them all;
|
||||
// once init is a real service supervisor it will spawn them itself (system_spawn).
|
||||
startInitrdBinaries(boot_information);
|
||||
|
||||
// Become the idle task: drop below every real task and halt until an
|
||||
// interrupt. The timer keeps preempting into init and any other work.
|
||||
scheduler.setPriority(0);
|
||||
status("\nkernel idle; /sbin/init is running.\n");
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
/// Spawn every program bundled in the initrd as its own ring-3 process. A bad
|
||||
/// image or a program that fails to load is logged and skipped — the rest of the
|
||||
/// system still runs.
|
||||
fn startInitrdBinaries(boot_information: *const danos.BootInformation) void {
|
||||
if (boot_information.initrd_len == 0) return;
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
|
||||
const rd = initrd.Reader.init(image) orelse {
|
||||
status("initrd: bad image, skipping\n");
|
||||
return;
|
||||
};
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
statusPrint("starting /sbin/{s} (from initrd)...\n", .{item.name});
|
||||
process.spawnProcess(item.blob, 4) catch |err| {
|
||||
statusPrint("initrd: {s} failed to load: {s}\n", .{ item.name, @errorName(err) });
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/// Wake the application processors the firmware left parked. Allocates the low
|
||||
/// trampoline page (and makes it executable), then wakes each non-boot core in turn,
|
||||
/// handing it a fresh kernel stack and its per-CPU slot. Cores that don't report in
|
||||
/// are left parked — the running system is unaffected. See docs/smp.md.
|
||||
fn bringUpSecondaries() void {
|
||||
const cores = platform.cpus();
|
||||
if (cores.len <= 1) return;
|
||||
|
||||
// A low (<1 MiB) frame was reserved at boot for the real-mode trampoline (a SIPI
|
||||
// vector addresses it). It's kept for the system's life — armed only during a
|
||||
// wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later.
|
||||
if (ap_trampoline_page == 0) {
|
||||
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
||||
return;
|
||||
}
|
||||
architecture.setTrampolinePage(ap_trampoline_page);
|
||||
architecture.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop
|
||||
|
||||
// Test hook: the smp-retry case forces the first wake to fail, so the retry below
|
||||
// must still bring every core online. Inert in a normal build (test_case is null).
|
||||
if (build_options.test_case) |tc| {
|
||||
if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1);
|
||||
}
|
||||
|
||||
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||
const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
|
||||
for (cores[1..], 1..) |core, index| {
|
||||
const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch {
|
||||
log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id});
|
||||
continue;
|
||||
};
|
||||
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
|
||||
// This core's dedicated fault stack — allocated only now that the core is
|
||||
// real, rather than reserved statically for every possible core.
|
||||
const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch {
|
||||
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||
continue;
|
||||
};
|
||||
architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
||||
const pc = scheduler.prepareSecondary(index, core.apic_id);
|
||||
var attempt: u32 = 1;
|
||||
while (attempt <= maximum_wake_attempts) : (attempt += 1) {
|
||||
if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
||||
pc.online = true;
|
||||
log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
||||
break;
|
||||
}
|
||||
if (attempt == maximum_wake_attempts)
|
||||
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
||||
}
|
||||
}
|
||||
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||
}
|
||||
|
||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
||||
/// touches the framebuffer.
|
||||
fn status(message: []const u8) void {
|
||||
log.write(message);
|
||||
console.write(message);
|
||||
}
|
||||
|
||||
fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
||||
var buffer: [256]u8 = undefined;
|
||||
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// Frames (4 KiB pages) to whole MiB.
|
||||
fn mib(pages: u64) u64 {
|
||||
return pages * danos.page_size / (1024 * 1024);
|
||||
}
|
||||
|
||||
fn kib(frames: u64) u64 {
|
||||
return frames * danos.page_size / (1024);
|
||||
}
|
||||
|
||||
/// Report a CPU exception and halt **this core**. There's no fault recovery yet, so
|
||||
/// the faulting core is terminal — but the fault is *contained* to it: on an
|
||||
/// application processor only that core stops, and the rest of the system keeps
|
||||
/// running (full recovery — kill the task, keep the core — is the resilience track,
|
||||
/// see docs/resilience.md). The report names the core so an AP fault is attributed,
|
||||
/// and goes to every output sink plus a POST code and a persistent breadcrumb.
|
||||
fn onException(state: *const architecture.CpuState) noreturn {
|
||||
log.checkpoint(cp_exception);
|
||||
const core = scheduler.currentCpuIndex();
|
||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
||||
// top of the diagnostic log.
|
||||
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
|
||||
var buffer: [128]u8 = undefined;
|
||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
|
||||
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
|
||||
/// no live console), then halt. Assumes no console — the sinks self-guard.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(message: []const u8, first_trace_address: ?usize) noreturn {
|
||||
_ = first_trace_address;
|
||||
log.checkpoint(cp_panic);
|
||||
log.recordPanic(message);
|
||||
status("\nKERNEL PANIC: ");
|
||||
status(message);
|
||||
status("\n");
|
||||
architecture.halt();
|
||||
}
|
||||
}.panic);
|
||||
@@ -0,0 +1,174 @@
|
||||
//! Physical memory manager: a bitmap frame allocator. It hands out and reclaims
|
||||
//! 4 KiB physical frames — the primitive every later memory feature (page
|
||||
//! tables, the heap) is built on top of.
|
||||
//!
|
||||
//! This is generic kernel code: it works on the neutral `danos.MemoryRegion`
|
||||
//! array the loader hands over (see docs/memory-map.md), so it carries no UEFI
|
||||
//! and nothing architecture-specific beyond the 4 KiB page.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
/// One bit per frame, covering physical RAM from 0 up to the highest usable
|
||||
/// address: 1 = used/unavailable, 0 = free. The bitmap itself lives in a frame
|
||||
/// we carve out of usable memory during init.
|
||||
var bitmap: []u8 = &.{};
|
||||
var total_frames: usize = 0;
|
||||
var used_frames: usize = 0;
|
||||
/// Where the next allocation scan begins, so we don't rescan from frame 0 every
|
||||
/// time. Pulled back on free() so reclaimed low frames get reused.
|
||||
var next_hint: usize = 0;
|
||||
|
||||
pub const Stats = struct {
|
||||
total_frames: usize,
|
||||
used_frames: usize,
|
||||
free_frames: usize,
|
||||
};
|
||||
|
||||
pub fn stats() Stats {
|
||||
return .{
|
||||
.total_frames = total_frames,
|
||||
.used_frames = used_frames,
|
||||
.free_frames = total_frames - used_frames,
|
||||
};
|
||||
}
|
||||
|
||||
inline fn bit(frame: usize) u3 {
|
||||
return @intCast(frame & 7);
|
||||
}
|
||||
inline fn isUsed(frame: usize) bool {
|
||||
return (bitmap[frame >> 3] >> bit(frame)) & 1 != 0;
|
||||
}
|
||||
inline fn setUsed(frame: usize) void {
|
||||
bitmap[frame >> 3] |= @as(u8, 1) << bit(frame);
|
||||
}
|
||||
inline fn setFree(frame: usize) void {
|
||||
bitmap[frame >> 3] &= ~(@as(u8, 1) << bit(frame));
|
||||
}
|
||||
|
||||
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(map.regions)))[0..map.len];
|
||||
}
|
||||
|
||||
/// Build the allocator from the loader's memory map. Reaches physical memory
|
||||
/// (the region array, the bitmap's own storage) through the physmap, which the
|
||||
/// loader's bootstrap tables already provide — so this works before the kernel
|
||||
/// installs its own tables. Invariant: the bitmap lands in the first usable
|
||||
/// region (lowest address), which must sit under the bootstrap physmap's reach
|
||||
/// (4 GiB); it always does, as both this and the page-table allocator scan from
|
||||
/// low addresses up.
|
||||
pub fn init(map: danos.MemoryMap) void {
|
||||
const regs = regions(map);
|
||||
|
||||
// 1. Size the bitmap to cover every frame up to the highest RAM address —
|
||||
// including reserved RAM, so those frames are trackable (e.g. to free the
|
||||
// boot buffers later). Only MMIO (device address space) is excluded.
|
||||
// Everything starts unallocatable; usable regions are freed below.
|
||||
var highest: u64 = 0;
|
||||
for (regs) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
const end = r.base + r.pages * page_size;
|
||||
if (end > highest) highest = end;
|
||||
}
|
||||
total_frames = @intCast(highest / page_size);
|
||||
if (total_frames == 0) @panic("pmm: no usable memory");
|
||||
const bitmap_bytes = (total_frames + 7) / 8;
|
||||
const bitmap_pages = (bitmap_bytes + page_size - 1) / page_size;
|
||||
|
||||
// 2. Park the bitmap in the first usable region large enough to hold it.
|
||||
// Start at least one page in, so we never place it on frame 0 (which is
|
||||
// kept reserved as the "none" address, and is an awkward pointer besides).
|
||||
var storage: ?u64 = null;
|
||||
for (regs) |r| {
|
||||
if (r.kind != .usable) continue;
|
||||
const base = if (r.base == 0) page_size else r.base;
|
||||
const skipped = (base - r.base) / page_size;
|
||||
if (r.pages - skipped >= bitmap_pages) {
|
||||
storage = base;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
|
||||
bitmap = @as([*]u8, @ptrFromInt(danos.physicalToVirtual(bitmap_base)))[0..bitmap_bytes];
|
||||
|
||||
// 3. Start with everything marked used, then free the usable regions. Doing
|
||||
// it this way means every gap, reserved span and MMIO hole is unallocatable
|
||||
// by default — we only ever hand back memory the firmware called usable.
|
||||
@memset(bitmap, 0xff);
|
||||
used_frames = total_frames;
|
||||
for (regs) |r| {
|
||||
if (r.kind != .usable) continue;
|
||||
var f: usize = @intCast(r.base / page_size);
|
||||
const end = f + @as(usize, @intCast(r.pages));
|
||||
while (f < end and f < total_frames) : (f += 1) {
|
||||
setFree(f);
|
||||
used_frames -= 1;
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Take back the frames the bitmap occupies, and frame 0 — so a 0 result
|
||||
// stays reserved to mean "no frame".
|
||||
reserve(bitmap_base, bitmap_pages);
|
||||
reserve(0, 1);
|
||||
}
|
||||
|
||||
/// Mark `count` frames from physical `base` as used, counting only those that
|
||||
/// were actually free.
|
||||
fn reserve(base: u64, count: usize) void {
|
||||
var f: usize = @intCast(base / page_size);
|
||||
const end = f + count;
|
||||
while (f < end and f < total_frames) : (f += 1) {
|
||||
if (!isUsed(f)) {
|
||||
setUsed(f);
|
||||
used_frames += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate one physical frame, or null if none are free. The address is
|
||||
/// page-aligned; the frame's contents are undefined.
|
||||
pub fn alloc() ?u64 {
|
||||
var scanned: usize = 0;
|
||||
var f = next_hint;
|
||||
while (scanned < total_frames) : (scanned += 1) {
|
||||
if (f >= total_frames) f = 0;
|
||||
if (!isUsed(f)) {
|
||||
setUsed(f);
|
||||
used_frames += 1;
|
||||
next_hint = f + 1;
|
||||
return @as(u64, f) * page_size;
|
||||
}
|
||||
f += 1;
|
||||
}
|
||||
return null; // out of physical memory
|
||||
}
|
||||
|
||||
/// Allocate one free frame whose physical address is below `limit`, or null if
|
||||
/// none is free down there. The AP trampoline needs this: an x86 STARTUP IPI vectors
|
||||
/// a waking core to physical `vector << 12`, and `vector` is a byte — so the
|
||||
/// trampoline must live under 1 MiB. A short linear scan of the low frames; only run
|
||||
/// a handful of times at boot, so it needn't be fast.
|
||||
pub fn allocBelow(limit: u64) ?u64 {
|
||||
const cap = @min(total_frames, @as(usize, @intCast(limit / page_size)));
|
||||
var f: usize = 1; // frame 0 stays reserved as the "none" address
|
||||
while (f < cap) : (f += 1) {
|
||||
if (!isUsed(f)) {
|
||||
setUsed(f);
|
||||
used_frames += 1;
|
||||
return @as(u64, f) * page_size;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Return a frame obtained from alloc() to the pool. Bogus or double frees are
|
||||
/// ignored rather than corrupting the count.
|
||||
pub fn free(address: u64) void {
|
||||
const f: usize = @intCast(address / page_size);
|
||||
if (f >= total_frames or !isUsed(f)) return;
|
||||
setFree(f);
|
||||
used_frames -= 1;
|
||||
if (f < next_hint) next_hint = f;
|
||||
}
|
||||
@@ -0,0 +1,579 @@
|
||||
//! User-space processes: loading a user ELF and running it in ring 3. danos is a
|
||||
//! microkernel, so this only ever loads *user* binaries — there is no kernel-space
|
||||
//! loader; in-kernel code is linked into the kernel image, not loaded here.
|
||||
//!
|
||||
//! Two entry points:
|
||||
//! - `spawnProcess` loads a user ELF (`/sbin/init`, and later servers/drivers)
|
||||
//! into a fresh address space and schedules it as a real preemptive ring-3
|
||||
//! process on its own page tables. This is the production path.
|
||||
//! - `run` executes a raw code blob (the user-pf isolation test program) on the
|
||||
//! *current* kernel context via the borrowed-thread path — a minimal probe of
|
||||
//! the ring-transition mechanisms, kept for that test.
|
||||
//! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program
|
||||
//! talks to the kernel only through the system_call instruction (or the int 0x80
|
||||
//! gate). The shared handler is installed once by `init`.
|
||||
//!
|
||||
//! Borrowed-path caveat (`run` only): it publishes TSS.rsp0 on the *current*
|
||||
//! core and uses a single global unwind slot (`user_saved_rsp` in isr.s), so the
|
||||
//! caller must disable preemption and only one core may be inside it at a time.
|
||||
//! Real processes (`spawnProcess`) have none of these limits — the scheduler
|
||||
//! maintains rsp0/CR3 per switch.
|
||||
|
||||
const std = @import("std");
|
||||
const elf = std.elf;
|
||||
const danos = @import("danos");
|
||||
const architecture = @import("architecture");
|
||||
const pmm = @import("pmm.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const ipc = @import("ipc-synchronous.zig");
|
||||
const device_service = @import("device-service.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const log = @import("log.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const SystemCall = danos.SystemCall;
|
||||
|
||||
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
||||
/// the identity map (low indices) and the vmm test address (index 128), so
|
||||
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
|
||||
/// An ELF image may occupy [code_virtual, stack_virtual); the stack page sits above.
|
||||
pub const code_virtual: u64 = 0x0000_7000_0000_0000;
|
||||
pub const stack_virtual: u64 = 0x0000_7000_0020_0000;
|
||||
|
||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
|
||||
/// 1 GiB window is far more than any user heap needs today.
|
||||
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
||||
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
||||
|
||||
/// End of the user (low) canonical half. Any legitimate user pointer is below it;
|
||||
/// used to bound the addresses a system_call will dereference on the caller's behalf.
|
||||
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
|
||||
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
||||
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
||||
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
|
||||
/// `Task.device_map_next`.
|
||||
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
|
||||
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||
|
||||
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
|
||||
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
||||
const maximum_mmap_pages = 256;
|
||||
|
||||
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
|
||||
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
|
||||
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
|
||||
|
||||
/// The isolation-proof program: reads a kernel-only page, must #PF.
|
||||
pub fn pfBlob() []const u8 {
|
||||
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
|
||||
}
|
||||
|
||||
/// What debug_write syscalls produced (accumulated), and the exit system_call's code.
|
||||
pub var write_buffer: [256]u8 = undefined;
|
||||
pub var write_len: usize = 0;
|
||||
pub var write_from_user: bool = false;
|
||||
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
|
||||
pub var exit_code: u64 = 0;
|
||||
|
||||
/// The system_call surface, dispatched on the saved system_call number (`danos.SystemCall`).
|
||||
/// This is the microkernel-minimal set — memory + scheduling only; file/device
|
||||
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
|
||||
/// written back into the trap frame, since the entry paths restore user registers
|
||||
/// from it. One handler serves both the system_call/sysret and int-0x80 entry paths.
|
||||
///
|
||||
/// Install it once at boot (before any user code runs) via `init`.
|
||||
pub fn init() void {
|
||||
architecture.setSystemCallHandler(system_call);
|
||||
}
|
||||
|
||||
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
||||
fn fail(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, @bitCast(@as(i64, -1)));
|
||||
}
|
||||
|
||||
fn system_call(state: *architecture.CpuState) void {
|
||||
switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) {
|
||||
.exit => {
|
||||
exit_code = architecture.systemCallArg(state, 0);
|
||||
// A scheduled process drops its endpoint references, frees its address
|
||||
// space, and reschedules; a borrowed test thread unwinds back to the
|
||||
// kernel that entered it.
|
||||
if (scheduler.currentIsUserProcess()) {
|
||||
// Unbind before closeHandles: dropping the last reference destroys the
|
||||
// Endpoint, and a still-bound GSI would have an ISR call
|
||||
// notifyFromIsr on freed memory the next time the device fired.
|
||||
// unbindAll also leaves the line masked, so a dead driver's device
|
||||
// goes quiet rather than storming.
|
||||
releaseIrqs(scheduler.current());
|
||||
ipc.closeHandles(scheduler.current());
|
||||
scheduler.exitUser();
|
||||
} else architecture.userExit();
|
||||
},
|
||||
.yield => {
|
||||
scheduler.yield();
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
},
|
||||
.sleep => {
|
||||
scheduler.sleep(architecture.systemCallArg(state, 0));
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
},
|
||||
.debug_write => systemDebugWrite(state),
|
||||
.mmap => systemMmap(state),
|
||||
.munmap => systemMunmap(state),
|
||||
.create_endpoint => systemCreateEndpoint(state),
|
||||
.ipc_register => systemIpcRegister(state),
|
||||
.ipc_lookup => systemIpcLookup(state),
|
||||
.ipc_call => systemIpcCall(state),
|
||||
.ipc_reply_wait => systemIpcReplyWait(state),
|
||||
.device_enumerate => systemDeviceEnumerate(state),
|
||||
.device_claim => systemDeviceClaim(state),
|
||||
.mmio_map => systemMmioMap(state),
|
||||
.irq_bind => systemIrqBind(state),
|
||||
.irq_ack => systemIrqAck(state),
|
||||
.device_register => systemDeviceRegister(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
|
||||
/// Return `-errno` in the system_call result register.
|
||||
fn failErr(state: *architecture.CpuState, errno: i64) void {
|
||||
architecture.setSystemCallResult(state, @bitCast(-errno));
|
||||
}
|
||||
|
||||
/// create_endpoint() -> handle: allocate an endpoint and install it in the
|
||||
/// caller's handle table.
|
||||
fn systemCreateEndpoint(state: *architecture.CpuState) void {
|
||||
const endpoint = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM);
|
||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||
if (h < 0) {
|
||||
ipc.dropRef(endpoint);
|
||||
return failErr(state, ipc.ENOSPC);
|
||||
}
|
||||
architecture.setSystemCallResult(state, @intCast(h));
|
||||
}
|
||||
|
||||
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
||||
/// well-known id so other processes can find it.
|
||||
fn systemIpcRegister(state: *architecture.CpuState) void {
|
||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
|
||||
}
|
||||
|
||||
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
||||
/// handle to it in the caller.
|
||||
fn systemIpcLookup(state: *architecture.CpuState) void {
|
||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||
if (h < 0) {
|
||||
ipc.dropRef(endpoint);
|
||||
return failErr(state, ipc.ENOSPC);
|
||||
}
|
||||
architecture.setSystemCallResult(state, @intCast(h));
|
||||
}
|
||||
|
||||
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
|
||||
/// Blocks until the server replies; the trap frame lives on this task's kernel
|
||||
/// stack, so it survives the block and receives the result on resume.
|
||||
fn systemIpcCall(state: *architecture.CpuState) void {
|
||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4));
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
}
|
||||
|
||||
/// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len,
|
||||
/// with the sender's badge in the secondary result register (rdx).
|
||||
fn systemIpcReplyWait(state: *architecture.CpuState) void {
|
||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
var badge: u64 = 0;
|
||||
const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), &badge);
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
architecture.setSystemCallResult2(state, badge);
|
||||
}
|
||||
|
||||
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
|
||||
/// buffer (up to `maximum` entries), returning the total device count.
|
||||
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||
const maximum = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
const sz = @sizeOf(danos.DeviceDescriptor);
|
||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||
const out: [*]danos.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
||||
architecture.setSystemCallResult(state, device_service.enumerate(out[0..@intCast(cap)]));
|
||||
}
|
||||
|
||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
if (device_service.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
||||
architecture.setSystemCallResult(state, 0)
|
||||
else
|
||||
fail(state);
|
||||
}
|
||||
|
||||
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
||||
/// this address space (strong-uncacheable) and return the register base address.
|
||||
/// The claim is the capability — a process can only map hardware it owns.
|
||||
fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const resource_index = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const owner = device_service.ownerOf(device_id) orelse return fail(state);
|
||||
if (owner != t.id) return fail(state); // not claimed by this process
|
||||
const r = device_service.resourceOf(device_id, resource_index) orelse return fail(state);
|
||||
if (r.kind != @intFromEnum(danos.ResourceKind.memory)) return fail(state);
|
||||
|
||||
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
||||
const first = r.start & ~@as(u64, page_size - 1);
|
||||
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
|
||||
const pages = (last - first) / page_size + 1;
|
||||
const base_v = t.device_map_next;
|
||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||
|
||||
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len);
|
||||
t.device_map_next = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||
}
|
||||
|
||||
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
||||
/// this process has claimed. The bus-driver primitive: a process that owns a bus
|
||||
/// enumerates it and hands each device it finds to the table, where a class driver
|
||||
/// can claim it.
|
||||
///
|
||||
/// The kernel copies the descriptor into a kernel local *once* (via the same
|
||||
/// physmap-walking path as IPC, so an unmapped user page fails the call rather than
|
||||
/// faulting the kernel), then validates and uses that copy — no second read of user
|
||||
/// memory, so nothing it checked can change under it. It refuses any child resource
|
||||
/// that escapes the parent's windows: a descriptor is a licence to map physical
|
||||
/// memory, so a bus may only subdivide what it already holds. `id`/`parent` in the
|
||||
/// supplied descriptor are ignored.
|
||||
fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||
const parent_id = architecture.systemCallArg(state, 0);
|
||||
const descriptor_ptr = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
|
||||
var descriptor: danos.DeviceDescriptor = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
|
||||
const id = device_service.register(parent_id, t.id, &descriptor) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, id);
|
||||
}
|
||||
|
||||
/// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed
|
||||
/// (which is what frees the endpoints an ISR would otherwise notify into).
|
||||
fn releaseIrqs(t: *scheduler.Task) void {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
irq.releaseOwner(t.id);
|
||||
}
|
||||
|
||||
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
||||
/// The two checks are the whole security story: the device must be *claimed* by the
|
||||
/// caller, and the resource must be one of that device's `irq` resources as recorded
|
||||
/// by discovery. Neither a raw GSI nor an unclaimed device can get through — which
|
||||
/// is why irq_bind takes a resource index and not an interrupt number.
|
||||
fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
||||
const owner = device_service.ownerOf(device_id) orelse return null;
|
||||
if (owner != t.id) return null;
|
||||
const r = device_service.resourceOf(device_id, resource_index) orelse return null;
|
||||
if (r.kind != @intFromEnum(danos.ResourceKind.irq)) return null;
|
||||
if (r.start >= irq.maximum_gsi) return null;
|
||||
return @intCast(r.start);
|
||||
}
|
||||
|
||||
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
|
||||
/// endpoint as an asynchronous IPC notification. The driver then blocks in
|
||||
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
||||
fn systemIrqBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
irq.bind(gsi, endpoint, t.id) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// irq_ack(device_id, resource_index) -> 0/-1: re-arm a bound IRQ. The ISR left the line
|
||||
/// masked (it could not quiet the device — that's this driver's job), so nothing
|
||||
/// more arrives until the driver says it has serviced the hardware.
|
||||
fn systemIrqAck(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (irq.ack(gsi)) architecture.setSystemCallResult(state, 0) else fail(state);
|
||||
}
|
||||
|
||||
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
|
||||
/// A bring-up diagnostic — real output goes through the VFS/console later.
|
||||
///
|
||||
/// The pointer must lie in the user (low) half, so kernel addresses and
|
||||
/// non-canonical values fall outside it and the read below can't be steered at
|
||||
/// kernel data. Length is checked first so the upper-bound add can't overflow.
|
||||
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
|
||||
/// the user half passes the check and the read #PFs -> on_fault halts — a
|
||||
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
|
||||
fn systemDebugWrite(state: *architecture.CpuState) void {
|
||||
const ptr = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
||||
const source: [*]const u8 = @ptrFromInt(ptr);
|
||||
@memcpy(write_buffer[0..len], source[0..len]); // keep the latest message
|
||||
write_len = len;
|
||||
write_from_user = architecture.fromUser(state);
|
||||
write_count += 1;
|
||||
log.write("DANOS-INIT: ");
|
||||
log.write(source[0..len]);
|
||||
architecture.setSystemCallResult(state, len);
|
||||
} else {
|
||||
fail(state);
|
||||
}
|
||||
}
|
||||
|
||||
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
|
||||
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
||||
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
||||
/// always RW+NX; W^X for user code stays with the ELF loader). Failure returns
|
||||
/// -1. The user-space allocator (lib `runtime`) carves these pages into malloc blocks.
|
||||
fn systemMmap(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
|
||||
|
||||
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
|
||||
const base = t.heap_next;
|
||||
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||
|
||||
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
|
||||
// (no partially-mapped grant leaks into the address space).
|
||||
var frames: [maximum_mmap_pages]u64 = undefined;
|
||||
var got: usize = 0;
|
||||
while (got < pages) : (got += 1) {
|
||||
frames[got] = pmm.alloc() orelse {
|
||||
for (frames[0..got]) |f| pmm.free(f);
|
||||
return fail(state);
|
||||
};
|
||||
}
|
||||
|
||||
for (frames[0..pages], 0..) |frame, i| {
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||
architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
|
||||
}
|
||||
t.heap_next = base + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base);
|
||||
}
|
||||
|
||||
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
||||
/// each page and frees its frame. The arena is a bump allocator, so the virtual
|
||||
/// range is not recycled (the user-space allocator reuses freed *blocks* itself);
|
||||
/// this just returns the physical frames to the kernel. Returns 0, or -1 if the
|
||||
/// range is not page-aligned or lies outside the arena.
|
||||
fn systemMunmap(state: *architecture.CpuState) void {
|
||||
const base = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or base % page_size != 0) return fail(state);
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base + i * page_size;
|
||||
if (architecture.translate(t.aspace, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.aspace, va);
|
||||
pmm.free(physical);
|
||||
}
|
||||
}
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// Reset the recorded system_call evidence before a user-mode run.
|
||||
fn resetRecords() void {
|
||||
write_len = 0;
|
||||
write_from_user = false;
|
||||
write_count = 0;
|
||||
exit_code = 0;
|
||||
}
|
||||
|
||||
pub const RunError = error{ ProgramTooBig, OutOfMemory };
|
||||
|
||||
/// Map `blob` at code_virtual with a fresh user stack, drop to ring 3, and return
|
||||
/// once the program exits via system_call 0. See the migration caveat in the module
|
||||
/// doc. A program that faults instead never returns (on_fault halts the core).
|
||||
pub fn run(blob: []const u8) RunError!void {
|
||||
if (blob.len > page_size) return error.ProgramTooBig;
|
||||
const code_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const stack_frame = pmm.alloc() orelse {
|
||||
pmm.free(code_frame);
|
||||
return error.OutOfMemory;
|
||||
};
|
||||
|
||||
// Fill the code frame through the physmap (supervisor RW): the user-facing
|
||||
// mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is
|
||||
// padded with int3 so a stray jump traps instead of sliding.
|
||||
const code: [*]u8 = @ptrFromInt(danos.physicalToVirtual(code_frame));
|
||||
@memcpy(code[0..blob.len], blob);
|
||||
@memset(code[blob.len..page_size], 0xCC);
|
||||
|
||||
architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X
|
||||
architecture.mapUserPage(stack_virtual, stack_frame, true, false); // RW + NX
|
||||
resetRecords();
|
||||
|
||||
architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_virtual + page_size);
|
||||
|
||||
// Back via the exit system_call; the interrupt gate left IF clear.
|
||||
architecture.enableInterrupts();
|
||||
architecture.unmapPage(code_virtual);
|
||||
architecture.unmapPage(stack_virtual);
|
||||
pmm.free(code_frame);
|
||||
pmm.free(stack_frame);
|
||||
}
|
||||
|
||||
// --- user ELF loading (/sbin/init) ------------------------------------------
|
||||
|
||||
pub const InitError = error{
|
||||
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
|
||||
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
|
||||
BadEntry, // e_entry not inside an executable segment
|
||||
ProgramTooBig, // more pages than the loader's budget
|
||||
OutOfMemory,
|
||||
};
|
||||
|
||||
const maximum_segments = 16;
|
||||
const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
||||
|
||||
const Segment = struct {
|
||||
vaddr: u64,
|
||||
memsz: u64,
|
||||
filesz: u64,
|
||||
off: u64,
|
||||
writable: bool,
|
||||
executable: bool,
|
||||
|
||||
fn pages(self: Segment) u64 {
|
||||
return (self.memsz + page_size - 1) / page_size;
|
||||
}
|
||||
};
|
||||
|
||||
/// Parse and validate every PT_LOAD before touching memory. Bounds are checked
|
||||
/// against the image and the user region; segments must be page-aligned,
|
||||
/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only
|
||||
/// segment, mapped RO+NX).
|
||||
fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!struct { count: usize, entry: u64 } {
|
||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf;
|
||||
const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]);
|
||||
if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf;
|
||||
if (image[elf.EI_CLASS] != elf.ELFCLASS64) return error.BadElf;
|
||||
if (ehdr.e_machine != .X86_64) return error.BadElf;
|
||||
if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation
|
||||
if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf;
|
||||
if (ehdr.e_phnum > maximum_segments) return error.BadElf;
|
||||
const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize;
|
||||
if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf;
|
||||
|
||||
var count: usize = 0;
|
||||
var total_pages: u64 = 0;
|
||||
for (0..ehdr.e_phnum) |i| {
|
||||
const off = ehdr.e_phoff + i * ehdr.e_phentsize;
|
||||
const phdr = std.mem.bytesToValue(elf.Elf64_Phdr, image[off..][0..@sizeOf(elf.Elf64_Phdr)]);
|
||||
if (phdr.p_type != elf.PT_LOAD) continue;
|
||||
if (phdr.p_memsz == 0) continue;
|
||||
|
||||
if (phdr.p_vaddr % page_size != 0) return error.BadSegment;
|
||||
if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment;
|
||||
if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment;
|
||||
// Inside the user image region, strictly below the stack page.
|
||||
if (phdr.p_vaddr < code_virtual) return error.BadSegment;
|
||||
if (phdr.p_memsz > stack_virtual - phdr.p_vaddr) return error.BadSegment;
|
||||
|
||||
const w = phdr.p_flags & elf.PF_W != 0;
|
||||
const x = phdr.p_flags & elf.PF_X != 0;
|
||||
if (w and x) return error.BadSegment; // W^X, even for init
|
||||
|
||||
const seg = Segment{
|
||||
.vaddr = phdr.p_vaddr,
|
||||
.memsz = phdr.p_memsz,
|
||||
.filesz = phdr.p_filesz,
|
||||
.off = phdr.p_offset,
|
||||
.writable = w,
|
||||
.executable = x,
|
||||
};
|
||||
// No overlap with any earlier segment (page-granular, since mapping is).
|
||||
for (segs[0..count]) |other| {
|
||||
const a_end = seg.vaddr + seg.pages() * page_size;
|
||||
const b_end = other.vaddr + other.pages() * page_size;
|
||||
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
|
||||
}
|
||||
total_pages += seg.pages();
|
||||
if (total_pages > maximum_pages) return error.ProgramTooBig;
|
||||
segs[count] = seg;
|
||||
count += 1;
|
||||
}
|
||||
if (count == 0) return error.BadElf;
|
||||
|
||||
// The entry point must land inside an executable segment.
|
||||
for (segs[0..count]) |seg| {
|
||||
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
|
||||
return .{ .count = count, .entry = ehdr.e_entry };
|
||||
}
|
||||
return error.BadEntry;
|
||||
}
|
||||
|
||||
/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed
|
||||
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
||||
/// On a later failure the whole address space is torn down, which frees every
|
||||
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0);
|
||||
const page_off = page_index * page_size;
|
||||
if (page_off < seg.filesz) {
|
||||
const n = @min(page_size, seg.filesz - page_off);
|
||||
@memcpy(destination[0..n], image[seg.off + page_off ..][0..n]);
|
||||
}
|
||||
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
||||
}
|
||||
|
||||
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
||||
/// ring-3 process at `priority`. Returns immediately — the process runs
|
||||
/// preemptively on its own page tables alongside everything else, and its exit
|
||||
/// is handled by the system_call layer. The whole build (address space + ELF load +
|
||||
/// task) runs under the kernel lock so it appears atomically and can't race
|
||||
/// pmm/heap on another core.
|
||||
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
|
||||
var segs: [maximum_segments]Segment = undefined;
|
||||
const parsed = try parseSegments(image, &segs);
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
||||
errdefer architecture.destroyAddressSpace(aspace);
|
||||
|
||||
for (segs[0..parsed.count]) |seg| {
|
||||
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
||||
}
|
||||
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
architecture.mapUserPageInto(aspace, stack_virtual, stack_frame, true, false); // RW + NX
|
||||
|
||||
if (!scheduler.spawnUserLocked(aspace, parsed.entry, stack_virtual + page_size, priority))
|
||||
return error.OutOfMemory;
|
||||
}
|
||||
@@ -0,0 +1,561 @@
|
||||
//! The scheduler: fixed-priority preemptive multitasking.
|
||||
//!
|
||||
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
||||
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
||||
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
||||
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
||||
//! docs/vision.md).
|
||||
//!
|
||||
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
||||
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
||||
//! makes those two paths coexist.
|
||||
//!
|
||||
//! Cross-core safety is the **big kernel lock** (`sync.zig`): every critical
|
||||
//! section here runs under it, and it is held across a context switch and released
|
||||
//! by the task that resumes (see sync.zig's hand-off rule). On a single core the
|
||||
//! lock is never contended, so the behaviour is exactly the old interrupt-flag
|
||||
//! model; it's what lets a second core enter `schedule()` without corrupting the
|
||||
//! shared queues.
|
||||
|
||||
const std = @import("std");
|
||||
const parameters = @import("parameters");
|
||||
const architecture = @import("architecture");
|
||||
const heap = @import("heap.zig");
|
||||
const sync = @import("sync.zig");
|
||||
|
||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
||||
pub const Priority = u3;
|
||||
const number_priorities = 8;
|
||||
|
||||
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
|
||||
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
|
||||
|
||||
const State = enum { free, ready, running, blocked };
|
||||
|
||||
pub const Task = struct {
|
||||
id: u32 = 0,
|
||||
state: State = .free,
|
||||
priority: Priority = 0,
|
||||
sp: usize = 0, // saved stack pointer, valid while not running
|
||||
stack: []u8 = &.{},
|
||||
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
|
||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
||||
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
|
||||
// Physical root of this task's address space, or 0 for a kernel task (which
|
||||
// runs on the shared kernel page tables). A user task carries its own.
|
||||
aspace: u64 = 0,
|
||||
user_ip: u64 = 0, // user-mode entry point (user task only)
|
||||
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
||||
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
|
||||
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
||||
// as the user heap grows; user task only.
|
||||
heap_next: u64 = 0,
|
||||
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
|
||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
||||
device_map_next: u64 = 0,
|
||||
// --- synchronous IPC (ipc_sync.zig) ---
|
||||
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
||||
// opaque here so the scheduler and IPC modules don't import each other.
|
||||
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
|
||||
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||
ipc_client: ?*Task = null,
|
||||
ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS)
|
||||
ipc_send_len: u64 = 0,
|
||||
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
|
||||
ipc_reply_cap: u64 = 0,
|
||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||
};
|
||||
|
||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
pub const ipc_maximum_handles = 16;
|
||||
|
||||
var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||
var next_id: u32 = 1;
|
||||
|
||||
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
||||
/// queue of tasks **pinned** to it. One entry per core; the architecture layer stashes a
|
||||
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
|
||||
/// with a single read and no lock.
|
||||
///
|
||||
/// Most work stays in the **global** ready queue (below), which any idle core pulls
|
||||
/// from — work-conserving. A task given an *affinity* instead goes to that core's
|
||||
/// `pinned_*` queue and is only ever run there (no surprise migration — the more
|
||||
/// real-time-predictable model, docs/smp.md). The two queues are merged at selection
|
||||
/// time. Both are still mutated only under the big kernel lock, so one core enqueuing
|
||||
/// into another core's pinned queue is safe.
|
||||
pub const PerCpu = struct {
|
||||
current: *Task = undefined, // the task running on this core
|
||||
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
||||
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
|
||||
index: u32 = 0, // dense 0-based core index
|
||||
online: bool = false, // has this core finished bring-up?
|
||||
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
||||
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||
pinned_bitmap: u8 = 0,
|
||||
};
|
||||
|
||||
const maximum_cpus = parameters.maximum_cpus;
|
||||
var cpus = [_]PerCpu{.{}} ** maximum_cpus;
|
||||
|
||||
/// This core's per-CPU state, via the architecture layer's GS-base pointer. Valid only once
|
||||
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
|
||||
inline fn thisCpu() *PerCpu {
|
||||
return @ptrFromInt(architecture.cpuLocal());
|
||||
}
|
||||
|
||||
/// The task running on this core — the per-CPU replacement for the old global
|
||||
/// `current`. A convenience reader; writes go through `thisCpu().current`.
|
||||
pub inline fn current() *Task {
|
||||
return thisCpu().current;
|
||||
}
|
||||
|
||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
|
||||
// are shared across all cores and mutated only under the big kernel lock.
|
||||
var ready_head: [number_priorities]?*Task = .{null} ** number_priorities;
|
||||
var ready_tail: [number_priorities]?*Task = .{null} ** number_priorities;
|
||||
var ready_bitmap: u8 = 0;
|
||||
|
||||
var preemption_enabled = true;
|
||||
|
||||
/// Bring up scheduling on the bootstrap processor: register the currently-running
|
||||
/// kernel context as task 0, publish this core's per-CPU state (via the GS base),
|
||||
/// give the core an idle task, and hook the timer for preemption. Runs once, at
|
||||
/// boot, before interrupts are enabled — so no lock is needed here.
|
||||
pub fn init(boot_priority: Priority) void {
|
||||
const pc = &cpus[0];
|
||||
pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() };
|
||||
architecture.setCpuLocal(0, @intFromPtr(pc));
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
pc.current = &tasks[0];
|
||||
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
|
||||
architecture.setTickHook(tick);
|
||||
}
|
||||
|
||||
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
||||
/// for the next interrupt at near-zero power (see docs/halting.md).
|
||||
fn idle() void {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
/// Reserve and initialise the per-CPU slot for an application processor at dense
|
||||
/// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a
|
||||
/// pointer the architecture bring-up hands to the core (it publishes it in its GS base).
|
||||
/// Called on the BSP before waking each AP; the AP marks itself `online`.
|
||||
pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu {
|
||||
const pc = &cpus[index];
|
||||
pc.* = .{ .index = @intCast(index), .hw_id = hw_id, .online = false };
|
||||
return pc;
|
||||
}
|
||||
|
||||
/// Entry for an application processor once the architecture layer has set up its per-CPU
|
||||
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
|
||||
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
|
||||
/// interrupts enabled the timer preempts this idle context into whatever the global
|
||||
/// ready queue offers, so the core runs real work in parallel with the others. The
|
||||
/// `.c` calling convention lets the architecture trampoline path jump here. Never returns.
|
||||
pub fn secondaryMain() callconv(.c) noreturn {
|
||||
const flags = sync.enter();
|
||||
const pc = thisCpu();
|
||||
const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");
|
||||
t.* = .{ .id = next_id, .state = .running, .priority = 0 };
|
||||
next_id += 1;
|
||||
pc.current = t;
|
||||
pc.idle = t;
|
||||
pc.online = true;
|
||||
pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||
sync.leave(flags);
|
||||
|
||||
architecture.enableInterrupts(); // the timer now preempts this idle context into work
|
||||
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
|
||||
}
|
||||
|
||||
/// Number of cores that have finished bring-up (the BSP plus every online AP).
|
||||
pub fn onlineCount() usize {
|
||||
var n: usize = 0;
|
||||
for (&cpus) |*pc| {
|
||||
if (pc.online) n += 1;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue;
|
||||
/// everything else goes to the shared global queue.
|
||||
fn enqueue(t: *Task) void {
|
||||
if (t.affinity) |cpu| {
|
||||
const pc = &cpus[cpu];
|
||||
enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
|
||||
} else {
|
||||
enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t);
|
||||
}
|
||||
}
|
||||
|
||||
fn enqueueTo(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, t: *Task) void {
|
||||
t.next = null;
|
||||
const p: usize = t.priority;
|
||||
if (tail[p]) |tl| tl.next = t else head[p] = t;
|
||||
tail[p] = t;
|
||||
bitmap.* |= @as(u8, 1) << t.priority;
|
||||
}
|
||||
|
||||
/// The highest non-empty priority level in a bitmap, or -1 if empty.
|
||||
fn topLevel(bitmap: u8) i32 {
|
||||
if (bitmap == 0) return -1;
|
||||
return @as(i32, number_priorities - 1) - @as(i32, @clz(bitmap));
|
||||
}
|
||||
|
||||
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
|
||||
/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task
|
||||
/// wins an equal-priority tie, so it can't be starved by global work at its level.
|
||||
fn dequeueHighest(pc: *PerCpu) ?*Task {
|
||||
const g = topLevel(ready_bitmap);
|
||||
const p = topLevel(pc.pinned_bitmap);
|
||||
if (g < 0 and p < 0) return null;
|
||||
if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p));
|
||||
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
|
||||
}
|
||||
|
||||
fn dequeueFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
|
||||
const t = head[level].?;
|
||||
head[level] = t.next;
|
||||
if (head[level] == null) {
|
||||
tail[level] = null;
|
||||
bitmap.* &= ~(@as(u8, 1) << @intCast(level));
|
||||
}
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
/// Create a task that runs `entry` at `priority`, runnable on any core. It becomes
|
||||
/// ready immediately. Takes the kernel lock: it mutates the shared task table and
|
||||
/// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be
|
||||
/// serialised.
|
||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
||||
const flags = sync.enter();
|
||||
_ = create(entry, priority, null);
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there.
|
||||
/// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case
|
||||
/// the task is still created but left unpinned (so it runs *somewhere* rather than
|
||||
/// stranding in a queue no core services). Callers that require the pin (e.g. tests)
|
||||
/// should check the result.
|
||||
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const ok = cpu < maximum_cpus and cpus[cpu].online;
|
||||
_ = create(entry, priority, if (ok) cpu else null);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
||||
/// in user mode at `entry` on `user_sp`. It gets a fresh kernel stack for
|
||||
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
|
||||
/// Returns false (creating nothing) if the table is full or out of memory.
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
||||
/// across the whole spawn, so the address space and the task appear atomically).
|
||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority) bool {
|
||||
const t = freeSlot() orelse return false;
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
||||
t.* = .{
|
||||
.id = next_id,
|
||||
.state = .ready,
|
||||
.priority = priority,
|
||||
.stack = stack,
|
||||
.aspace = aspace,
|
||||
.user_ip = entry,
|
||||
.user_sp = user_sp,
|
||||
};
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.kstack_top = top;
|
||||
// First switch-in lands in startUserTask (no register smuggling — it reads
|
||||
// the user entry/stack from the Task itself).
|
||||
t.sp = architecture.initTaskStack(top, @intFromPtr(&startUserTask));
|
||||
enqueue(t);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It
|
||||
/// drops to ring 3 at the task's recorded entry/stack. Reading them from the
|
||||
/// Task avoids smuggling values through callee-saved registers across the
|
||||
/// context switch and lock release.
|
||||
fn startUserTask() void {
|
||||
const t = current();
|
||||
var buffer: [96]u8 = undefined;
|
||||
architecture.serialWrite(std.fmt.bufPrint(&buffer, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
|
||||
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
||||
}
|
||||
|
||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
|
||||
/// Returns the new task so a core can keep a handle to its idle task.
|
||||
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
||||
const t = freeSlot() orelse @panic("sched: task table full");
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.kstack_top = top;
|
||||
t.sp = architecture.initTaskStack(top, @intFromPtr(entry));
|
||||
enqueue(t);
|
||||
return t;
|
||||
}
|
||||
|
||||
fn freeSlot() ?*Task {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .free) return t;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Pick the highest-priority ready task and switch this core to it. The big kernel
|
||||
/// lock must be held by the caller (which also keeps local interrupts disabled);
|
||||
/// it serialises every core's scheduling, so no other core can touch the shared
|
||||
/// queues while we requeue `previous` and dequeue `next`. A dequeued task is `.ready`,
|
||||
/// never running elsewhere, so two cores never run the same task.
|
||||
fn schedule() void {
|
||||
const pc = thisCpu();
|
||||
const previous = pc.current;
|
||||
if (previous.state == .running) {
|
||||
previous.state = .ready;
|
||||
enqueue(previous); // back of its level's queue (round-robin)
|
||||
}
|
||||
const next = dequeueHighest(pc) orelse {
|
||||
previous.state = .running; // nothing else ready — keep running
|
||||
return;
|
||||
};
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
if (next != previous) switchTo(pc, &previous.sp, next);
|
||||
}
|
||||
|
||||
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
||||
/// user-mode interrupt lands on a good stack) and its address space (only when
|
||||
/// it differs from what's loaded — every page-table switch is a full TLB flush),
|
||||
/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used
|
||||
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
|
||||
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
|
||||
/// workload. The big kernel lock is held and interrupts are off throughout, so no
|
||||
/// interrupt can observe a half-updated (kernel stack, address space) pair.
|
||||
/// `save_sp` receives the outgoing task's stack pointer.
|
||||
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
||||
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
|
||||
const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable();
|
||||
if (want != pc.loaded_aspace) {
|
||||
architecture.loadPageTable(want);
|
||||
pc.loaded_aspace = want;
|
||||
}
|
||||
architecture.switchContext(save_sp, next.sp);
|
||||
}
|
||||
|
||||
/// Voluntarily give up the CPU to the next ready task.
|
||||
pub fn yield() void {
|
||||
const flags = sync.enter();
|
||||
schedule();
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// Block the current task for `ms` milliseconds, then let it become runnable
|
||||
/// again. The idle task (or other work) runs in the meantime.
|
||||
pub fn sleep(ms: u64) void {
|
||||
const flags = sync.enter();
|
||||
const t = current();
|
||||
t.wake_at = architecture.millis() + ms;
|
||||
t.state = .blocked;
|
||||
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
// --- event-based blocking -------------------------------------------------
|
||||
//
|
||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
||||
// message). Tasks link into it through the same `next` field the ready queues
|
||||
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
||||
// semaphores and IPC channels are built on.
|
||||
|
||||
pub const WaitQueue = struct {
|
||||
head: ?*Task = null,
|
||||
};
|
||||
|
||||
/// Block the current task on `wait_queue` and switch away. Precondition: the big kernel
|
||||
/// lock is held (so a condition can be checked and the block committed atomically;
|
||||
/// it also keeps local interrupts disabled). On return — when woken — the lock is
|
||||
/// still held.
|
||||
pub fn waitLocked(wait_queue: *WaitQueue) void {
|
||||
const t = current();
|
||||
t.state = .blocked;
|
||||
t.next = wait_queue.head;
|
||||
wait_queue.head = t;
|
||||
schedule();
|
||||
}
|
||||
|
||||
/// Move the highest-priority waiter on `wait_queue` (if any) to the ready queue.
|
||||
/// Precondition: the big kernel lock is held. Does not preempt — the caller decides.
|
||||
pub fn wakeLocked(wait_queue: *WaitQueue) void {
|
||||
// Find the highest-priority waiter (bounded scan) and unlink it.
|
||||
var best_previous: ?*Task = null;
|
||||
var best: ?*Task = null;
|
||||
var previous: ?*Task = null;
|
||||
var node = wait_queue.head;
|
||||
while (node) |t| : ({
|
||||
previous = t;
|
||||
node = t.next;
|
||||
}) {
|
||||
if (best == null or t.priority > best.?.priority) {
|
||||
best = t;
|
||||
best_previous = previous;
|
||||
}
|
||||
}
|
||||
const t = best orelse return;
|
||||
if (best_previous) |p| p.next = t.next else wait_queue.head = t.next;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Block the current task and switch away, without putting it on any wait queue —
|
||||
/// the caller has already linked it wherever it belongs (e.g. an endpoint's sender
|
||||
/// FIFO). Precondition: the big kernel lock is held; still held on return (when the
|
||||
/// task is made ready again). The IPC layer's counterpart to `waitLocked`.
|
||||
pub fn blockCurrentLocked() void {
|
||||
current().state = .blocked;
|
||||
schedule();
|
||||
}
|
||||
|
||||
/// Make a specific (currently blocked) task ready to run again. Precondition: the
|
||||
/// big kernel lock is held. Used by the IPC layer to wake a specific caller/server
|
||||
/// rather than "some waiter on a queue".
|
||||
pub fn readyLocked(t: *Task) void {
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Block on `wait_queue` (a self-contained critical section).
|
||||
pub fn wait(wait_queue: *WaitQueue) void {
|
||||
const flags = sync.enter();
|
||||
waitLocked(wait_queue);
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// Wake the highest-priority waiter on `wait_queue`, preempting if it outranks us.
|
||||
pub fn wake(wait_queue: *WaitQueue) void {
|
||||
const flags = sync.enter();
|
||||
const pc = thisCpu();
|
||||
wakeLocked(wait_queue);
|
||||
// If a task this core would now pick outranks the running one, run it at once.
|
||||
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
|
||||
// next tick; this core doesn't preempt for work it can't run.)
|
||||
if (highestReadyPriority(pc)) |p| {
|
||||
if (p > pc.current.priority) schedule();
|
||||
}
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// The highest-priority task core `pc` could run right now — the better of the global
|
||||
/// queue and this core's pinned queue — or null if it would fall back to idle.
|
||||
fn highestReadyPriority(pc: *PerCpu) ?Priority {
|
||||
const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap));
|
||||
if (top < 0) return null;
|
||||
return @intCast(top);
|
||||
}
|
||||
|
||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
||||
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
||||
fn wakeExpired() void {
|
||||
const now = architecture.millis();
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
||||
t.wake_at = 0;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Called from the timer interrupt (interrupts already disabled): wake due
|
||||
/// sleepers, then preempt. Takes the kernel lock like any other critical section,
|
||||
/// but releases it *without* touching the interrupt flag — the handler's `iretq`
|
||||
/// restores the interrupted context's flags, so re-enabling here would open a
|
||||
/// nested-interrupt window before the return.
|
||||
pub fn tick() void {
|
||||
_ = sync.enter();
|
||||
wakeExpired();
|
||||
if (preemption_enabled) schedule();
|
||||
sync.leaveIsr();
|
||||
}
|
||||
|
||||
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
||||
pub fn setPreemption(enabled: bool) void {
|
||||
preemption_enabled = enabled;
|
||||
}
|
||||
|
||||
/// End the current task and switch away for good; never returns. The task's stack
|
||||
/// is leaked for now (no reaper yet). Acquires the kernel lock and hands it off to
|
||||
/// the task we switch into (which releases it) — this frame never returns to leave.
|
||||
pub fn exit() noreturn {
|
||||
_ = sync.enter();
|
||||
const pc = thisCpu();
|
||||
pc.current.state = .free;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
var discard: usize = 0;
|
||||
switchTo(pc, &discard, next);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// End the current **user** task: free its address space, then exit. Runs on the
|
||||
/// dying task's kernel stack (in the shared kernel half, so it survives the CR3
|
||||
/// switch to the kernel tables that must happen before we free the process's own
|
||||
/// tables — we can't free the page tables we're standing on). The kernel stack
|
||||
/// itself is leaked, as in `exit` (no reaper yet). Never returns.
|
||||
pub fn exitUser() noreturn {
|
||||
_ = sync.enter();
|
||||
const pc = thisCpu();
|
||||
const dying = pc.current;
|
||||
const as = dying.aspace;
|
||||
if (as != 0) {
|
||||
const kroot = architecture.kernelPageTable();
|
||||
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
||||
pc.loaded_aspace = kroot;
|
||||
architecture.destroyAddressSpace(as);
|
||||
}
|
||||
dying.state = .free;
|
||||
dying.aspace = 0;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
var discard: usize = 0;
|
||||
switchTo(pc, &discard, next);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Whether the running task is a user process (has its own address space).
|
||||
pub fn currentIsUserProcess() bool {
|
||||
return current().aspace != 0;
|
||||
}
|
||||
|
||||
pub fn currentId() u32 {
|
||||
return current().id;
|
||||
}
|
||||
|
||||
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
||||
/// per-CPU state, so a task calling it on different cores sees different values —
|
||||
/// which is how a test can prove work is running in parallel. Returns 0 if the GS
|
||||
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
|
||||
/// reporter can call it unconditionally without a second fault.
|
||||
pub fn currentCpuIndex() u32 {
|
||||
if (architecture.cpuLocal() == 0) return 0;
|
||||
return thisCpu().index;
|
||||
}
|
||||
|
||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||
pub fn setPriority(p: Priority) void {
|
||||
current().priority = p;
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
//! The big kernel lock (BKL) — the coarse mutual exclusion that lets more than one
|
||||
//! CPU run kernel code safely.
|
||||
//!
|
||||
//! Until SMP, the kernel's mutual exclusion *was* the interrupt flag: a critical
|
||||
//! section did `cli`, and since only one core existed, nothing else could touch
|
||||
//! kernel state (the discipline in docs/scheduling.md). That invariant dies the
|
||||
//! instant a second core runs kernel code — `cli` on one core does nothing to
|
||||
//! another. So the kernel's shared state (the scheduler queues, IPC channels) is
|
||||
//! guarded by a spinlock, and the lock is **always held with local interrupts
|
||||
//! disabled**, so a core's own timer interrupt can't re-enter the kernel and
|
||||
//! deadlock against the lock it already holds.
|
||||
//!
|
||||
//! This is deliberately *one coarse lock*, not many fine ones: it's philosophically
|
||||
//! aligned with a tiny kernel and it keeps the single-core correctness model
|
||||
//! (docs/scheduling.md) largely intact — one lock around kernel entry instead of
|
||||
//! rethinking every critical section. It's the first-design choice seL4 makes and
|
||||
//! docs/smp.md endorses; per-core run queues + fine-grained locking come later, if
|
||||
//! contention ever bites. Because the kernel does little, the lock is held briefly.
|
||||
//!
|
||||
//! **The hand-off rule.** The lock is held *across* a context switch and released
|
||||
//! by whichever task resumes, not by the one that switched away. A task that blocks
|
||||
//! or yields calls `enter`, mutates the queues, `schedule()`s — switching to another
|
||||
//! task *with the lock still held* — and only calls `leave` once it is eventually
|
||||
//! resumed and its critical section runs to the end. So every call into `schedule()`
|
||||
//! (and thus `switch_context`) happens with the lock held, and every task resumes
|
||||
//! from a switch holding it. A freshly-spawned task has no `enter`/`leave` frame to
|
||||
//! resume into, so `task_trampoline` releases the lock explicitly on its behalf via
|
||||
//! `releaseForFreshTask` before running the task body.
|
||||
|
||||
const std = @import("std");
|
||||
const architecture = @import("architecture");
|
||||
|
||||
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
||||
var held = std.atomic.Value(u32).init(0);
|
||||
|
||||
/// Enter the kernel: disable interrupts on this core, then spin until we own the
|
||||
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
|
||||
/// Interrupts stay off for the whole critical section so this core's timer tick
|
||||
/// can't try to re-acquire the lock we're holding.
|
||||
pub fn enter() u64 {
|
||||
const flags = architecture.saveInterrupts();
|
||||
acquire();
|
||||
return flags;
|
||||
}
|
||||
|
||||
/// Release the lock and restore the interrupt flags `enter` returned (re-enabling
|
||||
/// interrupts only if they were on beforehand). The normal exit for a critical
|
||||
/// section reached from task context (`yield`, `sleep`, `wait`, `wake`, IPC).
|
||||
pub fn leave(flags: u64) void {
|
||||
release();
|
||||
architecture.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// Release the lock but leave interrupts as they are. The exit for a critical
|
||||
/// section running inside an interrupt handler (the timer `tick`): the handler's
|
||||
/// `iretq` is what restores the interrupted context's flags, so restoring them
|
||||
/// here too would open a nested-interrupt window before the return. Release only.
|
||||
pub fn leaveIsr() void {
|
||||
release();
|
||||
}
|
||||
|
||||
/// Release the lock on behalf of a freshly-spawned task. Such a task is switched to
|
||||
/// (with the lock held) but has no `enter`/`leave` frame of its own to release
|
||||
/// through — `task_trampoline` calls this before running the task body. Interrupts
|
||||
/// are enabled separately by the trampoline. Exported for the assembly trampoline.
|
||||
export fn releaseForFreshTask() callconv(.c) void {
|
||||
release();
|
||||
}
|
||||
|
||||
fn acquire() void {
|
||||
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
||||
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
||||
while (held.swap(1, .acquire) != 0) {
|
||||
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
||||
}
|
||||
}
|
||||
|
||||
fn release() void {
|
||||
held.store(0, .release);
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,30 @@
|
||||
//! Kernel tunables — the compile-time knobs, gathered in one place.
|
||||
//!
|
||||
//! These constants would otherwise be scattered across the files that use them,
|
||||
//! hiding the trade-offs. Keeping them here makes them visible at a glance and gives
|
||||
//! one spot to change them. They're plain `comptime` constants (zero runtime cost);
|
||||
//! any one can later be promoted to a `-D` build option if a target needs to vary it
|
||||
//! (see build.zig's `-Dtest-case` for the pattern). This keeps root.zig to what it
|
||||
//! actually is — the bootloader↔kernel handoff *contract* — with tunables living here.
|
||||
|
||||
/// Ceiling on logical CPUs the kernel tracks — the size of the per-CPU bookkeeping
|
||||
/// arrays (discovery pool, scheduler state, per-core GDT/TSS). Generous headroom:
|
||||
/// those structs are small, and the *large* per-core resources (kernel and IST
|
||||
/// stacks) are allocated at bring-up for cores that actually come online, so this
|
||||
/// ceiling is cheap. A machine with more logical CPUs has its surplus reported and
|
||||
/// left parked (see acpi `cpusDropped`).
|
||||
pub const maximum_cpus = 128;
|
||||
|
||||
/// Maximum tasks (kernel threads) alive at once — the static task-table size. Each
|
||||
/// online core consumes one slot for its idle task, plus task 0 on the BSP.
|
||||
pub const maximum_tasks = 16;
|
||||
|
||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||
pub const kernel_stack_size = 16 * 1024;
|
||||
|
||||
/// Each core's IST (double-fault) stack, in bytes. The BSP's is static; an AP's is
|
||||
/// heap-allocated at bring-up.
|
||||
pub const ist_stack_size = 16 * 1024;
|
||||
|
||||
/// Scheduler tick / preemption rate, in Hz (the timer's periodic frequency).
|
||||
pub const timer_hz = 1000;
|
||||
@@ -0,0 +1,39 @@
|
||||
//! /sbin/init — the first user-space program, PID 1. Built as its own
|
||||
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
|
||||
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
|
||||
//! kernel (system/kernel/process.zig). It links against the shared user runtime
|
||||
//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers.
|
||||
//!
|
||||
//! Today it proves the C-convention heap works, then settles into a heartbeat:
|
||||
//! it prints a line and sleeps, forever — enough to show the system reaches user
|
||||
//! space and stays alive with a real process scheduled alongside the kernel's
|
||||
//! idle loop. It grows into the real init (service supervision) once there are
|
||||
//! other user programs to supervise.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
|
||||
pub fn main() void {
|
||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||
// mmaps pages from the kernel and carves them with the free list), write into
|
||||
// that heap buffer (exercising the widened debug_write bounds check), and
|
||||
// free it. A fault here would kill init before it heartbeats — so the init
|
||||
// test doubles as the heap regression test. (C code links the same heap via
|
||||
// the extern malloc/free symbols; Zig code uses this allocator.)
|
||||
const gpa = runtime.allocator();
|
||||
if (gpa.alloc(u8, 64)) |buffer| {
|
||||
const message = "init: heap ok\n";
|
||||
@memcpy(buffer[0..message.len], message);
|
||||
_ = runtime.system.write(buffer[0..message.len]);
|
||||
gpa.free(buffer);
|
||||
} else |_| {}
|
||||
|
||||
while (true) {
|
||||
_ = runtime.system.write("init: heartbeat\n");
|
||||
runtime.system.sleep(1000);
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
//! The VFS wire protocol — the message format spoken between a client (via the
|
||||
//! `runtime` file API) and the user-space VFS server over IPC. A request is a fixed
|
||||
//! `Request` header followed by an inline payload (a path, or write bytes); a
|
||||
//! reply is a fixed `Reply` header followed by an inline payload (read bytes, or
|
||||
//! a Stat). Everything fits in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes).
|
||||
//!
|
||||
//! This is user-space only — the kernel knows nothing of files or paths; it only
|
||||
//! moves the bytes. Shared by library/runtime/unistd.zig (client) and system/services/vfs/vfs.zig (server).
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
open, // open(path) -> node id
|
||||
close, // close(node)
|
||||
read, // read(node, offset, len) -> bytes
|
||||
write, // write(node, offset, bytes) -> count
|
||||
stat, // stat(node) -> Stat
|
||||
};
|
||||
|
||||
/// Request header. `node` is the server-side open-file id (from a prior open);
|
||||
/// for `open` the path is the payload and `len` is its length. `offset`/`len`
|
||||
/// carry the read/write position and count.
|
||||
pub const Request = extern struct {
|
||||
operation: Operation,
|
||||
node: u64,
|
||||
offset: u64,
|
||||
len: u32,
|
||||
flags: u32,
|
||||
};
|
||||
|
||||
/// Reply header. `status` is 0 on success or a negative errno; `node` is the new
|
||||
/// open-file id (for `open`); `len` is the payload length (bytes read, or the
|
||||
/// Stat size).
|
||||
pub const Reply = extern struct {
|
||||
status: i32,
|
||||
_padding: u32 = 0,
|
||||
node: u64 = 0,
|
||||
len: u32 = 0,
|
||||
_padding2: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Stat = extern struct {
|
||||
size: u64,
|
||||
kind: u32,
|
||||
_padding: u32 = 0,
|
||||
};
|
||||
|
||||
pub const message_maximum: usize = 256;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
/// Largest inline payload that still fits one IPC message alongside a header.
|
||||
pub const maximum_payload: usize = message_maximum - request_size;
|
||||
|
||||
/// Open flags.
|
||||
pub const O_CREAT: u32 = 1;
|
||||
@@ -0,0 +1,47 @@
|
||||
//! /sbin/vfstest — a client that proves the VFS round trip end to end: open a
|
||||
//! file through the `runtime` file API, write to it, seek back, read it, and compare.
|
||||
//! On success it heartbeats "vfstest: ok" so the kernel test can observe it;
|
||||
//! on failure it reports what went wrong. Shipped in the initrd alongside vfs.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
|
||||
pub fn main() void {
|
||||
const u = runtime.unistd;
|
||||
const payload = "hello-vfs";
|
||||
|
||||
// The VFS server may not have registered yet — retry open until it's up.
|
||||
var fd: i32 = -1;
|
||||
var tries: u32 = 0;
|
||||
while (fd < 0 and tries < 200) : (tries += 1) {
|
||||
fd = u.open("greeting", u.O_CREAT);
|
||||
if (fd < 0) runtime.system.sleep(20);
|
||||
}
|
||||
if (fd < 0) {
|
||||
_ = runtime.system.write("vfstest: open failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
if (u.write(fd, payload) != @as(isize, payload.len)) {
|
||||
_ = runtime.system.write("vfstest: write failed\n");
|
||||
return;
|
||||
}
|
||||
_ = u.lseek(fd, 0, u.SEEK_SET);
|
||||
|
||||
var buffer: [32]u8 = undefined;
|
||||
const n = u.read(fd, &buffer);
|
||||
u.close(fd);
|
||||
|
||||
if (n == @as(isize, payload.len) and std.mem.eql(u8, buffer[0..@intCast(n)], payload)) {
|
||||
while (true) {
|
||||
_ = runtime.system.write("vfstest: ok\n");
|
||||
runtime.system.sleep(1000);
|
||||
}
|
||||
}
|
||||
_ = runtime.system.write("vfstest: mismatch\n");
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,140 @@
|
||||
//! /sbin/vfs — the user-space VFS server. Shipped in the initrd, spawned as a
|
||||
//! ring-3 process, and reached by every other process through IPC (the `runtime`
|
||||
//! file API marshals open/read/write/stat/close into calls to this server's
|
||||
//! endpoint, published under the well-known `vfs` service id).
|
||||
//!
|
||||
//! For now the namespace is a small in-memory ramfs (opening a name creates it):
|
||||
//! enough to prove the whole path — client file API -> IPC -> server dispatch ->
|
||||
//! reply. Device nodes backed by user-space drivers (/device) layer on top in M10,
|
||||
//! where `open` on a /device name forwards to the owning driver's endpoint.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const protocol = runtime.vfs_protocol;
|
||||
|
||||
const Node = struct {
|
||||
used: bool = false,
|
||||
name: [24]u8 = undefined,
|
||||
name_len: usize = 0,
|
||||
data: [512]u8 = undefined,
|
||||
size: usize = 0,
|
||||
};
|
||||
|
||||
const OpenFile = struct {
|
||||
used: bool = false,
|
||||
node: usize = 0,
|
||||
};
|
||||
|
||||
var nodes = [_]Node{.{}} ** 8;
|
||||
var opens = [_]OpenFile{.{}} ** 16;
|
||||
|
||||
fn findNode(name: []const u8) ?usize {
|
||||
for (&nodes, 0..) |*n, i| {
|
||||
if (n.used and std.mem.eql(u8, n.name[0..n.name_len], name)) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn createNode(name: []const u8) ?usize {
|
||||
for (&nodes, 0..) |*n, i| {
|
||||
if (!n.used) {
|
||||
const l = @min(name.len, n.name.len);
|
||||
@memcpy(n.name[0..l], name[0..l]);
|
||||
n.* = .{ .used = true, .name = n.name, .name_len = l, .size = 0 };
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn openAt(id: u64) ?*OpenFile {
|
||||
if (id >= opens.len) return null;
|
||||
const o = &opens[@intCast(id)];
|
||||
return if (o.used) o else null;
|
||||
}
|
||||
|
||||
/// Serialise a reply header + payload into `out`; returns the total length.
|
||||
fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize {
|
||||
@memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply));
|
||||
const n = @min(payload.len, out.len - protocol.reply_size);
|
||||
@memcpy(out[protocol.reply_size..][0..n], payload[0..n]);
|
||||
return protocol.reply_size + n;
|
||||
}
|
||||
|
||||
fn fail(out: []u8) usize {
|
||||
return writeReply(out, .{ .status = -1 }, &.{});
|
||||
}
|
||||
|
||||
/// Handle one request; write the reply into `out`, return its length.
|
||||
fn handle(message: []const u8, out: []u8) usize {
|
||||
if (message.len < protocol.request_size) return fail(out);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
|
||||
switch (request.operation) {
|
||||
.open => {
|
||||
const name = payload[0..@min(payload.len, request.len)];
|
||||
const ni = findNode(name) orelse createNode(name) orelse return fail(out);
|
||||
for (&opens, 0..) |*o, i| {
|
||||
if (!o.used) {
|
||||
o.* = .{ .used = true, .node = ni };
|
||||
return writeReply(out, .{ .status = 0, .node = i }, &.{});
|
||||
}
|
||||
}
|
||||
return fail(out);
|
||||
},
|
||||
.read => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
const nd = &nodes[of.node];
|
||||
const off: usize = @intCast(request.offset);
|
||||
if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF
|
||||
const n = @min(@min(nd.size - off, request.len), protocol.maximum_payload);
|
||||
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]);
|
||||
},
|
||||
.write => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
const nd = &nodes[of.node];
|
||||
const off: usize = @intCast(request.offset);
|
||||
if (off > nd.data.len) return fail(out);
|
||||
const n = @min(@min(payload.len, request.len), nd.data.len - off);
|
||||
@memcpy(nd.data[off .. off + n], payload[0..n]);
|
||||
if (off + n > nd.size) nd.size = off + n;
|
||||
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{});
|
||||
},
|
||||
.stat => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
const st = protocol.Stat{ .size = nodes[of.node].size, .kind = 0 };
|
||||
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.Stat) }, std.mem.asBytes(&st));
|
||||
},
|
||||
.close => {
|
||||
if (request.node < opens.len) opens[@intCast(request.node)].used = false;
|
||||
return writeReply(out, .{ .status = 0 }, &.{});
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const endpoint = runtime.ipc.createEndpoint() orelse {
|
||||
_ = runtime.system.write("vfs: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
if (!runtime.ipc.register(.vfs, endpoint)) {
|
||||
_ = runtime.system.write("vfs: register failed\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("vfs: ready\n");
|
||||
|
||||
var reply_buffer: [protocol.message_maximum]u8 = undefined;
|
||||
var reply_len: usize = 0;
|
||||
var receive: [protocol.message_maximum]u8 = undefined;
|
||||
while (true) {
|
||||
const got = runtime.ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive);
|
||||
// Ignore notifications (none expected here); handle a request.
|
||||
reply_len = handle(receive[0..got.len], &reply_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
Reference in New Issue
Block a user