Re-organize the source tree as a monorepo mirroring the FHS

The source layout now mirrors the runtime filesystem hierarchy
(docs/danos-file-system-hierarchy-FSH.md): what lives under system/ in the
source is what a running danos represents under /system. Each service and
driver is a sub-project directory that is its own Zig module — cross-project
references go by module name, never by a path into another project's files.

Moves (all git mv, history preserved):
- src/            -> system/            (danos internals; the self-representation)
    root.zig      -> danos.zig          (the kernel<->user contract module)
    kernel/arch/  -> kernel/architecture/   (arch -> architecture)
    device/       -> devices/           (what /system/devices reflects)
    boot/         -> /boot              (the loaders, top level)
- sbin/           -> split by role:
    init, vfs     -> system/services/<name>/<name>.zig
    hpetd, busd   -> system/drivers/<name>/<name>.zig
    vfs-test      -> system/services/vfs/vfs-test.zig  (inside the vfs project)
- lib/            -> library/runtime/   (room for other libraries beside runtime)

The VFS wire protocol becomes its own module, system/services/vfs/protocol.zig
("vfs-protocol"): the vfs sub-project exposes its interface, and the runtime's
file layer imports it by name. First instance of the "protocol module" pattern
(docs/driver-model.md); usb/block will expose theirs the same way.

Also: fix a naming-standard violation in the protocol — Op -> Operation (and
req -> request, _pad -> _padding). Docs updated: /system/services added to the
FHS doc, a repository-layout section added to the docs index, and stale source
paths swept across comments and docs.

Runtime boot paths are unchanged (the bootloader still loads /sbin/init);
aligning the runtime filesystem to the FHS is a separate follow-up. Suite 35/35
plus host tests green.
This commit is contained in:
Daniel Samson
2026-07-10 12:55:56 +01:00
parent 15b70856c9
commit 8754d4e46a
83 changed files with 334 additions and 177 deletions
+413
View File
@@ -0,0 +1,413 @@
//! Local APIC and its timer — the source of device interrupts.
//!
//! Modern x86 routes interrupts through the per-CPU Local APIC (the legacy 8259
//! PIC is remapped out of the way and masked). The LAPIC also has a built-in
//! timer, which is the simplest device interrupt to bring up: it needs no
//! external routing, just a vector and a count. We use it as danos's heartbeat.
//!
//! The LAPIC is memory-mapped (default physical 0xFEE00000, inside our identity
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
//! the LAPIC won't deliver the next one.
const danos = @import("danos");
const io = @import("io.zig");
const paging = @import("paging.zig");
/// The ACPI PM timer, as a calibration reference: an I/O port or MMIO counter.
pub const PmTimer = struct { mmio: bool, address: u64, is_32bit: bool };
// Platform facts from discovery (set by `configure` before bring-up). Defaults are
// the legacy-safe assumptions so the code still works if discovery never ran.
var configuration_pic_present: bool = true;
var configuration_hpet_base: u64 = 0; // 0 = no HPET discovered
var configuration_pm_timer: ?PmTimer = null;
/// Which reference the last calibration used, for logging.
var cal_source: []const u8 = "none";
/// Hand the LAPIC bring-up the discovered platform facts. Call before `init`.
pub fn configure(pic_present: bool, hpet_base: u64, pm_timer: ?PmTimer) void {
configuration_pic_present = pic_present;
configuration_hpet_base = hpet_base;
configuration_pm_timer = pm_timer;
}
/// The calibration reference the timer was measured against ("cpuid"/"hpet"/…).
pub fn calibrationSource() []const u8 {
return cal_source;
}
/// IDT vector the timer fires on (in the device range, >= 32).
pub const timer_vector = 32;
/// Spurious-interrupt vector. Low nibble 0xF by convention; also in our gate
/// range so a stray spurious interrupt lands on a valid (no-op) handler.
const spurious_vector = 47;
// LAPIC register offsets.
const register_spurious = 0x0F0;
const register_eoi = 0x0B0;
const register_id = 0x020; // this core's LAPIC id, in bits 24-31
const register_icr_low = 0x300; // interrupt command register, low dword (writing it sends)
const register_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31)
const register_lvt_timer = 0x320;
const register_timer_initial = 0x380;
const register_timer_current = 0x390;
const register_timer_divide = 0x3E0;
const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight
const lvt_masked = 1 << 16;
const lvt_periodic = 1 << 17;
const timer_divide_16 = 0x3;
const ia32_apic_base_msr = 0x1B;
/// LAPIC MMIO base. A runtime var (not a constant) both because we read it from
/// the MSR and so register writes compile to normal stores rather than a
/// `mov moffs`, which the self-hosted backend can't encode.
var base: usize = 0xFEE00000;
var tick_count: u64 = 0;
/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate).
/// At divide-by-16, this is the effective counting rate.
var ticks_per_ms: u32 = 0;
/// The periodic-interrupt frequency the timer is armed at, once initTimer runs.
var timer_hz: u32 = 0;
/// TSC (Time Stamp Counter) calibration: cycles per second, and the count at boot.
/// The TSC is a per-core cycle counter, giving a ~nanosecond high-resolution
/// monotonic clock — far finer than the millisecond timer tick.
var tsc_hz: u64 = 0;
var tsc_base: u64 = 0;
/// Read the 64-bit Time Stamp Counter.
fn rdtsc() u64 {
var low: u32 = undefined;
var high: u32 = undefined;
asm volatile ("rdtsc"
: [low] "={eax}" (low),
[high] "={edx}" (high),
);
return (@as(u64, high) << 32) | low;
}
fn read(register: u32) u32 {
return @as(*volatile u32, @ptrFromInt(base + register)).*;
}
fn write(register: u32, value: u32) void {
@as(*volatile u32, @ptrFromInt(base + register)).* = value;
}
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
/// vectors) and mask every line, so it can't deliver interrupts behind the APIC.
fn remapAndMaskPic() void {
io.outb(0x20, 0x11); // start init (cascade mode)
io.outb(0xA0, 0x11);
io.outb(0x21, 0x20); // master offset 0x20
io.outb(0xA1, 0x28); // slave offset 0x28
io.outb(0x21, 0x04); // tell master about slave on IRQ2
io.outb(0xA1, 0x02);
io.outb(0x21, 0x01); // 8086 mode
io.outb(0xA1, 0x01);
io.outb(0x21, 0xFF); // mask all
io.outb(0xA1, 0xFF);
}
/// Enable the Local APIC: mask the PIC (only if one is present — a legacy-free
/// UEFI Class 3 machine may have none), set the global-enable MSR bit, and
/// software-enable the APIC via its spurious-vector register.
pub fn init() void {
if (configuration_pic_present) remapAndMaskPic();
const msr = io.rdmsr(ia32_apic_base_msr);
// Reach the LAPIC through the physmap (paging.init maps its page there).
base = @intCast(danos.physicalToVirtual(msr & 0xFFFFF000)); // physical base is bits 12+
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
write(register_spurious, 0x100 | spurious_vector); // bit 8 = software enable
}
/// Software-enable *this* core's Local APIC — the application-processor counterpart
/// of `init`, minus the one-time PIC remap (the BSP already masked it) and minus
/// calibration (the timer rate is a shared hardware constant, measured once). Each
/// core has its own LAPIC at the same MMIO address, so no per-core base is needed.
pub fn initSecondary() void {
const msr = io.rdmsr(ia32_apic_base_msr);
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
write(register_spurious, 0x100 | spurious_vector); // software enable
}
// --- application-processor wakeup (INIT–SIPI–SIPI) --------------------------
/// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of
/// the wake sequence. Blocks until the LAPIC reports the IPI was delivered.
pub fn sendInit(apic_id: u32) void {
write(register_icr_high, apic_id << 24);
write(register_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered
waitIcrIdle();
}
/// Send a STARTUP IPI (SIPI) telling the target core to begin executing at physical
/// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is
/// sent twice after the INIT; both calls block until delivery completes.
pub fn sendStartup(apic_id: u32, vector: u8) void {
write(register_icr_high, apic_id << 24);
write(register_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector
waitIcrIdle();
}
fn waitIcrIdle() void {
while (read(register_icr_low) & icr_delivery_pending != 0) {}
}
/// The calibration window: we time everything against a 10 ms reference interval.
const calib_ms = 10;
/// Measure the LAPIC timer's and the TSC's rates. The PIT (legacy 8254) can be
/// absent on UEFI Class 3 firmware — and polling it would hang — so we pick a
/// reference clock in order of preference: the CPU's own TSC frequency (CPUID leaf
/// 0x15, no external timer needed), then the discovered HPET, then the ACPI PM
/// timer, and only the PIT as a last resort. Each path yields the same two rates.
pub fn calibrate() void {
var done = false;
// 1. CPUID leaf 0x15 gives the TSC frequency directly — measure the LAPIC
// against the TSC itself, needing no external timer at all.
if (cpuidTscHz()) |hz| {
measure(hz, ~@as(u64, 0), rdtsc);
tsc_hz = hz; // keep the exact enumerated value
cal_source = "cpuid";
done = true;
}
// 2. The discovered HPET.
if (!done and configuration_hpet_base != 0) {
if (hpetHz()) |hpet_hz| {
measure(hpet_hz, hpetMask(), readHpet);
cal_source = "hpet";
done = true;
}
}
// 3. The ACPI PM timer (fixed 3.579545 MHz).
if (!done) {
if (configuration_pm_timer) |pt| {
measure(3_579_545, if (pt.is_32bit) 0xFFFF_FFFF else 0xFF_FFFF, readPmTimer);
cal_source = "pm-timer";
done = true;
}
}
// 4. The legacy PIT, last resort.
if (!done) {
calibratePit();
cal_source = "pit";
}
// A bad measurement (no reference actually ticked) leaves nonsense; fall back.
if (ticks_per_ms == 0 or tsc_hz == 0) {
calibratePit();
cal_source = "pit";
}
tsc_base = rdtsc(); // the clock's zero point (boot)
}
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
/// clock (frequency `ref_hz`, counter width `ref_mask`) counts out `calib_ms`, and
/// snapshot the TSC across the same window. Yields `ticks_per_ms` and `tsc_hz`.
fn measure(ref_hz: u64, ref_mask: u64, refNow: *const fn () u64) void {
const calib_ticks = ref_hz / (1000 / calib_ms); // reference ticks in calib_ms
write(register_timer_divide, timer_divide_16);
write(register_lvt_timer, lvt_masked);
write(register_timer_initial, 0xFFFFFFFF);
const ref0 = refNow();
const tsc0 = rdtsc();
while (((refNow() -% ref0) & ref_mask) < calib_ticks) {}
const tsc1 = rdtsc();
const elapsed = 0xFFFFFFFF - read(register_timer_current);
write(register_timer_initial, 0);
ticks_per_ms = elapsed / calib_ms;
tsc_hz = (tsc1 -% tsc0) * (1000 / calib_ms);
}
/// The PIT fallback (legacy 8254 channel 2, polled). Only reached when no better
/// reference exists — on a legacy-free machine this path isn't taken.
fn calibratePit() void {
const pit_hz = 1_193_182;
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
write(register_timer_divide, timer_divide_16);
write(register_lvt_timer, lvt_masked);
write(register_timer_initial, 0xFFFFFFFF);
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
io.outb(0x42, @truncate(pit_count));
io.outb(0x42, @truncate(pit_count >> 8));
const tsc_start = rdtsc();
io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start
var guard: u64 = 0;
while (io.inb(0x61) & 0x20 == 0 and guard < 100_000_000) : (guard += 1) {} // bounded
const tsc_end = rdtsc();
const elapsed = 0xFFFFFFFF - read(register_timer_current);
write(register_timer_initial, 0);
ticks_per_ms = elapsed / calib_ms;
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms);
}
// --- reference clocks ------------------------------------------------------
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
/// null if the CPU doesn't enumerate it (common under QEMU).
fn cpuidTscHz() ?u64 {
if (cpuid(0).eax < 0x15) return null;
const r = cpuid(0x15);
if (r.eax == 0 or r.ebx == 0 or r.ecx == 0) return null; // ratio/crystal not given
return @as(u64, r.ecx) * r.ebx / r.eax;
}
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
fn cpuid(leaf: u32) CpuidRegs {
var a: u32 = undefined;
var b: u32 = undefined;
var c: u32 = undefined;
var d: u32 = undefined;
asm volatile ("cpuid"
: [a] "={eax}" (a),
[b] "={ebx}" (b),
[c] "={ecx}" (c),
[d] "={edx}" (d),
: [leaf] "{eax}" (leaf),
[sub] "{ecx}" (@as(u32, 0)),
);
return .{ .eax = a, .ebx = b, .ecx = c, .edx = d };
}
// HPET registers: capabilities at +0x00 (period in the high dword, in fs; bit 13 =
// 64-bit-counter capable), general configuration at +0x10, main counter at +0xF0.
fn hpetRead64(off: usize) u64 {
return @as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).*;
}
fn hpetWrite64(off: usize, value: u64) void {
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
}
/// Map + enable the HPET and return its tick frequency, or null if unusable.
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
/// address, so the register accessors reach it without the identity map.
fn hpetHz() ?u64 {
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
const caps = hpetRead64(0x00);
const period_fs = caps >> 32; // femtoseconds per tick
if (period_fs == 0) return null;
hpetWrite64(0x10, hpetRead64(0x10) | 1); // ENABLE_CNF: start the main counter
return 1_000_000_000_000_000 / period_fs; // 1e15 fs/s ÷ fs/tick
}
/// The HPET counter width mask (64- or 32-bit, per caps bit 13).
fn hpetMask() u64 {
return if (hpetRead64(0x00) & (1 << 13) != 0) ~@as(u64, 0) else 0xFFFF_FFFF;
}
fn readHpet() u64 {
return hpetRead64(0xF0);
}
fn readPmTimer() u64 {
const pt = configuration_pm_timer.?;
// MMIO PM timer via the physmap (mapMmio is idempotent); the common case is
// a legacy I/O port.
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(paging.mapMmio(pt.address, 4, false))).*;
return io.inl(@intCast(pt.address));
}
/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires
/// calibrate() to have run.
pub fn initTimer(hz: u32) void {
timer_hz = hz;
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
write(register_timer_divide, timer_divide_16);
write(register_lvt_timer, timer_vector | lvt_periodic);
write(register_timer_initial, @intCast(count));
}
/// Configured periodic-interrupt frequency (Hz).
pub fn frequencyHz() u32 {
return timer_hz;
}
/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks.
pub fn lapicHz() u64 {
return @as(u64, ticks_per_ms) * 1000;
}
/// Measured TSC frequency (Hz).
pub fn tscHz() u64 {
return tsc_hz;
}
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
// the scheduler uses for sleep deadlines.
pub fn nanos() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
}
pub fn micros() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
}
pub fn millis() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
}
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
pub fn eoi() void {
write(register_eoi, 0);
}
/// Optional callback run each tick (the scheduler registers it for preemption).
var on_tick: ?*const fn () void = null;
pub fn setTickHook(hook: *const fn () void) void {
on_tick = hook;
}
/// The timer interrupt handler: advance the monotonic tick count, then run the
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
/// the dispatcher before we get here, so a task switch here doesn't stall it.
pub fn timerTick() void {
// Acknowledge before the tick hook: `on_tick` is the scheduler, which may switch
// tasks and not return promptly, and the LAPIC mustn't wait on it to deliver the
// next interrupt. (Each device handler now owns its own EOI — see
// `idt.interruptDispatch` — because a *routed* interrupt must be masked at the
// I/O APIC before it is acknowledged, an ordering the dispatcher can't impose.)
eoi();
tick_count +%= 1;
if (on_tick) |hook| hook();
}
/// This core's Local APIC id — the interrupt destination for `routeGsi`.
pub fn localId() u8 {
return @truncate(read(register_id) >> 24);
}
/// Number of timer ticks so far. Volatile load: the count is bumped
/// asynchronously by the interrupt handler, so callers must re-read memory.
pub fn ticks() u64 {
return @as(*const volatile u64, &tick_count).*;
}
+578
View File
@@ -0,0 +1,578 @@
//! x86_64 CPU operations. This is the "architecture" module: the generic kernel imports
//! it as `@import("architecture")` and never names x86_64 directly, so a second
//! architecture is added by pointing that module at a different directory in
//! build.zig — no change to the generic code. Keep everything CPU-specific here
//! (halt, the descriptor tables, later paging), and nothing generic.
const danos = @import("danos");
const parameters = @import("parameters");
const gdt = @import("gdt.zig");
const tss = @import("tss.zig");
const idt = @import("idt.zig");
const paging = @import("paging.zig");
const serial = @import("serial.zig");
const apic = @import("apic.zig");
const ioapic = @import("ioapic.zig");
const io = @import("io.zig");
const smp = @import("smp.zig");
const pcpu = @import("per-cpu.zig");
/// The saved register/trap frame passed to a fault handler.
pub const CpuState = idt.CpuState;
// --- trap-frame accessors ---------------------------------------------------
// The frame's fields are x86_64 registers; the generic kernel reads it through
// these accessors so it never names one.
/// The interrupted/faulting instruction address (RIP here; ELR_EL1 on aarch64,
/// sepc on riscv64).
pub fn instructionPointer(state: *const CpuState) u64 {
return state.rip;
}
/// The interrupted stack pointer (RSP here).
pub fn stackPointer(state: *const CpuState) u64 {
return state.rsp;
}
/// Whether the trap came from user mode (CPL 3 here; EL0 on aarch64, U-mode on
/// riscv64).
pub fn fromUser(state: *const CpuState) bool {
return state.cs & 3 == 3;
}
/// The faulting virtual address, if this trap is a page fault (CR2 here;
/// FAR_EL1 on aarch64, stval on riscv64). Null for any other exception.
pub fn faultAddress(state: *const CpuState) ?u64 {
if (state.vector != 14) return null;
return asm volatile ("mov %%cr2, %[out]"
: [out] "=r" (-> u64),
);
}
// --- system_call ABI ------------------------------------------------------------
// The System V-style register convention (number in rax, arguments in
// rdi/rsi/rdx/r10/r8/r9, result in rax), exposed positionally so the generic
// dispatcher never names a register.
/// The system_call number the user program passed.
pub fn systemCallNumber(state: *const CpuState) u64 {
return state.rax;
}
/// Positional system_call argument `n`.
pub fn systemCallArg(state: *const CpuState, n: u8) u64 {
return switch (n) {
0 => state.rdi,
1 => state.rsi,
2 => state.rdx,
3 => state.r10,
4 => state.r8,
5 => state.r9,
else => 0,
};
}
/// Write the system_call's return value into the frame — the entry paths restore
/// user registers from it.
pub fn setSystemCallResult(state: *CpuState, value: u64) void {
state.rax = value;
}
/// Write a *second* system_call return value (rdx here — restored by both the
/// system_call/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by
/// sysretq). Used by IPC_ReplyWait to hand back the sender's badge alongside the
/// message length in rax.
pub fn setSystemCallResult2(state: *CpuState, value: u64) void {
state.rdx = value;
}
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
/// so it can be the very first thing called.
pub fn serialInit() void {
serial.init();
}
/// Write bytes to the serial port.
pub fn serialWrite(bytes: []const u8) void {
serial.write(bytes);
}
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
/// displays. The last-resort progress signal when there's no text output at all.
/// Writing 0x80 is universally safe (it's the legacy I/O-delay port).
pub fn checkpoint(code: u8) void {
io.outb(0x80, code);
}
/// Whether a Bochs/QEMU-style debug console is on port 0xE9 (it returns 0xE9 when
/// read). On real hardware the port reads back 0xFF, so this stays false — a safe
/// probe before we write to it.
pub fn debugconPresent() bool {
return io.inb(0xE9) == 0xE9;
}
/// Output sink: write bytes to the 0xE9 debug console (see `debugconPresent`).
pub fn debugconWrite(bytes: []const u8) void {
for (bytes) |b| io.outb(0xE9, b);
}
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
/// stack for double faults), then the IDT with exception handlers. After this a
/// CPU fault is reported instead of triple-faulting. Install the fault handler
/// (setFaultHandler) first so early faults are caught.
pub fn init() void {
gdt.init();
tss.init();
idt.init();
pcpu.initSystemCall();
}
/// Build the kernel's own page tables (with real permissions) and switch onto
/// them. Needs the frame allocator and the boot info (for the memory map and the
/// kernel's segment layout). Call once the frame allocator is up.
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
paging.init(allocFrame, freeFrame, boot_information);
}
/// Create a new address space (returns the physical address of its root table —
/// the PML4 here — or null). Shares the kernel's higher half; the user (low)
/// half starts empty.
pub fn createAddressSpace() ?u64 {
return paging.createAddressSpace();
}
/// Free an address space and everything mapped in its user half. Caller must not
/// be running on it.
pub fn destroyAddressSpace(root: u64) void {
paging.destroyAddressSpace(root);
}
/// Map a user page into address space `root` (W^X is the caller's contract).
pub fn mapUserPageInto(root: u64, virtual: u64, physical: u64, writable: bool, executable: bool) void {
paging.mapUserInto(root, virtual, physical, writable, executable);
}
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void {
paging.mapUserDeviceInto(root, virtual, physical, len);
}
/// Map a page into the kernel address space (non-executable). For the heap, etc.
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
paging.map(virtual, physical, writable);
}
/// Map a device MMIO range and return the virtual address to reach it at. This
/// is the device layer's `Hal.mapMmio` — it hands back a physmap pointer and
/// never exposes how the mapping is placed.
pub fn mapMmio(physical: u64, len: u64, writable: bool) u64 {
return paging.mapMmio(physical, len, writable);
}
/// The kernel's page-table root (physical), shared into every address space.
pub fn kernelPageTable() u64 {
return paging.kernelPml4();
}
/// Switch the active address space (load CR3 with a physical root table).
pub fn loadPageTable(root: u64) void {
paging.loadCr3(root);
}
/// The physical root of the currently active page tables (CR3 here; TTBR0/satp
/// elsewhere).
pub fn activePageTable() u64 {
return asm volatile ("mov %%cr3, %[out]"
: [out] "=r" (-> u64),
);
}
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
/// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the
/// per-CPU `kernel_rsp` (for the system_call stub, which switches by hand). Updated
/// by the scheduler when it switches to a user task.
pub fn setKernelStack(cpu: usize, top: usize) void {
tss.rsp0Ptr(cpu).* = top;
pcpu.setKernelRsp(cpu, top);
}
/// Remove a kernel mapping.
pub fn unmapPage(virtual: u64) void {
paging.unmap(virtual);
}
/// Remove a page mapping from address space `root` (for munmap of user pages).
/// Clears the leaf entry only; freeing the underlying frame is the caller's job.
pub fn unmapUserPageInto(root: u64, virtual: u64) void {
paging.unmapInto(root, virtual);
}
/// Resolve `virtual` to its physical address in the address space rooted at `root`
/// (any address space, not just the live one), or null if unmapped. Used to find
/// the frame behind a user page for munmap, and for cross-address-space copies.
pub fn translate(root: u64, virtual: u64) ?u64 {
return paging.translateIn(root, virtual);
}
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
/// W^X: code read-only + executable, data writable + no-execute.
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
paging.mapUser(virtual, physical, writable, executable);
}
// --- ring 3 entry/exit -----------------------------------------------------
/// Drop to ring 3 at `rip` on `rsp` (defined in isr.s). Saves the kernel context,
/// publishes the kernel stack pointer through `rsp0_slot` (this core's TSS.rsp0,
/// so ring-3 interrupts land on a good stack), builds an iretq frame with the
/// user selectors, and iretq's. "Returns" only when the user program triggers
/// the exit path (user_exit_to_kernel).
extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void;
/// Abandon the in-flight ring-3 trap context and resume the kernel as if
/// `enter_user` had returned (defined in isr.s). Called by the exit system_call.
extern fn user_exit_to_kernel() callconv(.c) noreturn;
/// Run user code at `entry` with stack `stack_top` on this core (`cpu` = the
/// caller's CPU index; the architecture layer can't ask the scheduler). Returns after the
/// user program exits via system_call. Interrupts are disabled on return (the exit
/// arrives through an interrupt gate) — the caller re-enables.
pub fn enterUser(cpu: usize, entry: u64, stack_top: u64) void {
enter_user(entry, stack_top, tss.rsp0Ptr(cpu));
}
/// Never returns to the user program: unwind to the kernel context that called
/// `enterUser`. For the exit system_call's handler.
pub fn userExit() noreturn {
user_exit_to_kernel();
}
/// Register the handler for the user system_call gate (int 0x80, vector 128). The
/// handler may write the trap frame (see `setSystemCallResult`).
pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void {
idt.setSystemCallHandler(handler);
}
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
/// core calls this once, after its GDT is in place (a GS *selector* reload would
/// clobber the base). See percpu.zig for the swapgs discipline.
pub fn setCpuLocal(cpu: usize, ptr: usize) void {
pcpu.setLocal(cpu, ptr);
}
/// This core's scheduler pointer (via the GS base) — a per-core register, so each
/// core sees its own without locking. Valid in any ring-0 context.
pub fn cpuLocal() usize {
return pcpu.scheduler();
}
// --- SMP: application-processor bring-up ----------------------------------
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
/// The frame stays inert (zeroed, non-executable) between wakes and is armed only
/// while a core is climbing — so a core can be (re)woken at any time (retry, or a
/// future power manager) without leaving an executable page resident. See smp.zig.
pub fn setTrampolinePage(physical: u64) void {
smp.setTrampolinePage(physical);
}
/// Wake the core with hardware id `hw_id` (its Local APIC id here; MPIDR on
/// aarch64, hart id on riscv64) as dense CPU `index`, giving it `stack_top` and
/// its per-CPU pointer `percpu`; it adopts the kernel page tables. Returns false
/// if it doesn't come online within the timeout. Blocks until the core reports in.
pub fn startSecondary(hw_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
// The AP adopts the kernel page tables explicitly — never the caller's live
// CR3, which a future re-wake from a core running a process would make a
// process address space.
return smp.startAp(hw_id, stack_top, percpu, index, paging.kernelPml4());
}
/// Register the generic entry a woken AP jumps to once its architecture state is up (its own
/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here.
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
smp.setSecondaryEntry(entry);
}
/// Bytes the kernel should allocate for a secondary core's dedicated fault stack
/// (the IST double-fault stack here), and where to record its top before waking
/// the core. The stack is heap-allocated per online core (the boot CPU's is
/// static — it's needed before the allocator exists). See tss.zig.
pub const fault_stack_size = tss.ist_stack_size;
pub fn setFaultStack(cpu: usize, top: usize) void {
tss.setApIstStack(cpu, top);
}
/// Test hook: force the next `n` AP wake attempts to fail, so the retry path can be
/// exercised deterministically (see the smp-retry test). No effect when `n` is 0.
pub fn testFailNextWakes(n: u32) void {
smp.testFailNextWakes(n);
}
/// The reserved AP-trampoline frame (0 if none). For tests that check it's inert.
pub fn trampolinePage() u64 {
return smp.trampolinePage();
}
/// Whether the page at `virtual` is currently mapped executable (present, NX clear).
pub fn pageExecutable(virtual: u64) bool {
return paging.isExecutable(virtual);
}
/// Kernel tick rate (the scheduler's time quantum), from configuration.
pub const timer_hz = parameters.timer_hz;
/// The ACPI PM timer, as a calibration reference (re-exported for the configuration).
pub const PmTimer = apic.PmTimer;
/// A MADT interrupt-source override (re-exported for the configuration).
pub const IsoEntry = ioapic.IsoEntry;
/// Discovered platform facts the architecture layer needs so it makes no legacy
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
pub const PlatformConfiguration = struct {
/// Whether the legacy 8259 PIC is present (skip programming it if not).
pic_present: bool = true,
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
hpet_base: u64 = 0,
/// The ACPI PM timer, another calibration reference.
pm_timer: ?PmTimer = null,
/// I/O APIC MMIO base + its first global system interrupt (0 = none).
ioapic_base: u64 = 0,
ioapic_gsi_base: u32 = 0,
/// MADT ISA-IRQ overrides, for I/O APIC routing.
overrides: []const IsoEntry = &.{},
};
/// Apply the discovered platform configuration. Must run before `startTimer` (the timer
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
/// Maps + masks the I/O APIC immediately.
pub fn configurePlatform(configuration: PlatformConfiguration) void {
apic.configure(configuration.pic_present, configuration.hpet_base, configuration.pm_timer);
ioapic.configure(configuration.ioapic_base, configuration.ioapic_gsi_base, configuration.overrides);
ioapic.init();
}
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
pub fn serialReconfigure(is_mmio: bool, address: u64) void {
serial.reconfigure(is_mmio, address);
}
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
pub fn timerCalibrationSource() []const u8 {
return apic.calibrationSource();
}
/// External-interrupt-router diagnostics, for boot logging / verification (the
/// I/O APIC's redirection entries here; a GIC distributor or PLIC elsewhere).
pub fn irqRouteCount() u32 {
return ioapic.entryCount();
}
pub fn irqRouteRaw(n: u32) u32 {
return ioapic.entryLow(n);
}
// --- device-IRQ plumbing, for system/kernel/irq.zig -----------------------------
//
// The generic IRQ layer speaks GSIs and vectors; everything below hides the fact
// that on x86_64 those mean "I/O APIC redirection entry" and "IDT gate". The
// vector window is bounded by the stubs isr.s actually emits: `gate_count` = 48,
// vector 32 is the LAPIC timer and 47 is the spurious vector, leaving 33..46.
pub const irq_vector_base: u8 = 33;
pub const irq_vector_count: u8 = 14; // 33..46 inclusive
/// True if `gsi` is one this machine's interrupt router can deliver.
pub fn irqOwnsGsi(gsi: u32) bool {
return ioapic.ownsGsi(gsi);
}
/// Install `handler` on `vector` (an absolute IDT gate index).
pub fn irqSetHandler(vector: u8, handler: *const fn () void) void {
idt.setHandler(vector, handler);
}
/// Route `gsi` to `vector` on *this* core, masked. Unmask with `irqUnmask` once bound.
pub fn irqRoute(gsi: u32, vector: u8, level: bool, active_low: bool) void {
ioapic.routeGsi(gsi, vector, apic.localId(), level, active_low);
}
pub fn irqMask(gsi: u32) void {
ioapic.maskGsi(gsi);
}
pub fn irqUnmask(gsi: u32) void {
ioapic.unmaskGsi(gsi);
}
/// Acknowledge the interrupt currently in service on this core's LAPIC.
pub fn irqEoi() void {
apic.eoi();
}
/// Enable the Local APIC, calibrate its timer against the best available reference
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
/// unmasked with enableInterrupts() to be delivered. Run `configurePlatform` first.
pub fn startTimer() void {
apic.init();
apic.calibrate();
idt.setHandler(apic.timer_vector, apic.timerTick);
apic.initTimer(timer_hz);
}
/// Number of timer ticks since startTimer().
pub fn ticks() u64 {
return apic.ticks();
}
// Monotonic high-resolution clock (from the TSC), one function per resolution.
pub fn nanos() u64 {
return apic.nanos();
}
pub fn micros() u64 {
return apic.micros();
}
pub fn millis() u64 {
return apic.millis();
}
/// Measured frequency of the tick timer's input clock (the LAPIC timer here), in
/// Hz, from calibration.
pub fn timerClockHz() u64 {
return apic.lapicHz();
}
/// Measured frequency of the monotonic clock's underlying counter (the TSC here;
/// CNTVCT on aarch64, `time` on riscv64), in Hz.
pub fn clockHz() u64 {
return apic.tscHz();
}
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
pub fn enableInterrupts() void {
asm volatile ("sti");
}
/// Mask maskable interrupts (`cli`).
pub fn disableInterrupts() void {
asm volatile ("cli");
}
/// Disable interrupts and return the previous flags, so a nested critical section
/// can restore the caller's state rather than blindly re-enabling. Pairs with
/// restoreInterrupts.
pub fn saveInterrupts() u64 {
var flags: u64 = undefined;
asm volatile (
\\pushfq
\\pop %[f]
\\cli
: [f] "=r" (flags),
:
: .{ .memory = true }
);
return flags;
}
/// Re-enable interrupts only if they were enabled when `flags` was captured.
pub fn restoreInterrupts(flags: u64) void {
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
}
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
pub fn setTickHook(hook: *const fn () void) void {
apic.setTickHook(hook);
}
// --- context switching (for the scheduler) -------------------------------
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
/// pointer is written to `old_rsp`. Defined in isr.s.
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
pub fn switchContext(old_sp: *usize, new_sp: usize) void {
switch_context(old_sp, new_sp);
}
/// Build the initial stack for a new task so that switching to it lands in
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
/// The layout must match switch_context's push order (callee-saved, then the
/// return address on top); `entry` is smuggled in via the r15 slot.
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
var sp = stack_top;
const push = struct {
fn f(p: *usize, value: usize) void {
p.* -= @sizeOf(usize);
@as(*usize, @ptrFromInt(p.*)).* = value;
}
}.f;
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
push(&sp, 0); // rbx
push(&sp, 0); // rbp
push(&sp, 0); // r12
push(&sp, 0); // r13
push(&sp, 0); // r14
push(&sp, entry); // r15 -> task entry, read by task_trampoline
return sp;
}
/// Drop the current (kernel-context) task to ring 3 at `rip` on `rsp`, never
/// returning (defined in isr.s). Used by the scheduler's user-task trampoline
/// once it has switched onto the task and read its entry/stack. Interrupts are
/// disabled across the swapgs+iretq so no interrupt observes the user GS base in
/// ring 0; the pushed RFLAGS re-enables them in ring 3.
extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn;
pub fn jumpToUser(entry: u64, stack_top: u64) noreturn {
jump_to_user(entry, stack_top);
}
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
/// return. Until set, faults just halt the core.
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
idt.on_fault = handler;
}
/// A human-readable name for a CPU exception vector.
pub fn exceptionName(vector: u64) []const u8 {
return idt.vectorName(vector);
}
/// Read `width` bytes (1/2/4) from an I/O port. The generic device layer drives
/// ACPI registers through this rather than naming x86 port instructions; on an
/// MMIO-only architecture this would be implemented differently.
pub fn pioRead(width: u8, port: u16) u32 {
return switch (width) {
1 => io.inb(port),
2 => io.inw(port),
4 => io.inl(port),
else => 0,
};
}
/// Write `width` bytes (1/2/4) to an I/O port.
pub fn pioWrite(width: u8, port: u16, value: u32) void {
switch (width) {
1 => io.outb(port, @truncate(value)),
2 => io.outw(port, @truncate(value)),
4 => io.outl(port, value),
else => {},
}
}
/// Park the core forever. `hlt` drops it into a low-power idle until the next
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
/// docs/halting.md for the full reasoning.
pub fn halt() noreturn {
while (true) asm volatile ("hlt");
}
/// Spin-wait hint (`pause`). Emitted in the body of a spinlock's busy-wait: it
/// relaxes the core while it polls a contended lock — yielding pipeline resources
/// to a hyperthread sibling and easing the cache-coherency traffic on the lock
/// line. Purely a performance/power hint; correct to omit, but kinder on the bus.
pub fn cpuRelax() void {
asm volatile ("pause");
}
+89
View File
@@ -0,0 +1,89 @@
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
//! reference a code selector — so we install our own flat GDT with known
//! selectors (0x08/0x10 kernel code/data, 0x18/0x20 user data/code for ring 3)
//! rather than trusting whatever the firmware left in place.
//!
//! The code/data descriptors are identical on every core, but the **TSS descriptor
//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see
//! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its
//! own copy of the table with its own TSS descriptor. Slot 0 is the BSP.
const parameters = @import("parameters");
/// Selectors into the table (index * 8). Same on every core's GDT.
pub const kernel_code = 0x08;
pub const kernel_data = 0x10;
pub const user_data = 0x18;
pub const user_code = 0x20;
pub const tss_selector = 0x28;
/// Ring-3 selectors as loaded from user mode: RPL 3 or'd in. The user *data*
/// descriptor is load-bearing even in long mode — iretq to CPL 3 with a null SS
/// raises #GP(0).
pub const user_code_rpl3 = user_code | 3;
pub const user_data_rpl3 = user_data | 3;
const maximum_cpus = parameters.maximum_cpus;
const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high
/// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor,
/// filled in per core by `setTssFor`.
/// kernel code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
/// kernel data: present, ring 0, writable -> 0x00CF9200_0000FFFF
/// user data: present, ring 3, writable -> 0x00CFF200_0000FFFF
/// user code: present, ring 3, executable, readable, L=1 -> 0x00AFFA00_0000FFFF
/// User data sits below user code so a future SYSRET works unchanged: it loads
/// CS = STAR.SYSRET_CS + 16 and SS = STAR.SYSRET_CS + 8, so with SYSRET_CS = 0x10
/// those land on 0x20 (user code) and 0x18 (user data).
const template = [entries]u64{
0, // null descriptor (required)
0x00AF9A000000FFFF, // kernel code (0x08)
0x00CF92000000FFFF, // kernel data (0x10)
0x00CFF2000000FFFF, // user data (0x18)
0x00AFFA000000FFFF, // user code (0x20)
0, // TSS descriptor low (0x28)
0, // TSS descriptor high
};
/// One GDT per core (each a copy of the template, differing only in its TSS slot).
var gdts = [_][entries]u64{template} ** maximum_cpus;
/// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
/// TSS. Write it into that core's GDT before it loads the TSS selector.
pub fn setTssFor(cpu: usize, base: u64, limit: u64) void {
gdts[cpu][5] = (limit & 0xFFFF) |
((base & 0xFFFF) << 16) |
(((base >> 16) & 0xFF) << 32) |
(@as(u64, 0x89) << 40) |
(((limit >> 16) & 0xF) << 48) |
(((base >> 24) & 0xFF) << 56);
gdts[cpu][6] = (base >> 32) & 0xFFFFFFFF;
}
/// The operand `lgdt` wants: table byte-length minus one, then its address.
const Descriptor = packed struct {
limit: u16,
base: u64,
};
/// Loads the GDT and reloads the segment registers (including CS). Defined in
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match the table.
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
/// Load core `cpu`'s GDT and switch onto its segments. Note this reloads the segment
/// registers, which zeroes the GS base — so a core must publish its per-CPU pointer
/// (setCpuLocal) *after* calling this.
pub fn loadOnThisCpu(cpu: usize) void {
const descriptor = Descriptor{
.limit = @sizeOf([entries]u64) - 1,
.base = @intFromPtr(&gdts[cpu]),
};
gdt_flush(&descriptor);
}
/// Install the bootstrap processor's GDT (slot 0) and switch onto its segments.
pub fn init() void {
loadOnThisCpu(0);
}
+186
View File
@@ -0,0 +1,186 @@
//! Interrupt Descriptor Table, CPU-exception handlers, and device-interrupt
//! dispatch. Without this, any fault (a stray pointer, a bad page-table entry)
//! triple-faults and silently resets the machine. With it, the CPU vectors into
//! our stubs, which capture the register state and hand it to a dispatcher.
//!
//! Vectors split in two: 0-31 are CPU exceptions (terminal — reported and
//! halted); 32+ are device interrupts (a registered handler runs, the APIC is
//! acknowledged, and we return to the interrupted code).
const gdt = @import("gdt.zig");
const tss = @import("tss.zig");
/// Highest vector we install a gate/stub for (exceptions 0-31 plus the device
/// range 32-47, which covers the timer and the spurious vector).
const gate_count = 48;
/// A device-interrupt handler. It doesn't get the trap frame (a timer or keyboard
/// handler doesn't need the interrupted registers); add that if one ever does.
pub const Handler = *const fn () void;
var handlers = [_]?Handler{null} ** 256;
/// Register `handler` for a device-interrupt `vector` (>= 32).
pub fn setHandler(vector: usize, handler: Handler) void {
handlers[vector] = handler;
}
/// The ring-3 system_call gate's vector (`int $0x80`, the classic choice — well away
/// from the device range) and its handler. Unlike device handlers, a system_call
/// handler gets the (mutable) trap frame: it reads its arguments from the saved
/// user registers and writes rax as the return value, which isr_common then
/// restores into the user context.
pub const system_call_vector = 128;
var system_call_handler: ?*const fn (*CpuState) void = null;
pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void {
system_call_handler = handler;
}
/// The register + trap frame the ISR stubs build on the stack, laid out so the
/// lowest address (where RSP points when we call the handler) is the first field.
/// See the push order in `isrCommon` below.
pub const CpuState = extern struct {
r15: u64,
r14: u64,
r13: u64,
r12: u64,
r11: u64,
r10: u64,
r9: u64,
r8: u64,
rbp: u64,
rdi: u64,
rsi: u64,
rdx: u64,
rcx: u64,
rbx: u64,
rax: u64,
vector: u64, // pushed by the per-vector stub
error_code: u64, // real one from the CPU, or 0 pushed by the stub
rip: u64, // from here down: pushed by the CPU on entry
cs: u64,
rflags: u64,
rsp: u64,
ss: u64,
};
/// Where a fault is reported. The kernel overrides this (see setFaultHandler) with
/// something that prints to the console; until then, just stop.
pub var on_fault: *const fn (*const CpuState) noreturn = defaultFault;
fn defaultFault(_: *const CpuState) noreturn {
while (true) asm volatile ("hlt");
}
/// Names for the 32 defined exception vectors, for readable output.
const names = [_][]const u8{
"divide error", "debug",
"NMI", "breakpoint",
"overflow", "bound range exceeded",
"invalid opcode", "device not available",
"double fault", "coprocessor segment overrun",
"invalid TSS", "segment not present",
"stack-segment fault", "general protection fault",
"page fault", "reserved (15)",
"x87 floating-point", "alignment check",
"machine check", "SIMD floating-point",
"virtualization", "control protection",
"reserved (22)", "reserved (23)",
"reserved (24)", "reserved (25)",
"reserved (26)", "reserved (27)",
"hypervisor injection", "VMM communication",
"security exception", "reserved (31)",
};
pub fn vectorName(vector: u64) []const u8 {
return if (vector < names.len) names[vector] else "unknown";
}
/// A 64-bit IDT gate descriptor (16 bytes).
const Gate = packed struct {
offset_low: u16,
selector: u16,
ist: u8, // interrupt-stack-table index; 0 = use the current stack
flags: u8, // present, DPL, gate type
offset_mid: u16,
offset_high: u32,
reserved: u32 = 0,
};
var idt = [_]Gate{std.mem.zeroes(Gate)} ** 256;
const Descriptor = packed struct {
limit: u16,
base: u64,
};
/// Loads the IDT (`lidt`). Defined in isr.s.
extern fn idt_flush(descriptor: *const Descriptor) callconv(.c) void;
fn setGate(vector: usize, handler: u64) void {
idt[vector] = .{
.offset_low = @truncate(handler),
.selector = gdt.kernel_code,
.ist = 0,
.flags = 0x8E, // present, ring 0, 64-bit interrupt gate
.offset_mid = @truncate(handler >> 16),
.offset_high = @truncate(handler >> 32),
};
}
/// Point every installed vector at its stub (isr.s) and load the IDT.
pub fn init() void {
@setEvalBranchQuota(20000); // comptimePrint across all the gates adds up
inline for (0..gate_count) |vector| {
const stub = @extern(*const anyopaque, .{ .name = std.fmt.comptimePrint("isr{d}", .{vector}) });
setGate(vector, @intFromPtr(stub));
}
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
idt[8].ist = tss.double_fault_ist;
// The system_call gate. Installed outside the 0..gate_count loop (stubs 48-127
// don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a
// #GP. An interrupt gate (not trap): IF is cleared for the handler, which
// the ring-3 exit path relies on.
const system_call_stub = @extern(*const anyopaque, .{ .name = "isr128" });
setGate(system_call_vector, @intFromPtr(system_call_stub));
idt[system_call_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate
loadOnThisCpu();
}
/// Load the (shared, already-populated) IDT on the current core. The gate table is
/// read-only after `init`, so every core points its IDTR at the same one. Called by
/// the BSP via `init` and by each AP during bring-up.
pub fn loadOnThisCpu() void {
const descriptor = Descriptor{
.limit = @sizeOf(@TypeOf(idt)) - 1,
.base = @intFromPtr(&idt),
};
idt_flush(&descriptor);
}
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
/// assembly stubs can `call` it by name. Exceptions are terminal; device
/// interrupts run their handler, get acknowledged, and return.
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
if (state.vector < 32) {
on_fault(state); // CPU exception — never returns
} else if (state.vector == system_call_vector) {
// Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI.
if (system_call_handler) |handler| handler(state);
} else if (handlers[state.vector]) |handler| {
// The handler owns its EOI. It used to be issued here, before the call —
// correct for the LAPIC timer, but impossible to reconcile with a
// level-triggered device line, which must be **masked at the I/O APIC
// before** it is acknowledged or it redelivers instantly and storms
// (the driver that would quiet it lives in ring 3 and hasn't run yet).
// Only the handler knows which discipline its source needs, so only the
// handler can sequence it. See apic.timerTick and irq.dispatch.
handler();
}
// else: spurious/unhandled device interrupt — don't acknowledge it
}
const std = @import("std");
+68
View File
@@ -0,0 +1,68 @@
//! x86 port I/O and model-specific registers — the low-level primitives the
//! serial port and the APIC talk to hardware through.
pub fn outb(port: u16, value: u8) void {
asm volatile ("outb %[value], %[port]"
:
: [value] "{al}" (value),
[port] "{dx}" (port),
);
}
pub fn inb(port: u16) u8 {
return asm volatile ("inb %[port], %[value]"
: [value] "={al}" (-> u8),
: [port] "{dx}" (port),
);
}
pub fn outw(port: u16, value: u16) void {
asm volatile ("outw %[value], %[port]"
:
: [value] "{ax}" (value),
[port] "{dx}" (port),
);
}
pub fn inw(port: u16) u16 {
return asm volatile ("inw %[port], %[value]"
: [value] "={ax}" (-> u16),
: [port] "{dx}" (port),
);
}
pub fn outl(port: u16, value: u32) void {
asm volatile ("outl %[value], %[port]"
:
: [value] "{eax}" (value),
[port] "{dx}" (port),
);
}
pub fn inl(port: u16) u32 {
return asm volatile ("inl %[port], %[value]"
: [value] "={eax}" (-> u32),
: [port] "{dx}" (port),
);
}
/// Read a model-specific register (returns edx:eax combined).
pub fn rdmsr(msr: u32) u64 {
var low: u32 = undefined;
var high: u32 = undefined;
asm volatile ("rdmsr"
: [low] "={eax}" (low),
[high] "={edx}" (high),
: [msr] "{ecx}" (msr),
);
return (@as(u64, high) << 32) | low;
}
pub fn wrmsr(msr: u32, value: u64) void {
asm volatile ("wrmsr"
:
: [msr] "{ecx}" (msr),
[low] "{eax}" (@as(u32, @truncate(value))),
[high] "{edx}" (@as(u32, @truncate(value >> 32))),
);
}
@@ -0,0 +1,152 @@
//! I/O APIC — routes external device interrupts (a device's line) to a LAPIC
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
//! ACPI's MADT (via discovery), never assumed.
//!
//! `init` maps the I/O APIC and **masks every input** — the correct quiescent state
//! on a legacy-free machine. Lines are then unmasked one at a time, as user-space
//! drivers bind them (`routeGsi`/`unmaskGsi`, driven by system/kernel/irq.zig).
//!
//! Two entry points, for two kinds of caller. `routeIrq` takes a legacy **ISA IRQ**
//! and resolves it through the MADT overrides — for in-kernel use, and still without
//! a caller. `routeGsi` takes a **GSI** directly, which is what a device's own
//! routing capability names (e.g. the HPET's `Tn_INT_ROUTE_CAP`), and is the path a
//! bound driver interrupt takes.
const paging = @import("paging.zig");
/// A MADT Interrupt Source Override: an ISA IRQ that appears at a different global
/// system interrupt, with its own polarity/trigger (MPS INTI `flags`).
pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
var base: u64 = 0; // 0 = no I/O APIC discovered
var gsi_base: u32 = 0;
var maximum_entries: u32 = 0;
var overrides: [16]IsoEntry = undefined;
var override_count: usize = 0;
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
const register_ioregsel = 0x00;
const register_iowin = 0x10;
const register_version = 0x01;
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
const redir_mask = 1 << 16; // mask bit in the low dword
/// Supply the discovered I/O APIC location + the MADT IRQ overrides. Call before `init`.
pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) void {
base = ioapic_base;
gsi_base = ioapic_gsi_base;
override_count = @min(isos.len, overrides.len);
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
}
fn registerRead(index: u32) u32 {
@as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index;
return @as(*volatile u32, @ptrFromInt(base + register_iowin)).*;
}
fn registerWrite(index: u32, value: u32) void {
@as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index;
@as(*volatile u32, @ptrFromInt(base + register_iowin)).* = value;
}
fn writeEntry(n: u32, low: u32, high: u32) void {
registerWrite(redir_base + 2 * n, low);
registerWrite(redir_base + 2 * n + 1, high);
}
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
pub fn init() void {
if (base == 0) return;
// Reach the I/O APIC through the physmap; switch `base` to that virtual
// address so the register accessors work without the identity map.
base = paging.mapMmio(base, 0x1000, true);
maximum_entries = ((registerRead(register_version) >> 16) & 0xFF) + 1;
var n: u32 = 0;
while (n < maximum_entries) : (n += 1) writeEntry(n, redir_mask, 0);
}
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
/// for its GSI/polarity/trigger, and unmask it. No caller yet — groundwork for the
/// first device driver.
pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
if (base == 0) return;
var gsi: u32 = irq;
var flags: u16 = 0;
for (overrides[0..override_count]) |o| {
if (o.source == irq) {
gsi = o.gsi;
flags = o.flags;
}
}
if (gsi < gsi_base) return;
const n = gsi - gsi_base;
if (n >= maximum_entries) return;
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
var low: u32 = vector;
if (flags & 0x3 == 3) low |= (1 << 13);
if ((flags >> 2) & 0x3 == 3) low |= (1 << 15);
const high: u32 = @as(u32, apic_id) << 24; // destination APIC ID
writeEntry(n, low, high);
}
// --- GSI-level control (the user-space driver path) --------------------------
//
// `routeIrq` above takes an *ISA IRQ* and resolves it through the MADT overrides.
// A driver-bound interrupt is already a **GSI** (the device told us so, e.g. the
// HPET's `Tn_INT_ROUTE_CAP`), so it needs no override lookup — just the redirection
// entry. These three are what `system/kernel/irq.zig` drives.
//
// Callers must serialise: the I/O APIC is reached through an index/data register
// pair, so two cores interleaving `registerWrite` would corrupt each other. The kernel
// holds the big lock across these.
/// Redirection-entry index for `gsi`, or null if this I/O APIC doesn't own it.
fn entryFor(gsi: u32) ?u32 {
if (base == 0 or gsi < gsi_base) return null;
const n = gsi - gsi_base;
return if (n < maximum_entries) n else null;
}
/// True if `gsi` lands on this I/O APIC — the kernel's validity check before binding.
pub fn ownsGsi(gsi: u32) bool {
return entryFor(gsi) != null;
}
/// Point `gsi` at `vector` on the LAPIC `apic_id`, with explicit polarity/trigger,
/// and leave it **masked**. The caller unmasks once a handler is bound — otherwise a
/// device asserting between route and bind would fire into a null handler.
pub fn routeGsi(gsi: u32, vector: u8, apic_id: u8, level: bool, active_low: bool) void {
const n = entryFor(gsi) orelse return;
var low: u32 = @as(u32, vector) | redir_mask; // masked until bound
if (active_low) low |= (1 << 13);
if (level) low |= (1 << 15);
writeEntry(n, low, @as(u32, apic_id) << 24);
}
/// Stop `gsi` reaching any CPU. Called from the ISR *before* the LAPIC EOI: a
/// level-triggered line is still asserted at that point, so an unmasked entry would
/// redeliver immediately and storm before the user-space driver ever runs.
pub fn maskGsi(gsi: u32) void {
const n = entryFor(gsi) orelse return;
registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) | redir_mask);
}
/// Let `gsi` through again — the tail of `irq_ack`, once the driver has quieted the
/// device (so the line is deasserted and this can't immediately refire).
pub fn unmaskGsi(gsi: u32) void {
const n = entryFor(gsi) orelse return;
registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) & ~@as(u32, redir_mask));
}
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
pub fn entryCount() u32 {
return maximum_entries;
}
/// The low dword of redirection entry `n` — for diagnostics/read-back.
pub fn entryLow(n: u32) u32 {
if (base == 0) return 0;
return registerRead(redir_base + 2 * n);
}
+382
View File
@@ -0,0 +1,382 @@
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
# helpers. Kept in a dedicated assembly file rather than inline asm because these
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
#
# Each exception vector normalises the stack to a uniform trap frame — a dummy
# error code where the CPU pushes none, then the vector number — and jumps to the
# shared tail, which saves the general registers and calls the Zig handler with a
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
.text
# _start: the kernel entry. The loader jumps here (higher-half address) with
# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned
# stack in .bss (the loader stack is a low address that goes away once the low
# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never
# returns; the hlt loop is a belt-and-braces backstop.
.global _start
_start:
leaq bootstrap_stack_top(%rip), %rsp
call kmainEntry
1: hlt
jmp 1b
# The kernel's initial stack (used until the scheduler hands each task its own).
# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it
# needs more than a token stack. Lives in .bss (zeroed, higher-half).
.section .bss
.balign 16
bootstrap_stack:
.skip 65536
bootstrap_stack_top:
.text
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
# registers to the data selector, and reload CS to the code selector. CS can't be
# set with mov, so we far-return through the caller's own return address.
.global gdt_flush
gdt_flush:
lgdt (%rdi)
mov $0x10, %ax # kernel data selector
mov %ax, %ds
mov %ax, %es
mov %ax, %ss
mov %ax, %fs
mov %ax, %gs
pop %rax # caller's return address
push $0x08 # kernel code selector (new CS)
push %rax # return address (new RIP)
lretq
# idt_flush(rdi = *IDT descriptor): load the IDT.
.global idt_flush
idt_flush:
lidt (%rdi)
ret
# load_tr(di = TSS selector): load the task register.
.global load_tr
load_tr:
ltr %di
ret
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
# Cooperative context switch: save the callee-saved registers on the current
# stack, stash the stack pointer in the old task, load the new task's stack
# pointer, restore its callee-saved registers, and return into it. Caller-saved
# registers are the compiler's responsibility (this looks like a normal call).
.global switch_context
switch_context:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
mov %rsi, %rsp # switch to the new task's stack
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret # return into the new task's saved instruction pointer
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
# leaves its entry function in r15. A fresh task is switched to with the big kernel
# lock held (the hand-off rule in sync.zig) but has no enter/leave frame of its own,
# so it releases the lock here before running its body. r15 survives the call (it's
# callee-saved). New tasks then start with interrupts enabled.
.extern releaseForFreshTask
.global task_trampoline
task_trampoline:
call releaseForFreshTask # drop the kernel lock we inherited across the switch
sti
call *%r15 # call the task entry (fn() void)
1: hlt # if the entry returns, idle (still preemptible)
jmp 1b
# user_task_trampoline: the first thing a freshly-spawned *user* task runs.
# init_user_task_stack leaves the user entry in r15 and the user stack in r14
# (both callee-saved, so they survive the lock-release call). Like task_trampoline
# it drops the inherited kernel lock, then — instead of calling a kernel fn — it
# builds an iretq frame and drops to ring 3. The scheduler's switchTo already
# loaded this task's address space (CR3) and published its kernel stack
# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls
# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed).
# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context
# to ring 3, never returning. The scheduler calls this from a fresh user task's
# trampoline (after the lock is released and the entry/stack read from the Task).
# cli guards the swapgs..iretq window: an interrupt there would run in ring 0
# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set)
# re-enables interrupts on the drop to ring 3.
.global jump_to_user
jump_to_user:
cli
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
swapgs # user GS base (isr_common/syscall swap back on entry)
iretq
# --- ring 3 entry/exit ------------------------------------------------------
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames
# grow safely into it — then builds the 5-word iretq frame with the user
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
# timer keeps running in user mode.
.global enter_user
enter_user:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
mov %rsp, %gs:0 # kernel_rsp: the syscall stub stacks here too
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
swapgs # user GS base for ring 3 (isr_common swaps back)
iretq
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
# and stay off — the Zig caller re-enables.
.global user_exit_to_kernel
user_exit_to_kernel:
mov user_saved_rsp(%rip), %rsp
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret
# syscall_entry: the target of the SYSCALL instruction (LSTAR). The CPU does NOT
# switch stacks — it puts the return RIP in RCX, the saved RFLAGS in R11, loads
# CS/SS from STAR, masks RFLAGS with SFMASK (so IF is already clear), and jumps
# here with RSP still the *user* stack. We swap in the kernel GS, switch to the
# task's kernel stack via the per-CPU block, build a CpuState frame identical to
# the interrupt path's, and reuse interruptDispatch (vector 128) — then SYSRET.
#
# Hazard (acceptable while init is the only, trusted, user program): SYSRETQ #GPs
# in ring 0 if the return RIP (RCX) is non-canonical. A hostile user could arrange
# that; hardening (canonical check / iretq fallback) is a later security-track item.
.global syscall_entry
syscall_entry:
swapgs # kernel GS base
movq %rsp, %gs:8 # stash user rsp in the scratch slot
movq %gs:0, %rsp # switch to this task's kernel stack
# Build the trap frame (same field order as isr_common), highest field first.
pushq $0x1B # ss (user data | 3)
pushq %gs:8 # rsp (user, from scratch)
pushq %r11 # rflags (saved by syscall)
pushq $0x23 # cs (user code | 3)
pushq %rcx # rip (saved by syscall)
pushq $0 # error_code (none for a syscall)
pushq $128 # vector (same as the int 0x80 gate)
push %rax
push %rbx
push %rcx
push %rdx
push %rsi
push %rdi
push %rbp
push %r8
push %r9
push %r10
push %r11
push %r12
push %r13
push %r14
push %r15
mov %rsp, %rdi # trap-frame pointer
call interruptDispatch
pop %r15
pop %r14
pop %r13
pop %r12
pop %r11
pop %r10
pop %r9
pop %r8
pop %rbp
pop %rdi
pop %rsi
pop %rdx
pop %rcx
pop %rbx
pop %rax
add $16, %rsp # drop vector + error_code -> rsp at rip
popq %rcx # rip -> RCX (SYSRETQ restores RIP from RCX)
addq $8, %rsp # skip the cs slot (SYSRETQ loads CS from STAR)
popq %r11 # rflags -> R11 (SYSRETQ restores RFLAGS from R11)
popq %rsp # user rsp (the ss slot below is abandoned)
swapgs # user GS base
sysretq # -> ring 3: RIP=RCX, RFLAGS=R11, CS/SS from STAR
.section .bss
.balign 8
user_saved_rsp:
.skip 8
.text
# --- user-mode test program --------------------------------------------------
# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and
# entered via enter_user. Position-independent (immediates and short jumps only).
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
# from the user mapping. (The old hello/ping blob was retired once /sbin/init
# became the real ring-3 exerciser; only the isolation proof remains.)
.section .rodata
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
# page, so this must take a #PF with error code 0x5 (present | user) before any
# access happens. movabs loads the full 64-bit higher-half address (a disp32
# would sign-extend and miss).
.global user_pf_start
.global user_pf_end
user_pf_start:
movabs $0xFFFF8800FEE00000, %rcx
mov (%rcx), %rax
1: jmp 1b
user_pf_end:
.text
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
.macro STUB_NOERR vec
.global isr\vec
isr\vec:
pushq $0
pushq $\vec
jmp isr_common
.endm
# Stub for a vector the CPU DOES push an error code for: leave it in place.
.macro STUB_ERR vec
.global isr\vec
isr\vec:
pushq $\vec
jmp isr_common
.endm
STUB_NOERR 0
STUB_NOERR 1
STUB_NOERR 2
STUB_NOERR 3
STUB_NOERR 4
STUB_NOERR 5
STUB_NOERR 6
STUB_NOERR 7
STUB_ERR 8
STUB_NOERR 9
STUB_ERR 10
STUB_ERR 11
STUB_ERR 12
STUB_ERR 13
STUB_ERR 14
STUB_NOERR 15
STUB_NOERR 16
STUB_ERR 17
STUB_NOERR 18
STUB_NOERR 19
STUB_NOERR 20
STUB_ERR 21
STUB_NOERR 22
STUB_NOERR 23
STUB_NOERR 24
STUB_NOERR 25
STUB_NOERR 26
STUB_NOERR 27
STUB_NOERR 28
STUB_NOERR 29
STUB_NOERR 30
STUB_NOERR 31
# Device-interrupt vectors (timer, spurious, room for more). None push an error
# code, so they all use the dummy-zero form.
STUB_NOERR 32
STUB_NOERR 33
STUB_NOERR 34
STUB_NOERR 35
STUB_NOERR 36
STUB_NOERR 37
STUB_NOERR 38
STUB_NOERR 39
STUB_NOERR 40
STUB_NOERR 41
STUB_NOERR 42
STUB_NOERR 43
STUB_NOERR 44
STUB_NOERR 45
STUB_NOERR 46
STUB_NOERR 47
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
# vector; dispatched specially in interruptDispatch.
STUB_NOERR 128
.extern interruptDispatch
# Shared tail. Register push order here defines the CpuState field order.
# If the interrupt came from ring 3 the GS base holds the user's value, so swap
# in the kernel's before anything reads per-CPU data (swapgs discipline; see
# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24).
isr_common:
testb $3, 24(%rsp)
jz 1f
swapgs
1: push %rax
push %rbx
push %rcx
push %rdx
push %rsi
push %rdi
push %rbp
push %r8
push %r9
push %r10
push %r11
push %r12
push %r13
push %r14
push %r15
mov %rsp, %rdi # first argument: pointer to the trap frame
call interruptDispatch
pop %r15
pop %r14
pop %r13
pop %r12
pop %r11
pop %r10
pop %r9
pop %r8
pop %rbp
pop %rdi
pop %rsi
pop %rdx
pop %rcx
pop %rbx
pop %rax
add $16, %rsp # drop the vector and error code
# Symmetric to entry: if returning to ring 3, restore the user GS base. CS is
# now at offset 8 (RIP@0, CS@8).
testb $3, 8(%rsp)
jz 1f
swapgs
1: iretq
@@ -0,0 +1,52 @@
/* Kernel link layout — higher half.
*
* The kernel is linked to *run* in the higher half (virtual base
* 0xFFFF_FFFF_8000_0000, matching danos.kernel_virt_base and build.zig's
* image_base) but is *loaded* low. Each section's load address (LMA) is its
* virtual address minus KERNEL_VIRT_BASE via AT(), so the ELF's p_paddr lands
* at a low physical address (.text at 1 MiB) that the loader can allocate and
* copy into. The loader maps p_vaddr (high) -> p_paddr (low) in its bootstrap
* tables and jumps to the high entry; the kernel then builds its own tables
* with the physmap and abandons the identity map. Requires LLD (build.zig pins
* it) — the self-hosted linker ignores PHDRS/AT()/section order.
*/
KERNEL_VIRT_BASE = 0xFFFFFFFF80000000;
ENTRY(_start)
/* One loadable segment per permission set, so the loader can map .text as R+X,
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
PHDRS {
text PT_LOAD FLAGS(5); /* R + X */
rodata PT_LOAD FLAGS(4); /* R */
data PT_LOAD FLAGS(6); /* R + W */
}
SECTIONS {
.text ALIGN(4K) : AT(ADDR(.text) - KERNEL_VIRT_BASE) {
*(.text .text.*)
} :text
.rodata ALIGN(4K) : AT(ADDR(.rodata) - KERNEL_VIRT_BASE) {
*(.rodata .rodata.*)
} :rodata
.data ALIGN(4K) : AT(ADDR(.data) - KERNEL_VIRT_BASE) {
*(.data .data.*)
} :data
/* .bss occupies memory but not file space. The loader zeroes it via the
* gap between each PT_LOAD segment's file size and memory size, so no
* boundary symbols are needed here. */
.bss ALIGN(4K) : AT(ADDR(.bss) - KERNEL_VIRT_BASE) {
*(.bss .bss.*)
*(COMMON)
} :data
/DISCARD/ : {
*(.comment)
*(.note .note.*)
*(.eh_frame .eh_frame_hdr)
}
}
@@ -0,0 +1,389 @@
//! The kernel's page tables and virtual memory manager.
//!
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
//! which the kernel heap will build on.
//!
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
//! negligible against available RAM.
const danos = @import("danos");
const io = @import("io.zig");
const page_size = danos.page_size;
// Page-table entry bits.
const present: u64 = 1 << 0;
const writable: u64 = 1 << 1;
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
const pwt: u64 = 1 << 3; // page write-through
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
const no_execute: u64 = 1 << 63;
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
// ELF segment flags (p_flags).
const pf_x: u32 = 1;
const pf_w: u32 = 2;
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
var kernel_pml4: u64 = 0;
var alloc_frame: *const fn () ?u64 = undefined;
var free_frame: *const fn (u64) void = undefined; // for tearing down address spaces
/// Set once the kernel is running on its own tables (past the CR3 load in
/// `init`). Before that, the kernel reaches page-table frames through the
/// *loader's* bootstrap physmap, which only covers the low 4 GiB — so every
/// frame allocated for a table during that window must be below 4 GiB. Both the
/// frame allocator and this code scan from low addresses up, so it holds
/// naturally; the assertion in `allocTable` makes a violation loud rather than
/// a silent fault. After the switch the kernel's own physmap covers all RAM.
var on_own_tables = false;
/// Set at the end of `init`. Guards against a new *higher-half* PML4 entry being
/// created afterward: the kernel half is pre-populated at init and then shared
/// by copying PML4[256..512) into every process address space (M3), so a late
/// top-half entry would be invisible to already-created address spaces.
var init_done = false;
const bootstrap_physmap_limit: u64 = 4 << 30;
/// Dereference a page-table frame by its physical address, via the physmap.
/// This is the single hinge for the higher-half move: page tables hold physical
/// frame addresses (pmm gives out physical frames, and CR3/PTEs must be
/// physical), but the kernel reaches them at `physmap_base + physical`. Valid under
/// both the loader's bootstrap tables and the kernel's own, which share the
/// physmap base.
fn tableAt(physical: u64) *[512]u64 {
return @ptrFromInt(danos.physicalToVirtual(physical));
}
fn allocTable() u64 {
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
if (!on_own_tables and frame >= bootstrap_physmap_limit)
@panic("paging: table frame above the 4 GiB bootstrap physmap");
@memset(tableAt(frame)[0..], 0);
return frame;
}
/// Return the table an entry points at, creating it if empty. Intermediate
/// entries are writable and executable so the leaf's bits govern (a page is
/// writable only if every level is; non-executable if any level is).
fn descend(entry: *u64) u64 {
if (entry.* & present != 0) return entry.* & address_mask;
const frame = allocTable();
entry.* = frame | present | writable;
return frame;
}
/// Map one 4 KiB page `virtual` -> `physical` with `flags` (present is added).
fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
// The kernel half is fixed after init: every top-half PML4 entry is
// pre-created so address spaces can share it by copying these slots. A new
// one here would be invisible to address spaces already made.
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
@panic("paging: new higher-half PML4 entry after init");
const pdpt = descend(pml4e);
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
const pd = descend(pdpte);
const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF];
const pt = descend(pde);
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
}
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
/// window onto physical memory once the low identity map goes away.
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
var address = physical_base & ~@as(u64, page_size - 1);
const end = physical_base + len;
while (address < end) : (address += page_size) {
mapPage(pml4, danos.physicalToVirtual(address), address, flags);
}
}
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(mm.regions)))[0..mm.len];
}
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
fn enableNx() void {
const efer_msr = 0xC0000080;
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
}
/// Build the address space and switch onto it.
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
alloc_frame = allocFrame;
free_frame = freeFrame;
enableNx();
const pml4 = allocTable();
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
// mapping: the low half belongs to user space. MMIO is skipped here and
// mapped on demand (mapMmio) or explicitly below.
for (regions(boot_information.memory_map)) |r| {
if (r.kind == .mmio) continue;
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
}
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
// the kernel touches directly), RW + NX.
const fb = boot_information.framebuffer;
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
mapPage(pml4, danos.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
// 3. The kernel's own segments at their higher-half link addresses, mapped
// to their low physical load addresses with real ELF permissions: code
// R+X, rodata R, data R+W+NX. This is the W^X guarantee.
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
var flags: u64 = present;
if (seg.flags & pf_w != 0) flags |= writable;
if (seg.flags & pf_x == 0) flags |= no_execute;
var off: u64 = 0;
while (off < seg.pages * page_size) : (off += page_size) {
mapPage(pml4, seg.virtual + off, seg.physical + off, flags);
}
}
// 4. Pre-create every higher-half PML4 entry (an empty PDPT where none
// exists yet), so the whole kernel half is a fixed set of top-level
// slots. A process address space (M3) then shares the kernel half simply
// by copying PML4[256..512) — growth beneath these slots (heap, on-demand
// MMIO) propagates to every address space because they share the PDPTs.
for (256..512) |i| {
const e = &tableAt(pml4)[i];
if (e.* & present == 0) e.* = allocTable() | present | writable;
}
kernel_pml4 = pml4;
asm volatile ("mov %[pml4], %%cr3"
:
: [pml4] "r" (pml4),
: .{ .memory = true }
);
on_own_tables = true; // now on the kernel's physmap (covers all RAM)
init_done = true; // the kernel half is fixed from here
}
/// The kernel's own top-level page table (physical). Every kernel task and every
/// per-process address space shares this table's higher half.
pub fn kernelPml4() u64 {
return kernel_pml4;
}
/// Load CR3 (switch the active address space). `pml4` is a physical frame.
pub fn loadCr3(pml4: u64) void {
asm volatile ("mov %[pml4], %%cr3"
:
: [pml4] "r" (pml4),
: .{ .memory = true });
}
/// Map a page into the kernel address space on demand (for the heap, etc.).
/// `writable_page` controls W; pages are always mapped non-executable.
pub fn map(virtual: u64, physical: u64, writable_page: bool) void {
var flags: u64 = present | no_execute;
if (writable_page) flags |= writable;
mapPage(kernel_pml4, virtual, physical, flags);
invalidate(virtual);
}
/// Map a device MMIO range into the physmap and return the virtual address to
/// use for it (physicalToVirtual(physical)). The single way the kernel (and the device
/// layer, via the HAL) reaches memory-mapped registers once the identity map is
/// gone: physmap pages are RW + NX, so a driver never executes device memory.
/// Idempotent for already-mapped ranges. `len` 0 maps one page.
pub fn mapMmio(physical: u64, len: u64, writable_page: bool) u64 {
var flags: u64 = present | no_execute;
if (writable_page) flags |= writable;
const first = physical & ~@as(u64, page_size - 1);
const last = physical + (if (len == 0) 1 else len) - 1;
var address = first;
while (address <= (last & ~@as(u64, page_size - 1))) : (address += page_size) {
const virtual = danos.physicalToVirtual(address);
mapPage(kernel_pml4, virtual, address, flags);
invalidate(virtual);
}
return danos.physicalToVirtual(physical);
}
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves
/// existing entries untouched. Only used under user-exclusive virtual ranges, so
/// no kernel mapping's protection is widened (the leaf still governs).
fn descendUser(entry: *u64) u64 {
const table = descend(entry);
entry.* |= user;
return table;
}
/// Map one 4 KiB page `virtual` -> `physical` accessible from ring 3. W^X is the
/// caller's contract: code pages are read-only + executable, data pages are
/// writable + no-execute. `virtual` must lie in a user-exclusive region (see
/// `descendUser`).
pub fn mapUser(virtual: u64, physical: u64, writable_page: bool, executable: bool) void {
mapUserInto(kernel_pml4, virtual, physical, writable_page, executable);
}
/// Map a ring-3-accessible page into the address space rooted at `pml4` (which
/// may be a process's own table or the kernel's). W^X is the caller's contract.
pub fn mapUserInto(pml4: u64, virtual: u64, physical: u64, writable_page: bool, executable: bool) void {
var flags: u64 = present | user;
if (writable_page) flags |= writable;
if (!executable) flags |= no_execute;
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
const pdpt = descendUser(pml4e);
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
const pd = descendUser(pdpte);
const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF];
const pt = descendUser(pde);
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags;
invalidate(virtual);
}
/// Map a device MMIO window `[physical, physical+len)` into the user (low) half of the
/// address space rooted at `pml4`, page by page. Unlike `mapUserInto` these pages
/// are **strong-uncacheable** (PCD|PWT — device registers must not be cached) and
/// carry the `device_grant` bit so teardown does not return the MMIO frames to the
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
const first = physical & ~@as(u64, page_size - 1);
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
var off: u64 = 0;
while (first + off <= last) : (off += page_size) {
const v = virtual + off;
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
const pdpt = descendUser(pml4e);
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
const pd = descendUser(pdpte);
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
const pt = descendUser(pde);
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
invalidate(v);
}
}
/// Create a new address space: a fresh PML4 with an empty user half and the
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
/// the kernel half propagates to every address space). Returns the physical
/// PML4, or null if out of frames.
pub fn createAddressSpace() ?u64 {
const pml4 = alloc_frame() orelse return null;
const t = tableAt(pml4);
@memset(t[0..256], 0); // empty user half
@memcpy(t[256..512], tableAt(kernel_pml4)[256..512]); // shared kernel half
return pml4;
}
/// Tear down an address space created by `createAddressSpace`: free every frame
/// and table in the user half [0..256), then the PML4 itself. The shared kernel
/// half [256..512) is never touched. The caller must not be running on `pml4`.
pub fn destroyAddressSpace(pml4: u64) void {
const t = tableAt(pml4);
for (0..256) |i| {
if (t[i] & present != 0) freeSubtree(t[i] & address_mask, 3); // PDPT level
}
free_frame(pml4);
}
/// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At
/// level 1 the entries are leaf data frames; above, they are child tables.
fn freeSubtree(physical: u64, level: u32) void {
const t = tableAt(physical);
for (t) |e| {
if (e & present == 0) continue;
if (level > 1) {
freeSubtree(e & address_mask, level - 1);
} else if (e & device_grant == 0) {
// A device-grant leaf points at MMIO, not RAM — returning it to the
// frame allocator would corrupt the pool. Only reclaim real RAM.
free_frame(e & address_mask);
}
}
free_frame(physical); // page-table frames are always real RAM
}
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
/// case). Returns false if unmapped. Used for W^X checks in tests.
pub fn isExecutable(virtual: u64) bool {
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
if (pml4e & present == 0) return false;
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
if (pdpte & present == 0) return false;
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
if (pde & present == 0) return false;
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
if (pte & present == 0) return false;
return pte & no_execute == 0;
}
/// Make an already-identity-mapped RAM page **executable** (clear its NX bit),
/// leaving it present and writable. The blanket RAM mapping is NX for W^X, but the
/// application processors fetch the AP trampoline from a low RAM page under paging —
/// so that one page must be executable. A deliberate, temporary W^X exception for a
/// single bring-up page; the caller frees it once every AP is up.
pub fn setExecutable(physical: u64) void {
mapPage(kernel_pml4, physical, physical, present | writable); // note: no no_execute
invalidate(physical);
}
/// Remove a mapping and flush it from the TLB.
pub fn unmap(virtual: u64) void {
unmapInto(kernel_pml4, virtual);
}
/// Remove a mapping from the address space rooted at `pml4` (a process's own
/// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE —
/// the intermediate tables and any frame the PTE pointed at are left to the
/// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables).
pub fn unmapInto(pml4: u64, virtual: u64) void {
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
if (pml4e & present == 0) return;
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
if (pdpte & present == 0) return;
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
if (pde & present == 0) return;
tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF] = 0;
invalidate(virtual);
}
/// Resolve a virtual address to a physical one in the address space rooted at
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
/// any address space, not just the live one). Returns null if `virtual` is not
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
/// case. The foundation for cross-address-space copies and for munmap (which
/// needs the frame behind a user vaddr to free it).
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
if (pml4e & present == 0) return null;
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
if (pdpte & present == 0) return null;
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
if (pde & present == 0) return null;
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
if (pte & present == 0) return null;
return (pte & address_mask) | (virtual & (page_size - 1));
}
fn invalidate(virtual: u64) void {
// invlpg needs its operand via a register-indirect memory reference that Zig
// inline asm won't form directly, so stage the address in a register first.
asm volatile (
\\mov %[v], %%rax
\\invlpg (%%rax)
:
: [v] "r" (virtual),
: .{ .rax = true, .memory = true }
);
}
@@ -0,0 +1,77 @@
//! Per-CPU data reached through the GS segment base. The GS base holds a pointer
//! to this core's `ArchitecturePerCpu`, so kernel code gets the running core's block with
//! a single MSR read (`scheduler()`) and the system_call entry stub gets its kernel stack
//! with a `%gs`-relative load (no usable stack yet at that point).
//!
//! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds
//! the user's own GS (which ring 3 may set freely), and this pointer lives in the
//! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the
//! system_call stub and the conditional swapgs in isr_common) brings it back, and
//! every ring-0 -> ring-3 exit swaps it away. Because the very first ring
//! transition is always an exit (the kernel starts in ring 0), the swap pairs
//! keep the invariant without seeding KERNEL_GS_BASE. `scheduler()` is therefore
//! valid in any ring-0 context and never sees a user-controlled base.
const std = @import("std");
const io = @import("io.zig");
const parameters = @import("parameters");
const ia32_gs_base = 0xC000_0101;
/// Layout is load-bearing: the system_call entry stub in isr.s reaches `kernel_rsp`
/// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below
/// pin the offsets.
pub const ArchitecturePerCpu = extern struct {
kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for system_call entry (== TSS.rsp0)
scratch: u64 = 0, // %gs:8 — stashes the user rsp during system_call entry
scheduler: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns)
};
comptime {
std.debug.assert(@offsetOf(ArchitecturePerCpu, "kernel_rsp") == 0);
std.debug.assert(@offsetOf(ArchitecturePerCpu, "scratch") == 8);
}
var blocks = [_]ArchitecturePerCpu{.{}} ** parameters.maximum_cpus;
/// Publish core `index`'s per-CPU block: record the scheduler pointer and point
/// the GS base at the block. Called once per core during bring-up, after the GDT
/// is loaded (a GS *selector* reload would clobber the base).
pub fn setLocal(index: usize, scheduler_ptr: usize) void {
blocks[index].scheduler = scheduler_ptr;
io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index]));
}
/// The scheduler pointer for the running core (via the GS base). Valid in any
/// ring-0 context under the swapgs discipline.
pub fn scheduler() usize {
return @as(*const ArchitecturePerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).scheduler;
}
/// Record core `index`'s kernel stack top, used by the system_call entry stub to
/// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it
/// switches to a user task.
pub fn setKernelRsp(index: usize, top: usize) void {
blocks[index].kernel_rsp = top;
}
// Fast-system_call MSRs.
const ia32_efer = 0xC000_0080;
const ia32_star = 0xC000_0081;
const ia32_lstar = 0xC000_0082;
const ia32_sfmask = 0xC000_0084;
/// Enable the `system_call`/`sysret` fast path on this core (BSP and each AP). EFER.SCE
/// turns the instructions on; STAR sets the selectors system_call/sysret load; LSTAR
/// is the entry stub (isr.s); SFMASK clears RFLAGS bits on entry (notably IF —
/// the handler runs with interrupts off, like the int-gate path). The GDT is laid
/// out (kernel code 0x08, then user data 0x18 / code 0x20) precisely so these line
/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8
/// with RPL forced to 3, so base 0x10 gives CS 0x23 (user code|3) and SS 0x1B.
pub fn initSystemCall() void {
io.wrmsr(ia32_efer, io.rdmsr(ia32_efer) | 1); // SCE
io.wrmsr(ia32_star, (@as(u64, 0x08) << 32) | (@as(u64, 0x10) << 48));
const entry = @extern(*const anyopaque, .{ .name = "syscall_entry" });
io.wrmsr(ia32_lstar, @intFromPtr(entry));
io.wrmsr(ia32_sfmask, 0x4_0700); // clear IF, TF, DF, AC on entry
}
@@ -0,0 +1,86 @@
//! Serial console (16550-compatible UART) — the kernel's machine-readable output
//! channel. Unlike the framebuffer console, serial text can be captured to a file
//! by QEMU (`-serial file:...`), which is what the test harness asserts on.
//!
//! The UART defaults to the legacy PC COM1 at I/O port `0x3F8`, but a UEFI Class 3
//! (legacy-free) machine may have no COM1 — or its debug UART somewhere else, and
//! reachable via MMIO rather than port I/O. So the location is a runtime value:
//! `reconfigure` repoints it once ACPI's SPCR table has been read. Early boot logs
//! optimistically to COM1 (harmless if absent); the framebuffer console is the
//! always-present log.
const paging = @import("paging.zig");
/// How the UART registers are reached: legacy I/O ports or memory-mapped.
const Access = enum { port, mmio };
var access: Access = .port;
var base: u64 = 0x3F8; // COM1
fn portOut(p: u16, value: u8) void {
asm volatile ("outb %[value], %[p]"
:
: [value] "{al}" (value),
[p] "{dx}" (p),
);
}
fn portIn(p: u16) u8 {
return asm volatile ("inb %[p], %[value]"
: [value] "={al}" (-> u8),
: [p] "{dx}" (p),
);
}
/// Read UART register `off` through the active access method.
fn register(off: u64) u8 {
if (access == .mmio) return @as(*volatile u8, @ptrFromInt(base + off)).*;
return portIn(@intCast(base + off));
}
/// Write UART register `off` through the active access method.
fn setRegister(off: u64, value: u8) void {
if (access == .mmio) {
@as(*volatile u8, @ptrFromInt(base + off)).* = value;
} else {
portOut(@intCast(base + off), value);
}
}
/// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything
/// else; it has no dependencies, and is a harmless no-op if the port is absent.
pub fn init() void {
setRegister(1, 0x00); // disable interrupts
setRegister(3, 0x80); // enable DLAB (set baud divisor)
setRegister(0, 0x03); // divisor low: 38400 baud
setRegister(1, 0x00); // divisor high
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
setRegister(4, 0x0B); // RTS/DSR set
}
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
/// re-run the UART setup there. Called after discovery when an SPCR entry exists.
pub fn reconfigure(is_mmio: bool, address: u64) void {
access = if (is_mmio) .mmio else .port;
// An MMIO UART is reached through the physmap; an I/O-port UART keeps its
// port number unchanged.
base = if (is_mmio) paging.mapMmio(address, 0x100, true) else address;
init();
}
fn writeByte(c: u8) void {
// Wait for the transmit-holding register to empty — but bounded, so an absent
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
var guard: u32 = 0;
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
setRegister(0, c);
}
/// Write bytes, translating LF to CRLF so terminals and logs line up.
pub fn write(bytes: []const u8) void {
for (bytes) |c| {
if (c == '\n') writeByte('\r');
writeByte(c);
}
}
+182
View File
@@ -0,0 +1,182 @@
//! Application-processor (AP) bring-up: waking the cores the firmware left parked.
//!
//! The firmware starts only the bootstrap processor (BSP); the others sit idle until
//! the kernel wakes them with an INIT–SIPI–SIPI sequence (Intel SDM Vol.3, "MP
//! Initialization"). A woken core begins in 16-bit real mode at a low physical page,
//! runs the [trampoline](trampoline.s) up into 64-bit long mode, and lands in
//! `apEntry` here. This module copies the trampoline into place, patches its
//! per-AP parameters, drives the wake IPIs, and waits for each core to report in.
//!
//! Cores are brought up **one at a time**: a single trampoline page and parameter
//! block are reused, so the BSP patches, wakes, and waits for one AP before the
//! next. That also lets `apEntry` pick up its dense CPU index from a plain global.
//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic
//! scheduler entry and joins the run loop — mechanism here, policy there.
const danos = @import("danos");
const io = @import("io.zig");
const gdt = @import("gdt.zig");
const tss = @import("tss.zig");
const idt = @import("idt.zig");
const apic = @import("apic.zig");
const paging = @import("paging.zig");
const pcpu = @import("per-cpu.zig");
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
/// path doesn't depend on cpu.zig and risk an import cycle).
const ia32_gs_base = 0xC000_0101;
const page_size = 0x1000;
/// Physical address of the low (<1 MiB) frame reserved for the trampoline. Held for
/// the life of the system so any core can be (re)woken on demand — a retry, or a
/// future power manager bringing a core back online. The frame is kept **inert**
/// between wakes (zeroed and non-executable) and only armed for the brief moment a
/// core is actually climbing. Its low 20 bits are zero, so `physical >> 12` is the SIPI
/// vector.
var tramp_physical: u64 = 0;
/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own
/// bring-up. The BSP clears it before each wake and polls it afterwards — a simple
/// one-at-a-time handshake (only one AP is being started at any moment).
var ap_alive: u32 = 0;
/// The dense CPU index of the AP currently being started. Set by the BSP before the
/// wake, read by `apEntry` (safe because bring-up is strictly one core at a time).
var boot_index: usize = 0;
/// The generic scheduler entry a woken core jumps to once its architecture state is up. Set
/// by the kernel via `setSecondaryEntry`; never returns.
var secondary_entry: ?*const fn () callconv(.c) noreturn = null;
/// Register the generic entry an AP calls once its per-CPU tables/LAPIC/timer are up.
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
secondary_entry = entry;
}
/// Test hook: force the next `n` wake attempts to fail (skipping the actual
/// INIT-SIPI-SIPI), so the retry path can be exercised deterministically. Zero in
/// normal operation — the smp-retry test arms it via `architecture.testFailNextWakes`.
var fail_next_wakes: u32 = 0;
pub fn testFailNextWakes(n: u32) void {
fail_next_wakes = n;
}
/// Record the reserved low frame the trampoline uses. Call once at boot. The frame
/// starts inert (identity-mapped RW+NX like all RAM); each wake arms it and disarms
/// it again, so it's only ever executable while a core is climbing.
pub fn setTrampolinePage(physical: u64) void {
tramp_physical = physical;
}
/// The reserved trampoline frame (0 if SMP bring-up never ran). Exposed so a test
/// can verify it's inert — zeroed and non-executable — when dormant.
pub fn trampolinePage() u64 {
return tramp_physical;
}
/// Arm the trampoline for a wake: make its page executable (W^X exception for the
/// duration of the climb) and copy the blob in.
fn arm() void {
// The AP executes this page at its physical address (identity) while it
// climbs from real to long mode, so it needs a low identity mapping that is
// executable — the one deliberate, transient W^X exception. The BSP writes
// the blob into the frame through the physmap.
paging.setExecutable(tramp_physical);
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
const len = @intFromPtr(end) - @intFromPtr(start);
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
@memcpy(destination[0..len], start[0..len]);
}
/// Disarm after a wake: wipe the page through the physmap and remove its low
/// identity mapping, so no executable code (nor any stale bytes, nor any
/// low-half mapping) lingers between wakes. Safe once the woken core has
/// reported in — it's long past the trampoline by then, in the kernel image; a
/// core that never answered is dead and can't be mid-climb.
fn disarm() void {
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
@memset(destination[0..page_size], 0);
paging.unmap(tramp_physical); // drop the transient low identity mapping
}
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
/// base plus the field's offset within it (a same-section symbol difference). The
/// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86
/// tolerates unaligned stores, so we don't force layout constraints on the asm.
fn param(comptime name: []const u8) *align(1) volatile u64 {
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
return @ptrFromInt(danos.physicalToVirtual(tramp_physical + (sym - start)));
}
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive. This
/// is one self-contained attempt: it arms the trampoline, drives INIT–SIPI–SIPI, and
/// disarms again before returning — so it's safe to call repeatedly (a retry, or a
/// power manager re-waking a core; the INIT resets a core that was wedged). Returns
/// false if the core doesn't report in within the timeout (left parked, no harm to
/// the running system). `cr3` is the kernel page tables the AP adopts. Precondition:
/// `setTrampolinePage` has run.
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool {
// The trampoline loads CR3 with a 32-bit `movl` before it reaches long mode,
// so the page-table root must be addressable in 32 bits.
if (cr3 >= (1 << 32)) @panic("smp: kernel page tables above 4 GiB");
arm();
defer disarm();
if (fail_next_wakes > 0) { // test hook: simulate a core missing this attempt
fail_next_wakes -= 1;
return false;
}
boot_index = index;
param("ap_tramp_cr3").* = cr3;
param("ap_tramp_stack").* = stack_top;
param("ap_tramp_entry").* = @intFromPtr(&apEntry);
param("ap_tramp_percpu").* = percpu;
@atomicStore(u32, &ap_alive, 0, .seq_cst);
const vector: u8 = @intCast(tramp_physical >> 12);
apic.sendInit(apic_id);
delayMicros(10_000); // 10 ms INIT settle
apic.sendStartup(apic_id, vector);
delayMicros(200);
apic.sendStartup(apic_id, vector);
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
const deadline = apic.millis() + 100;
while (apic.millis() < deadline) {
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
asm volatile ("pause");
}
return false;
}
/// Busy-wait `us` microseconds against the calibrated TSC clock (the AP wake happens
/// after the timer is up, so the clock is available).
fn delayMicros(us: u64) void {
const start = apic.micros();
while (apic.micros() - start < us) asm volatile ("pause");
}
/// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU
/// pointer in RDI. Brings up this core's own descriptor tables, LAPIC and timer,
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
fn apEntry(percpu: usize) callconv(.c) noreturn {
const cpu = boot_index;
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
idt.loadOnThisCpu(); // the shared IDT
pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload
pcpu.initSystemCall(); // enable system_call/sysret on this core
apic.initSecondary(); // software-enable this core's LAPIC
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
while (true) asm volatile ("hlt"); // (only if no entry was registered)
}
@@ -0,0 +1,150 @@
# AP trampoline: brings a waking application processor from the 16-bit real mode it
# starts in (after INIT-SIPI-SIPI) up through protected mode into 64-bit long mode,
# then jumps to the Zig AP entry (arch/x86_64/smp.zig:apEntry).
#
# A STARTUP IPI vectors a core to physical address `vector << 12` in real mode, so
# this blob is copied to a low (<1 MiB) page and started there; at entry CS = that
# page >> 4 and IP = 0. It is fully **position-independent**: it derives its own
# linear base (CS << 4) into EBX and addresses every internal datum as
# `(label - ap_trampoline_start)(%ebx)` — a difference of two symbols in the same
# section, which the assembler folds to a constant page offset no matter where the
# blob was linked or copied to. The BSP patches the parameter block (CR3, stack,
# entry, per-CPU pointer) before each wake; see arch/x86_64/smp.zig.
#
# It lives in .rodata (not .text): it is data to be copied out and executed
# elsewhere, never run at its link address, so it must not be a normal code segment.
.section .rodata
.balign 16
.code16
.global ap_trampoline_start
ap_trampoline_start:
cli
cld
# Linear base of this page (CS << 4) into EBX; all data is addressed off it.
xorl %eax, %eax
mov %cs, %ax
shll $4, %eax
movl %eax, %ebx
mov %cs, %ax # DS = CS, so we address our data as DS:(label - start):
mov %ax, %ds # the segment base (CS<<4) already supplies the page base,
# so data operands use the page *offset*, not EBX.
# Relocate the pointers whose absolute (linear) targets depend on where we were
# copied: the GDT base and the two far-jump targets = EBX + their page offsets.
# EBX supplies the base for the *value* (via leal); the store address is DS-rel.
leal (gdt32 - ap_trampoline_start)(%ebx), %eax
movl %eax, gdtr32_base - ap_trampoline_start
leal (prot_entry - ap_trampoline_start)(%ebx), %eax
movl %eax, jmp32_off - ap_trampoline_start
leal (long_entry - ap_trampoline_start)(%ebx), %eax
movl %eax, jmp64_off - ap_trampoline_start
lgdtl gdtr32 - ap_trampoline_start
movl %cr0, %eax # enter protected mode (CR0.PE)
orl $1, %eax
movl %eax, %cr0
ljmpl *(jmp32_ptr - ap_trampoline_start) # -> prot_entry, CS = 0x08
.code32
prot_entry:
movw $0x10, %ax # flat 32-bit data segments
movw %ax, %ds
movw %ax, %es
movw %ax, %ss
movw %ax, %fs
movw %ax, %gs
# CR4: PAE (required for long mode) + OSFXSR/OSXMMEXCPT. The kernel is built with
# SSE (part of the x86_64 baseline), and the compiler emits SSE for things as
# ordinary as a struct copy — without OSFXSR those instructions #UD. The BSP got
# these bits from UEFI; an AP starts fresh, so we must set them ourselves.
movl %cr4, %eax
orl $((1 << 5) | (1 << 9) | (1 << 10)), %eax
movl %eax, %cr4
# CR0: clear EM (no x87 emulation) and set MP, so SSE/x87 don't fault.
movl %cr0, %eax
andl $~(1 << 2), %eax # ~EM
orl $(1 << 1), %eax # MP
movl %eax, %cr0
movl (param_cr3 - ap_trampoline_start)(%ebx), %eax # kernel page tables
movl %eax, %cr3
movl $0xC0000080, %ecx # EFER: long mode enable (LME) + NX enable (NXE, since
rdmsr # the kernel's PTEs set the NX bit)
orl $((1 << 8) | (1 << 11)), %eax
wrmsr
movl %cr0, %eax # paging on (CR0.PG) — now in long mode (compat sub-mode)
orl $(1 << 31), %eax
movl %eax, %cr0
ljmpl *(jmp64_ptr - ap_trampoline_start)(%ebx) # -> long_entry, CS = 0x18 (L=1)
.code64
long_entry:
movw $0x10, %ax # sane flat data segments
movw %ax, %ds
movw %ax, %es
movw %ax, %ss
# RBX = EBX (zero-extended) = page base. Load our stack and per-CPU pointer, then
# call the Zig entry — which runs from the kernel image and never returns.
movq (param_stack - ap_trampoline_start)(%rbx), %rsp
movq (param_percpu - ap_trampoline_start)(%rbx), %rdi # SysV arg 0
movq (param_entry - ap_trampoline_start)(%rbx), %rax
callq *%rax
1: hlt # unreachable; guard against a stray return
jmp 1b
# --- data: GDT, far pointers, and the BSP-patched parameter block -----------
.balign 8
gdt32:
.quad 0x0000000000000000 # 0x00 null
.quad 0x00CF9A000000FFFF # 0x08 32-bit code (G, D, present, exec/read)
.quad 0x00CF92000000FFFF # 0x10 data (valid in 32- and 64-bit)
.quad 0x00AF9A000000FFFF # 0x18 64-bit code (L=1)
gdt32_end:
gdtr32:
.word gdt32_end - gdt32 - 1
gdtr32_base:
.long 0 # patched (16-bit code): linear base of gdt32
jmp32_ptr: # indirect far-jump operand: offset then selector
jmp32_off:
.long 0 # patched: linear address of prot_entry
.word 0x08 # 32-bit code selector
jmp64_ptr:
jmp64_off:
.long 0 # patched: linear address of long_entry
.word 0x18 # 64-bit code selector
# The parameter block, filled in by the BSP (smp.zig) before each STARTUP IPI. Global
# so the Zig side can locate each field as (symbol - ap_trampoline_start).
.global ap_tramp_cr3
.global ap_tramp_stack
.global ap_tramp_entry
.global ap_tramp_percpu
param_cr3:
ap_tramp_cr3:
.quad 0 # kernel PML4 physical address (CR3)
param_stack:
ap_tramp_stack:
.quad 0 # top of this AP's kernel stack
param_entry:
ap_tramp_entry:
.quad 0 # address of apEntry (the Zig AP entry)
param_percpu:
ap_tramp_percpu:
.quad 0 # this AP's per-CPU pointer (goes in GS base)
.global ap_trampoline_end
ap_trampoline_end:
+87
View File
@@ -0,0 +1,87 @@
//! Task State Segment and its interrupt stacks. In long mode the TSS has two
//! jobs. First, the Interrupt Stack Table: an IDT gate can name an IST entry,
//! and the CPU switches to that stack when the exception fires — no matter how
//! broken the interrupted stack was. We use IST1 for the double-fault handler,
//! so a fault that happens *because* the current stack is unusable still lands
//! on solid ground instead of triple-faulting. Second, rsp0: the kernel stack
//! the CPU switches to when an interrupt arrives from ring 3 (published by the
//! user-mode entry path via `rsp0Ptr`).
//!
//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
//! indexed by CPU number; slot 0 is the BSP.
const parameters = @import("parameters");
const gdt = @import("gdt.zig");
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
const Tss = packed struct {
reserved0: u32 = 0,
rsp0: u64 = 0,
rsp1: u64 = 0,
rsp2: u64 = 0,
reserved1: u64 = 0,
ist1: u64 = 0,
ist2: u64 = 0,
ist3: u64 = 0,
ist4: u64 = 0,
ist5: u64 = 0,
ist6: u64 = 0,
ist7: u64 = 0,
reserved2: u64 = 0,
reserved3: u16 = 0,
iomap_base: u16 = 0,
};
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
pub const double_fault_ist = 1;
const maximum_cpus = parameters.maximum_cpus;
pub const ist_stack_size = parameters.ist_stack_size;
/// One TSS per core (small — kept static). The IST stacks are 16 KiB each, so only
/// the **BSP's** is static: it must exist before the frame allocator does, to catch a
/// fault during early boot. Each **AP** gets a heap-allocated IST stack at bring-up
/// (after the heap is up), the top of which the BSP records here before waking it —
/// so we reserve big stacks only for cores that actually come online.
var tss_table = [_]Tss{.{}} ** maximum_cpus;
var bsp_ist_stack: [ist_stack_size]u8 align(16) = undefined;
var ap_ist_top = [_]usize{0} ** maximum_cpus; // per-AP IST stack top (0 = BSP / not set)
/// Loads the task register with the TSS selector. Defined in isr.s.
extern fn load_tr(selector: u16) callconv(.c) void;
/// Address of core `cpu`'s rsp0 slot — the kernel stack the CPU switches to on a
/// ring-3 -> ring-0 interrupt. Computed as base + 4 (rsp0's architectural offset,
/// which is why the pointer is only 4-aligned) rather than `&t.rsp0`, which on a
/// packed struct would be an unaligned bit-pointer type. The ring-3 entry path
/// (enter_user in isr.s) writes the current kernel stack pointer through this
/// before dropping to user mode.
pub fn rsp0Ptr(cpu: usize) *align(4) u64 {
return @ptrFromInt(@intFromPtr(&tss_table[cpu]) + 4);
}
/// Record the top of the IST stack the kernel allocated for AP `cpu`. Called on the
/// BSP before waking that core; read by the core's own `setupThisCpu`.
pub fn setApIstStack(cpu: usize, top: usize) void {
ap_ist_top[cpu] = top;
}
/// Set up core `cpu`'s TSS: point IST1 at its stack (the BSP's static one for core 0,
/// the allocated one recorded via `setApIstStack` for an AP), install the TSS
/// descriptor into that core's GDT, and load it into the task register. Requires the
/// core's GDT to already be loaded (gdt.loadOnThisCpu first).
pub fn setupThisCpu(cpu: usize) void {
const t = &tss_table[cpu];
t.* = .{};
t.ist1 = if (cpu == 0) @intFromPtr(&bsp_ist_stack) + ist_stack_size else ap_ist_top[cpu];
t.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
gdt.setTssFor(cpu, @intFromPtr(t), @sizeOf(Tss) - 1);
load_tr(gdt.tss_selector);
}
/// Set up the bootstrap processor's TSS (slot 0). Requires gdt.init first.
pub fn init() void {
setupThisCpu(0);
}
+158
View File
@@ -0,0 +1,158 @@
//! A framebuffer text console: draws glyphs from an embedded PSF2 font directly
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
//! — just pixels.
//!
//! This is a **bootstrap** console — a stop-gap so early boot has something on
//! screen. The framebuffer is a general graphics surface, *not* inherently a text
//! terminal; once the driver machinery exists it becomes a proper graphics device
//! driver and this text-grid crutch goes away. It is therefore kept **separate
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file,
//! while this only paints the handful of user-facing status lines and panics.
//!
//! The module owns a single console and a `present` flag; `write` is a no-op when
//! the firmware handed over no framebuffer (a headless machine), so the kernel
//! never assumes a display exists.
const std = @import("std");
const danos = @import("danos");
/// The one framebuffer console, valid only when `con_present`.
var con: Console = undefined;
var con_present: bool = false;
/// Set up the console over `fb`, or mark it absent if there's no usable
/// framebuffer. Clears the screen when present.
pub fn init(fb: danos.Framebuffer) void {
if (!fb.present()) {
con_present = false;
return;
}
con = Console.init(fb);
if (con.cols == 0 or con.rows == 0) {
con_present = false;
return;
}
con.clear();
con_present = true;
}
/// Whether an on-screen console is available.
pub fn present() bool {
return con_present;
}
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present,
/// so it's always safe to call.
pub fn write(bytes: []const u8) void {
if (!con_present) return;
for (bytes) |c| con.putChar(c);
}
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
const font = @embedFile("font.psf");
const glyph_w = 8;
const glyph_h = 16;
const glyph_bytes = glyph_h; // 8 pixels wide => 1 byte per row
const glyph_data = 32; // PSF2 header size
pub const Console = struct {
fb: danos.Framebuffer,
cols: u32,
rows: u32,
col: u32 = 0,
row: u32 = 0,
fg: u32 = 0x00c8_c8c8, // light grey
bg: u32 = 0x0000_0000, // black
pub fn init(fb: danos.Framebuffer) Console {
// Reach the framebuffer through the physmap, so the pointer stays valid
// once the low identity map is gone. The base is mapped by both the
// loader's bootstrap tables and paging.init.
var mapped = fb;
if (fb.base != 0) mapped.base = danos.physicalToVirtual(fb.base);
return .{
.fb = mapped,
.cols = fb.width / glyph_w,
.rows = fb.height / glyph_h,
};
}
/// Fill the whole screen with the background colour and home the cursor.
pub fn clear(self: *Console) void {
var y: u32 = 0;
while (y < self.fb.height) : (y += 1) self.fillRow(y, self.bg);
self.col = 0;
self.row = 0;
}
pub fn putChar(self: *Console, ch: u8) void {
switch (ch) {
'\n' => self.newline(),
'\r' => self.col = 0,
else => {
if (self.col >= self.cols) self.newline();
self.drawGlyph(ch, self.col * glyph_w, self.row * glyph_h);
self.col += 1;
},
}
}
fn newline(self: *Console) void {
self.col = 0;
if (self.row + 1 >= self.rows) {
self.scroll();
} else {
self.row += 1;
}
}
fn drawGlyph(self: *Console, ch: u8, px: u32, py: u32) void {
const rows = font[glyph_data + @as(usize, ch) * glyph_bytes ..][0..glyph_bytes];
var gy: u32 = 0;
while (gy < glyph_h) : (gy += 1) {
const bits = rows[gy];
var gx: u32 = 0;
while (gx < glyph_w) : (gx += 1) {
// Leftmost pixel is the high bit.
const on = (bits >> @as(u3, @intCast(7 - gx))) & 1 != 0;
self.pixel(px + gx, py + gy, if (on) self.fg else self.bg);
}
}
}
/// Shift the visible text up one glyph row and clear the freed bottom row,
/// leaving the cursor on that now-blank last line.
fn scroll(self: *Console) void {
const visible = self.rows * glyph_h;
var y: u32 = 0;
while (y + glyph_h < visible) : (y += 1) self.copyRow(y, y + glyph_h);
while (y < visible) : (y += 1) self.fillRow(y, self.bg);
self.row = self.rows - 1;
}
inline fn rowPtr(self: *Console, y: u32) [*]volatile u32 {
const base: [*]volatile u8 = @ptrFromInt(self.fb.base);
return @ptrCast(@alignCast(base + y * self.fb.pitch));
}
inline fn pixel(self: *Console, x: u32, y: u32, color: u32) void {
self.rowPtr(y)[x] = color;
}
fn fillRow(self: *Console, y: u32, color: u32) void {
const row = self.rowPtr(y);
var x: u32 = 0;
while (x < self.fb.width) : (x += 1) row[x] = color;
}
fn copyRow(self: *Console, destination_y: u32, source_y: u32) void {
const destination = self.rowPtr(destination_y);
const source = self.rowPtr(source_y);
var x: u32 = 0;
while (x < self.fb.width) : (x += 1) destination[x] = source[x];
}
};
+181
View File
@@ -0,0 +1,181 @@
//! Device service: the kernel side of user-space driver access. At boot it
//! flattens the discovered device tree (src/device) into a stable, id-indexed
//! snapshot and a per-device claim table. User drivers enumerate the snapshot,
//! claim the device they own, and map its MMIO — the claim is the capability that
//! gates `mmio_map`/`irq_bind`, so a process can only ever touch hardware the
//! firmware-neutral device tree says it owns.
//!
//! The table is a **tree**: each entry carries its parent's id. Firmware discovery
//! seeds it, and a **bus driver** grows it — a process that has claimed a bus can
//! `register` children below it as it enumerates them (USB devices behind a hub, PCI
//! functions behind a bridge, comparators inside a timer block).
//!
//! Registration is where the capability model earns its keep. A `DeviceDescriptor` is, in
//! effect, a licence to map physical memory: whoever claims it may `mmio_map` its
//! `.memory` resources and `irq_bind` its `.irq` resources. If a bus driver could
//! invent arbitrary resources, it would invent one covering the kernel's RAM, claim
//! it, and map it. So `register` enforces **containment**: every resource of a child
//! must lie inside a resource of the same kind on its parent. A bus driver can only
//! ever subdivide what it was already given.
const std = @import("std");
const platform = @import("platform");
const danos = @import("danos");
const maximum_devices = 64;
/// Cap on children a single parent may have. A zero-resource child (legal — a USB
/// device is addressed through its controller, not by MMIO) sidesteps the containment
/// check, so without a bound a process that claimed one device could loop
/// `device_register` and exhaust the whole table, permanently denying it to every other
/// driver. This bounds the blast radius of one claim; a real quota (and a
/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md.
const maximum_children_per_parent = 16;
var devices: [maximum_devices]danos.DeviceDescriptor = undefined;
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
var count: usize = 0;
/// Devices discovery found but the table had no room for. Non-zero means the machine
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
/// which would otherwise be an entirely silent failure. Logged at boot.
pub var dropped: usize = 0;
/// Snapshot the device tree into the flat table. Run once, right after discovery.
pub fn init(device_tree: *const platform.DeviceTree) void {
count = 0;
dropped = 0;
for (&claimed) |*c| c.* = null;
walk(device_tree.root, danos.no_parent);
}
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
/// assigned it down to its children as their parent.
fn walk(node: *platform.Device, parent_id: u64) void {
const id = if (node.class == .root) danos.no_parent else record(node, parent_id);
var child = node.first_child;
while (child) |c| : (child = c.next_sibling) walk(c, id);
}
fn record(node: *platform.Device, parent_id: u64) u64 {
if (count >= maximum_devices) {
dropped += 1;
return danos.no_parent; // children of a dropped node become roots, not orphans
}
var d = std.mem.zeroes(danos.DeviceDescriptor);
d.id = count;
d.parent = parent_id;
d.class = @intFromEnum(node.class);
const h = node.hid();
d.hid_len = @min(h.len, d.hid.len);
@memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]);
const rc = @min(node.resource_count, danos.maximum_device_resources);
d.resource_count = rc;
for (0..rc) |i| {
const r = node.resources[i];
d.resources[i] = .{ .kind = @intFromEnum(r.kind), .start = r.start, .len = r.len };
}
devices[count] = d;
count += 1;
return d.id;
}
/// Copy up to `out.len` device descriptors into `out`; returns the total count
/// available (which may exceed `out.len`).
pub fn enumerate(out: []danos.DeviceDescriptor) usize {
const n = @min(count, out.len);
@memcpy(out[0..n], devices[0..n]);
return count;
}
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
/// out of range or already claimed.
pub fn claim(id: u64, owner: u32) bool {
if (id >= count) return false;
if (claimed[@intCast(id)] != null) return false;
claimed[@intCast(id)] = owner;
return true;
}
/// The task that owns device `id`, or null.
pub fn ownerOf(id: u64) ?u32 {
if (id >= count) return null;
return claimed[@intCast(id)];
}
/// Resource `index` of device `id`, or null if out of range.
pub fn resourceOf(id: u64, index: u64) ?danos.ResourceDescriptor {
if (id >= count) return null;
const d = &devices[@intCast(id)];
if (index >= d.resource_count) return null;
return d.resources[@intCast(index)];
}
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
/// interval containment; for an irq it's equality, since an interrupt line is not
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
/// and would otherwise vacuously "fit" anywhere.
fn contains(parent: danos.ResourceDescriptor, child: danos.ResourceDescriptor) bool {
if (parent.kind != child.kind) return false;
if (child.kind == @intFromEnum(danos.ResourceKind.irq)) return parent.start == child.start;
if (child.len == 0 or parent.len == 0) return false;
// No overflow: a resource that wraps the address space is not containable.
const child_end = std.math.add(u64, child.start, child.len) catch return false;
const parent_end = std.math.add(u64, parent.start, parent.len) catch return false;
return child.start >= parent.start and child_end <= parent_end;
}
pub const RegisterError = error{
NoSpace, // the device table is full
BadParent, // no such device, or not claimed by this task
TooManyResources,
TooManyChildren, // this parent is at maximum_children_per_parent
NotContained, // a child resource escapes its parent's window
};
/// Number of devices currently recorded with `parent_id` as their parent.
fn childCount(parent_id: u64) usize {
var n: usize = 0;
for (devices[0..count]) |d| {
if (d.parent == parent_id) n += 1;
}
return n;
}
/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new
/// device id. The child is left **unclaimed**, so another process (a class driver)
/// can claim it — that is how a bus hands a device to its driver.
///
/// `owner` must have claimed `parent_id`, and every resource in `descriptor` must be
/// contained in a parent resource of the same kind. A device with no resources is
/// fine and common: a USB device is addressed through its controller, not by MMIO.
pub fn register(parent_id: u64, owner: u32, descriptor: *const danos.DeviceDescriptor) RegisterError!u64 {
const parent_owner = ownerOf(parent_id) orelse return error.BadParent;
if (parent_owner != owner) return error.BadParent;
if (descriptor.resource_count > danos.maximum_device_resources) return error.TooManyResources;
if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;
if (count >= maximum_devices) return error.NoSpace;
const parent = &devices[@intCast(parent_id)];
for (0..@intCast(descriptor.resource_count)) |i| {
const r = descriptor.resources[i];
var ok = false;
for (0..@intCast(parent.resource_count)) |j| {
if (contains(parent.resources[j], r)) ok = true;
}
if (!ok) return error.NotContained;
}
var d = std.mem.zeroes(danos.DeviceDescriptor);
d.id = count;
d.parent = parent_id;
d.class = descriptor.class;
d.hid_len = @min(descriptor.hid_len, d.hid.len);
@memcpy(d.hid[0..@intCast(d.hid_len)], descriptor.hid[0..@intCast(d.hid_len)]);
d.resource_count = descriptor.resource_count;
for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i];
devices[count] = d;
count += 1;
return d.id;
}
Binary file not shown.
+174
View File
@@ -0,0 +1,174 @@
//! The kernel heap: dynamic allocation for the kernel.
//!
//! Where the frame allocator ([pmm]) hands out fixed 4 KiB physical frames, the
//! heap hands out arbitrary byte-sized blocks from a virtual region, growing on
//! demand by mapping fresh frames into it (architecture.mapPage) — the first real user of
//! the VMM (see docs/paging.md).
//!
//! The algorithm is a first-fit free list: an address-ordered singly linked list
//! of free blocks, split on allocation and coalesced with neighbours on free. It
//! is exposed as a std.mem.Allocator, so the kernel can use std containers.
//!
//! Not yet concurrency-safe: it assumes a single caller and no allocation from
//! interrupt handlers (ours don't). A lock comes with threads/SMP.
const std = @import("std");
const danos = @import("danos");
const architecture = @import("architecture");
const pmm = @import("pmm.zig");
const page_size = danos.page_size;
/// Virtual base of the heap: the start of the higher half, which is unmapped and
/// well clear of the identity-mapped low half. (Canonical on x86_64; an architecture that
/// splits the address space differently would choose its own.)
const heap_base: usize = 0xFFFF_8000_0000_0000;
/// Cap on heap growth for now.
const heap_maximum: usize = 64 * 1024 * 1024;
/// A block header, placed at the start of every block. While the block is free it
/// also links into the free list via `next`.
const Block = extern struct {
size: usize, // total block size in bytes, including this header; a multiple of 16
next: ?*Block, // free-list link (only meaningful while free)
};
const header_size = @sizeOf(Block); // 16
const minimum_block = header_size + 16; // smallest block worth splitting off
var free_list: ?*Block = null;
var heap_end: usize = heap_base; // [heap_base, heap_end) is currently mapped
fn alignUp(value: usize, alignment: usize) usize {
return (value + alignment - 1) & ~(alignment - 1);
}
fn payloadOf(block: *Block) [*]u8 {
return @ptrFromInt(@intFromPtr(block) + header_size);
}
/// Bring the heap up with an initial mapped region.
pub fn init() void {
free_list = null;
heap_end = heap_base;
_ = grow(page_size);
}
/// Map more pages onto the end of the heap and add them as a free block. Returns
/// false if out of heap virtual space or out of physical frames.
fn grow(minimum_bytes: usize) bool {
const start = heap_end;
const bytes = alignUp(minimum_bytes, page_size);
if (start + bytes > heap_base + heap_maximum) return false;
var virtual = start;
while (virtual < start + bytes) : (virtual += page_size) {
const frame = pmm.alloc() orelse return false;
architecture.mapPage(virtual, frame, true);
}
heap_end = start + bytes;
const block: *Block = @ptrFromInt(start);
block.size = bytes;
insertFree(block); // coalesces with the previous tail block if adjacent
return true;
}
/// Insert a block into the address-ordered free list, coalescing with the
/// physically adjacent free blocks on either side.
fn insertFree(block: *Block) void {
var previous: ?*Block = null;
var current = free_list;
while (current) |c| : (current = c.next) {
if (@intFromPtr(c) > @intFromPtr(block)) break;
previous = c;
}
block.next = current;
if (previous) |p| p.next = block else free_list = block;
// Merge forward into `current` if they're contiguous.
if (current) |c| {
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
block.size += c.size;
block.next = c.next;
}
}
// Merge `previous` forward into `block` if they're contiguous.
if (previous) |p| {
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
p.size += block.size;
p.next = block.next;
}
}
}
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
fn rawAlloc(len: usize) ?[*]u8 {
const need = alignUp(header_size + len, 16);
var attempts: u32 = 0;
while (attempts < 2) : (attempts += 1) {
var previous: ?*Block = null;
var current = free_list;
while (current) |block| : ({
previous = block;
current = block.next;
}) {
if (block.size < need) continue;
if (block.size >= need + minimum_block) {
// Split: carve `need` off the front, leave the rest free.
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
rest.size = block.size - need;
rest.next = block.next;
if (previous) |p| p.next = rest else free_list = rest;
block.size = need;
} else {
// Take the whole block.
if (previous) |p| p.next = block.next else free_list = block.next;
}
return payloadOf(block);
}
// Nothing fit: grow and try once more.
if (!grow(need)) return null;
}
return null;
}
fn rawFree(ptr: [*]u8) void {
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
insertFree(block);
}
// --- std.mem.Allocator interface -----------------------------------------
pub fn allocator() std.mem.Allocator {
return .{ .ptr = undefined, .vtable = &vtable };
}
const vtable = std.mem.Allocator.VTable{
.alloc = allocImpl,
.resize = resizeImpl,
.remap = remapImpl,
.free = freeImpl,
};
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
// Blocks are 16-byte aligned; larger alignments aren't supported yet.
if (alignment.toByteUnits() > 16) return null;
return rawAlloc(len);
}
fn resizeImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) bool {
return false; // no in-place resize; the caller reallocates
}
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
return null;
}
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
rawFree(memory.ptr);
}
+314
View File
@@ -0,0 +1,314 @@
//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a
//! rendezvous point; a client `call`s it (send a message, block for a reply) and
//! a server `replyWait`s on it (reply to the last client, then block for the next
//! request). This is the substrate the user-space VFS server and device drivers
//! are reached through — `open`/`read`/`write` become user-space wrappers that
//! marshal a request into a `call`.
//!
//! Design (see docs/syscall.md, the plan):
//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame
//! through the physmap (`copyAcross`), which is mapped in every address space's
//! shared kernel half — so the kernel reads/writes either process's user memory
//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing.
//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply
//! to exactly one client at a time; that caller is held in `Task.ipc_client`.
//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without
//! becoming runnable*, which a WaitQueue can't express, so callers queue on the
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
//! waiting for work use a normal WaitQueue.
//!
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
//! copy is a later security-track item, matching the existing debug_write gap.
const std = @import("std");
const danos = @import("danos");
const architecture = @import("architecture");
const scheduler = @import("scheduler.zig");
const sync = @import("sync.zig");
const heap = @import("heap.zig");
const page_size = danos.page_size;
const Task = scheduler.Task;
/// Largest message a single call/reply may carry. Bumping it is trivial; kept
/// small because the copy runs under the big kernel lock.
pub const MESSAGE_MAXIMUM: usize = 256;
pub const maximum_handles = scheduler.ipc_maximum_handles;
pub const maximum_services = 8;
/// Errno-style failures, returned as `-value` in the system_call result register.
pub const EBADF: i64 = 1; // bad handle
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
pub const ENOENT: i64 = 4; // no such registered service
pub const ENOSPC: i64 = 5; // handle table or registry full
pub const ENOMEM: i64 = 6; // out of memory
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
/// message from a client — there is no reply owed. The low bits carry the source
/// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in system/kernel/irq.zig;
/// the message path uses a plain task-id badge with this bit clear. Defined in the
/// shared contract (system/danos.zig), because ring 3 has to test the same bit.
pub const notify_badge_bit: u64 = danos.notify_badge_bit;
/// End of the user (low) canonical half — user buffers must lie below it.
const user_half_end: u64 = 0x0000_8000_0000_0000;
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
/// (per process) and/or by a registry slot, counted by `refcount`.
pub const Endpoint = struct {
refcount: u32 = 1,
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
// Task.next; each such task is .blocked and in no scheduler queue).
sender_head: ?*Task = null,
sender_tail: ?*Task = null,
// Servers blocked in `replyWait` awaiting a request.
receive_wait_queue: scheduler.WaitQueue = .{},
// Pending asynchronous notifications (badges), a small coalescing ring.
notify_buffer: [8]u64 = undefined,
notify_head: u8 = 0,
notify_tail: u8 = 0,
};
pub fn createEndpoint() ?*Endpoint {
const endpoint = heap.allocator().create(Endpoint) catch return null;
endpoint.* = .{};
return endpoint;
}
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
/// today like other kernel objects — but the refcount bookkeeping lands now.)
pub fn dropRef(endpoint: *Endpoint) void {
if (endpoint.refcount > 1) {
endpoint.refcount -= 1;
} else {
heap.allocator().destroy(endpoint);
}
}
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
t.next = null;
if (endpoint.sender_tail) |tail| tail.next = t else endpoint.sender_head = t;
endpoint.sender_tail = t;
}
fn dequeueSender(endpoint: *Endpoint) ?*Task {
const t = endpoint.sender_head orelse return null;
endpoint.sender_head = t.next;
if (endpoint.sender_head == null) endpoint.sender_tail = null;
t.next = null;
return t;
}
// --- cross-address-space copy ----------------------------------------------
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
/// unmapped or out of range. Handles page-straddling buffers.
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
const source_root = if (source_as != 0) source_as else architecture.kernelPageTable();
const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable();
if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false;
if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false;
var off: usize = 0;
while (off < len) {
const s = architecture.translate(source_root, source_va + off) orelse return false;
const d = architecture.translate(destination_root, destination_va + off) orelse return false;
const s_left = page_size - ((source_va + off) & (page_size - 1));
const d_left = page_size - ((destination_va + off) & (page_size - 1));
const n = @min(@min(s_left, d_left), len - off);
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(d));
@memcpy(destination[0..n], source[0..n]);
off += n;
}
return true;
}
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel
/// buffer `destination`, walking the user page tables through the physmap. Returns false if
/// the range escapes the user half or any source page is unmapped — so a bad user
/// pointer *fails the system_call* rather than faulting the kernel (danos has no
/// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt
/// the machine). The correct way to pull a fixed-size struct in from user space, and
/// a single fetch: no TOCTOU against a hostile pointer.
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
if (user_as == 0) return false; // not a user address space
if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false;
var off: usize = 0;
while (off < destination.len) {
const s = architecture.translate(user_as, user_va + off) orelse return false;
const s_left = page_size - ((user_va + off) & (page_size - 1));
const n = @min(s_left, destination.len - off);
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
@memcpy(destination[off..][0..n], source[0..n]);
off += n;
}
return true;
}
// --- the two IPC operations -------------------------------------------------
/// Client side of IPC_Call: send `[message_ptr, message_len)` to `endpoint` and block until a
/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a
/// negative errno. Runs as the current task.
pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64) i64 {
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
const flags = sync.enter();
defer sync.leave(flags);
const me = scheduler.current();
me.ipc_send_ptr = message_ptr;
me.ipc_send_len = message_len;
me.ipc_reply_ptr = reply_ptr;
me.ipc_reply_cap = reply_cap;
me.ipc_status = 0;
enqueueSender(endpoint, me); // join the FIFO, then...
scheduler.wakeLocked(&endpoint.receive_wait_queue); // ...wake a waiting server (no-op if none)
scheduler.blockCurrentLocked(); // block until the reply readies us again
return me.ipc_status; // reply length or -errno, written by the replier
}
/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client
/// we currently owe (if any), then receive the next request into
/// `[receive_ptr, receive_cap)`, blocking until one arrives. Writes the sender's badge
/// to `out_badge` and returns the request length, or a negative errno. A pending
/// notification is delivered ahead of client requests (length 0, badge with
/// `notify_badge_bit` set, no reply owed).
pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, out_badge: *u64) i64 {
if (reply_len > MESSAGE_MAXIMUM or receive_cap > MESSAGE_MAXIMUM) return -E2BIG;
const flags = sync.enter();
defer sync.leave(flags);
const me = scheduler.current();
// (1) Reply to the client we're still holding, if any.
if (me.ipc_client) |client| {
me.ipc_client = null;
const n = @min(reply_len, client.ipc_reply_cap);
if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
client.ipc_status = @intCast(n);
} else {
client.ipc_status = -EFAULT;
}
scheduler.readyLocked(client); // its `call` now returns
}
// (2) Receive the next request (or notification), blocking until one is ready.
while (true) {
if (popNotify(endpoint)) |badge| {
out_badge.* = badge | notify_badge_bit;
return 0; // notification: no payload, no reply owed
}
if (dequeueSender(endpoint)) |caller| {
const n = @min(caller.ipc_send_len, receive_cap);
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) {
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
scheduler.readyLocked(caller);
continue;
}
me.ipc_client = caller; // remember who to reply to
out_badge.* = caller.id;
return @intCast(n);
}
scheduler.waitLocked(&endpoint.receive_wait_queue); // nothing yet — sleep until woken, then retry
}
}
// --- asynchronous notification (for IRQ-as-message, M10) --------------------
fn popNotify(endpoint: *Endpoint) ?u64 {
if (endpoint.notify_head == endpoint.notify_tail) return null;
const badge = endpoint.notify_buffer[endpoint.notify_head % endpoint.notify_buffer.len];
endpoint.notify_head +%= 1;
return badge;
}
/// Post an asynchronous notification carrying `badge` to `endpoint` and wake a waiting
/// receiver. Precondition: the big kernel lock is held.
///
/// The lock must already cover whatever produced `endpoint` — an ISR that looked the
/// endpoint up in a table and *then* took the lock could be racing a process exit
/// that unbinds and frees it in between. See irq.dispatch, which holds one lock
/// region across the table read and this call.
///
/// A full ring drops the notification. That is the correct semantics, not a
/// concession: a notification is a *level* ("this device wants attention"), and the
/// driver re-reads device state on wake. It is never a count of events.
pub fn notifyLocked(endpoint: *Endpoint, badge: u64) void {
if (endpoint.notify_tail -% endpoint.notify_head < endpoint.notify_buffer.len) {
endpoint.notify_buffer[endpoint.notify_tail % endpoint.notify_buffer.len] = badge;
endpoint.notify_tail +%= 1;
}
scheduler.wakeLocked(&endpoint.receive_wait_queue);
}
/// `notifyLocked` as a self-contained ISR critical section, for a caller that holds
/// `endpoint` by some means other than a table the lock protects. Releases the lock without
/// touching the interrupt flag (the ISR's iretq restores it), like the timer tick.
pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
_ = sync.enter();
notifyLocked(endpoint, badge);
sync.leaveIsr();
}
// --- per-process handle table + name registry -------------------------------
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
for (&t.handles, 0..) |*slot, i| {
if (slot.* == null) {
slot.* = @ptrCast(endpoint);
return @intCast(i);
}
}
return -ENOSPC;
}
/// Resolve a handle to its endpoint, or null if out of range / unused.
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
if (h >= t.handles.len) return null;
const slot = t.handles[@intCast(h)] orelse return null;
return @ptrCast(@alignCast(slot));
}
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
/// exit path so a dead server's endpoints don't linger referenced.
pub fn closeHandles(t: *Task) void {
for (&t.handles) |*slot| {
if (slot.*) |p| {
dropRef(@ptrCast(@alignCast(p)));
slot.* = null;
}
}
}
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
pub fn register(id: u32, endpoint: *Endpoint) i64 {
if (id >= maximum_services) return -ENOENT;
if (registry[id]) |old| dropRef(old);
endpoint.refcount += 1;
registry[id] = endpoint;
return 0;
}
/// Find the endpoint published under `id`, taking a reference for the caller to
/// install in its handle table. Null if nothing is registered there.
pub fn lookup(id: u32) ?*Endpoint {
if (id >= maximum_services) return null;
const endpoint = registry[id] orelse return null;
endpoint.refcount += 1;
return endpoint;
}
+54
View File
@@ -0,0 +1,54 @@
//! Inter-process communication: message-passing channels.
//!
//! IPC is the backbone of a microkernel ([vision](../docs/vision.md)): once
//! drivers and services live in separate address spaces, a message is how they
//! talk. This first form is a **bounded blocking channel** — a ring buffer of
//! messages with a producer/consumer rendezvous, built on the scheduler's
//! [wait queues](scheduling.md). `send` blocks when the channel is full, `receive`
//! blocks when it's empty; neither busy-waits.
//!
//! For now both endpoints are kernel threads sharing the kernel address space.
//! When user mode arrives, the same primitive carries messages across the
//! isolation boundary (with the payload copied between address spaces).
const scheduler = @import("scheduler.zig");
const sync = @import("sync.zig");
/// A bounded blocking channel of `capacity` messages of type `T`.
pub fn Channel(comptime T: type, comptime capacity: usize) type {
return struct {
const Self = @This();
buffer: [capacity]T = undefined,
head: usize = 0, // next slot to read
tail: usize = 0, // next slot to write
count: usize = 0,
not_full: scheduler.WaitQueue = .{}, // senders wait here
not_empty: scheduler.WaitQueue = .{}, // receivers wait here
/// Send a message, blocking while the channel is full.
pub fn send(self: *Self, message: T) void {
const flags = sync.enter();
// Recheck the condition in a loop: a wakeup only means "try again"
// (another waiter may have taken the slot first).
while (self.count == capacity) scheduler.waitLocked(&self.not_full);
self.buffer[self.tail] = message;
self.tail = (self.tail + 1) % capacity;
self.count += 1;
scheduler.wakeLocked(&self.not_empty); // a receiver can now proceed
sync.leave(flags);
}
/// Receive a message, blocking while the channel is empty.
pub fn receive(self: *Self) T {
const flags = sync.enter();
while (self.count == 0) scheduler.waitLocked(&self.not_empty);
const message = self.buffer[self.head];
self.head = (self.head + 1) % capacity;
self.count -= 1;
scheduler.wakeLocked(&self.not_full); // a sender can now proceed
sync.leave(flags);
return message;
}
};
}
+190
View File
@@ -0,0 +1,190 @@
//! IRQ-as-IPC: delivering a hardware interrupt to a user-space driver.
//!
//! A microkernel can't run driver code in the ISR — the driver is a ring-3 process
//! in another address space. So the kernel's ISR does the least it can: quiet the
//! line, acknowledge the CPU, and post an asynchronous notification to the endpoint
//! the driver is blocked on (`ipc_sync.notifyFromIsr`). The driver wakes out of
//! `IPC_ReplyWait`, services the device, and calls `irq_ack` to re-arm.
//!
//! The full cycle, and why each step is where it is:
//!
//! ISR irqMask(gsi) -- the line is still asserted; stop it reaching a CPU
//! irqEoi() -- now safe to tell the LAPIC we're done
//! notifyFromIsr() -- wake the driver (it runs much later)
//! driver <services device> -- reads/clears the device's status register
//! driver irq_ack(device,resource) -- irqUnmask(gsi): the line is quiet, let it through
//!
//! Mask-before-EOI is the load-bearing part. A level-triggered line stays asserted
//! until the *device* is quieted, which only the ring-3 driver can do. EOI with the
//! entry unmasked and the I/O APIC redelivers immediately, forever, before the
//! driver is ever scheduled. Masking converts "level" into something a deferred
//! handler can cope with; `irq_ack` is what closes the loop.
//!
//! Binding is capability-gated exactly like `mmio_map`: the caller must have
//! `device_claim`ed the device, and the GSI must come from one of that device's `irq`
//! resources in the discovered device table (system/kernel/device-service.zig). A driver can
//! therefore never bind an interrupt it doesn't own — a raw-GSI system_call would let
//! any process steal the keyboard's line.
//!
//! KNOWN ISSUE (real hardware, not QEMU). Masking a level-triggered redirection entry
//! while its remote-IRR bit is set does not clear remote-IRR on some chipsets, and the
//! line then never fires again — a driver would take exactly one interrupt and block
//! forever. QEMU's I/O APIC clears remote-IRR on EOI regardless of the mask, so the
//! `hpet` test cannot see this. Linux's workaround is to flush remote-IRR by briefly
//! flipping the entry to edge-trigger and back. Revisit when danos first boots on
//! metal; MSI (no mask cycle at all) sidesteps it entirely.
const architecture = @import("architecture");
const sync = @import("sync.zig");
const ipc_sync = @import("ipc-synchronous.zig");
/// GSIs a single I/O APIC covers. 24 is the standard redirection-table size; a
/// second I/O APIC (none on QEMU's q35) would extend this.
pub const maximum_gsi = 24;
/// The endpoint to notify for each bound GSI, or null if unbound. Read from the ISR
/// and written from syscalls, always under the big kernel lock.
var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi;
/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint
/// pointer: an endpoint can be shared between processes (ipc_register/ipc_lookup hand
/// out extra references), so "every GSI pointing at this endpoint" is not the same
/// set as "every GSI this process bound", and releasing the former on exit would mask
/// a live sibling's device line.
var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi;
/// No GSI assigned to this vector.
const no_gsi: u32 = 0xFFFF_FFFF;
/// Reverse map for the ISR: which GSI does this vector carry? Populated at bind.
/// `interruptDispatch` hands a handler no arguments, so the vector→GSI edge has to
/// be recovered from somewhere — the trampolines below capture the vector at
/// comptime, and this turns it back into a GSI. Trampolines are installed on every
/// vector in the window at boot, so an unbound one must be distinguishable from
/// GSI 0 — hence the sentinel rather than a zero default.
var vector_gsi: [256]u32 = .{no_gsi} ** 256;
/// Vector currently assigned to each GSI (0 = none), so a rebind reuses it.
var gsi_vector: [maximum_gsi]u8 = .{0} ** maximum_gsi;
/// Set once the trampolines are installed.
var installed = false;
/// The ISR body for a bound device line. Runs with interrupts off, on the
/// interrupted task's kernel stack, on whichever core the I/O APIC picked.
fn dispatch(vector: u8) void {
const gsi = vector_gsi[vector];
if (gsi == no_gsi) {
// Nothing is routed here. Acknowledge so the LAPIC doesn't wedge on an
// in-service bit that never clears, but touch no redirection entry.
architecture.irqEoi();
return;
}
// One lock region for the whole cycle. Two reasons, and the second is subtle:
//
// - The I/O APIC is an index/data register pair, so two cores interleaving a
// read-modify-write of a redirection entry would corrupt it.
// - `bound[gsi]` must be *read and used* under the same acquisition that
// `unbind` writes it under. Dropping the lock between the load and
// `notifyLocked` would let a driver exiting on another core free the endpoint
// in the gap, and we would post a notification into freed memory. The GSI is
// routed to the core that bound it, but a driver may migrate and exit
// elsewhere, so this is reachable on SMP.
//
// No deadlock: the lock is non-recursive, but a core holding it runs with
// interrupts disabled and so cannot interrupt itself into here.
_ = sync.enter();
defer sync.leaveIsr();
architecture.irqMask(gsi); // the line is still asserted; stop it reaching a CPU
architecture.irqEoi(); // now safe to release the LAPIC's in-service bit
// Wakes the driver if it's blocked in ReplyWait; otherwise queues the badge on
// the endpoint's notify ring, so an interrupt taken while the driver is off
// doing something else is not lost.
if (bound[gsi]) |endpoint| ipc_sync.notifyLocked(endpoint, gsi);
}
/// Install one no-argument trampoline per usable vector. Each closes over its own
/// `vector` as a comptime constant — that's the trick that gets an argument into
/// `idt.Handler` (`*const fn () void`) without a per-vector hand-written stub.
pub fn init() void {
if (installed) return;
inline for (0..architecture.irq_vector_count) |i| {
const vector: u8 = @intCast(@as(usize, architecture.irq_vector_base) + i);
architecture.irqSetHandler(vector, &struct {
fn trampoline() void {
dispatch(vector);
}
}.trampoline);
}
installed = true;
}
/// Lowest unused vector in the device window, or null if they're all spoken for.
fn allocVector() ?u8 {
var v: u8 = architecture.irq_vector_base;
while (v < architecture.irq_vector_base + architecture.irq_vector_count) : (v += 1) {
var used = false;
for (gsi_vector) |gv| {
if (gv == v) used = true;
}
if (!used) return v;
}
return null;
}
pub const BindError = error{ BadGsi, InUse, NoVector };
/// Deliver `gsi` to `endpoint` as an IPC notification, on behalf of task `owner`. Routes the
/// line to this core, installs the binding, and unmasks. Caller must hold the big
/// kernel lock, and must already have checked that `owner` claimed the device this GSI
/// belongs to.
pub fn bind(gsi: u32, endpoint: *ipc_sync.Endpoint, owner: u32) BindError!void {
if (gsi >= maximum_gsi or !architecture.irqOwnsGsi(gsi)) return error.BadGsi;
if (bound[gsi] != null) return error.InUse;
const vector = allocVector() orelse return error.NoVector;
vector_gsi[vector] = gsi;
gsi_vector[gsi] = vector;
bound[gsi] = endpoint;
bound_owner[gsi] = owner;
// Level-triggered, active-high. Level is the general case a driver must survive
// (and what hpetd configures its comparator for); an edge source simply never
// leaves the line asserted, so the mask/ack cycle is harmless there.
//
// Hardcoded for now: a device whose MADT interrupt-source override declares the
// line active-*low* (most legacy PCI INTx) will need the polarity threaded
// through from discovery. Nothing danos binds today is such a device.
architecture.irqRoute(gsi, vector, true, false);
architecture.irqUnmask(gsi);
}
/// Re-arm `gsi` after the driver has quieted the device. Caller holds the big lock
/// and has verified ownership.
pub fn ack(gsi: u32) bool {
if (gsi >= maximum_gsi or bound[gsi] == null) return false;
architecture.irqUnmask(gsi);
return true;
}
/// Drop every binding made by task `owner` — called as that task exits, *before* its
/// endpoints are freed. Each line is left masked, so a dead driver's device goes quiet
/// rather than interrupting into a freed endpoint. Caller holds the big kernel lock,
/// which is what makes this safe against a concurrent `dispatch` on another core.
///
/// Keyed on the owner, not the endpoint: endpoints are shared (a registered service's
/// endpoint has references in several processes), so releasing "everything pointing at
/// this endpoint" would tear down bindings this task never made.
pub fn releaseOwner(owner: u32) void {
for (&bound, 0..) |*slot, gsi| {
if (slot.* != null and bound_owner[gsi] == owner) {
architecture.irqMask(@intCast(gsi));
slot.* = null;
bound_owner[gsi] = 0;
gsi_vector[gsi] = 0;
}
}
}
+83
View File
@@ -0,0 +1,83 @@
//! The kernel's multi-sink **diagnostic** log — the machine-readable stream of
//! what the kernel is doing, separate from any user-facing display.
//!
//! Output is a *diagnostic convenience, never a correctness dependency* — the
//! kernel must boot and run correctly with zero output channels. So logging fans
//! out to a set of registered **sinks**, each best-effort and self-guarding: the
//! serial UART, the 0xE9 debug console, and — later — a file on a ramdisk/USB/SSD.
//! A message reaches whatever channels exist; if none do, the kernel runs on,
//! silent but correct.
//!
//! The **framebuffer is deliberately not a sink here.** It's a separate output
//! surface (a bootstrap text console today, a graphics device driver later), so
//! the log never assumes the machine is text-based. `main.zig` mirrors a few
//! user-facing status lines and panics to it explicitly; the verbose log does not.
//!
//! No allocation: the sink table is fixed, so the log works before the heap is up
//! and inside a panic. Two channels don't go through the sink list because they
//! must survive even a total-output failure: `checkpoint` (a one-byte POST code)
//! and `recordPanic` (a breadcrumb in a fixed record).
const std = @import("std");
const architecture = @import("architecture");
pub const SinkFn = *const fn ([]const u8) void;
const maximum_sinks = 8;
var sinks: [maximum_sinks]SinkFn = undefined;
var sink_count: usize = 0;
/// Register an output sink. Every registered sink receives every message; sinks
/// must be self-guarding (safe to call when their device is absent).
pub fn addSink(sink: SinkFn) void {
if (sink_count < maximum_sinks) {
sinks[sink_count] = sink;
sink_count += 1;
}
}
/// Fan `bytes` out to every registered sink.
pub fn write(bytes: []const u8) void {
for (sinks[0..sink_count]) |sink| sink(bytes);
}
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
/// this is safe to call from interrupt context and from a panic.
pub fn print(comptime fmt: []const u8, args: anytype) void {
var buffer: [256]u8 = undefined;
write(std.fmt.bufPrint(&buffer, fmt, args) catch return);
}
/// Emit a one-byte checkpoint/POST code (I/O port 0x80) — the always-available
/// progress channel for when there is no text output at all. Independent of the
/// sink list, so it works even before any sink is registered.
pub fn checkpoint(code: u8) void {
architecture.checkpoint(code);
}
// --- persistent panic breadcrumb -------------------------------------------
//
// A fixed record in the kernel image that a panic fills in, so a post-mortem — an
// attached debugger, a RAM dump, or (later) a file/pstore reader — can recover
// what killed the kernel even when there was no live console. `magic` is written
// *last*, so a reader only trusts a fully-written record.
pub const panic_magic: u64 = 0xD1ED_B00B_5EED_F00D;
pub const PanicRecord = extern struct {
magic: u64 = 0,
len: u32 = 0,
_pad: u32 = 0,
message: [512]u8 = undefined,
};
/// Findable by symbol (`log.panic_record`) for a debugger or RAM dump.
pub var panic_record: PanicRecord = .{};
/// Stamp the panic message into the breadcrumb record.
pub fn recordPanic(message: []const u8) void {
const n: u32 = @intCast(@min(message.len, panic_record.message.len));
@memcpy(panic_record.message[0..n], message[0..n]);
panic_record.len = n;
panic_record.magic = panic_magic; // set last: a reader sees a complete record
}
+431
View File
@@ -0,0 +1,431 @@
const std = @import("std");
const danos = @import("danos");
const parameters = @import("parameters");
const architecture = @import("architecture");
const console = @import("console.zig");
const log = @import("log.zig");
const pmm = @import("pmm.zig");
const heap = @import("heap.zig");
const scheduler = @import("scheduler.zig");
const process = @import("process.zig");
const device_service = @import("device-service.zig");
const irq = @import("irq.zig");
const initrd = @import("initrd");
const platform = @import("platform");
const tests = @import("tests.zig");
const build_options = @import("build_options");
const BootInformation = danos.BootInformation;
/// The calling convention used to enter the kernel. Pinned to SystemV explicitly:
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
/// x64 (first argument in RCX), while the kernel is SystemV (first argument in
/// RDI). Both sides reference this so the `boot_information` pointer lands in the
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
pub const kernel_abi = danos.kernel_abi;
// POST/checkpoint codes emitted to I/O port 0x80 at boot milestones — the
// last-resort progress signal on a machine with no text output at all.
const cp_entry = 0x10;
const cp_paging = 0x20;
const cp_heap = 0x30;
const cp_discovery = 0x40;
const cp_scheduler = 0x50;
const cp_timer = 0x60;
const cp_running = 0x70;
const cp_exception = 0xE0;
const cp_panic = 0xEE;
/// Physical address of the low page reserved at boot for the AP trampoline (0 = none
/// was available). Claimed right after the frame allocator comes up, before paging
/// and the heap consume the scarce sub-1 MiB frames.
var ap_trampoline_page: u64 = 0;
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
/// caller to return to, so this never returns.
/// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss
/// then calls this with the loader's `boot_information` pointer in RDI. We can't keep
/// running on the loader's stack: it's a low physical address that the identity
/// map covers only transitionally, and vanishes once the kernel drops the low
/// half. `boot_information` (also low) is reached through the physmap — its base is the
/// same under the loader's bootstrap tables and the kernel's own.
export fn kmainEntry(boot_information: *const BootInformation) callconv(kernel_abi) noreturn {
kmain(@ptrFromInt(danos.physicalToVirtual(@intFromPtr(boot_information))));
}
fn kmain(boot_information: *const BootInformation) noreturn {
// The **log** is the machine-readable diagnostic stream: it fans out to every
// *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
// A headless, serial-less machine still boots correctly — it just goes quiet,
// with port-0x80 checkpoints as the only progress signal.
architecture.serialInit();
log.addSink(architecture.serialWrite);
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
// surface — a bootstrap text console today, a graphics device driver later — so
// we never assume the OS is text-based. Only a few user-facing status lines
// (via `status`) and panics are mirrored to it; the verbose log stays out.
const fb = boot_information.framebuffer;
console.init(fb);
log.checkpoint(cp_entry);
// Catch CPU exceptions before doing anything that might fault: install our
// reporter, then bring up the GDT + IDT.
architecture.setFaultHandler(onException);
architecture.init();
status("danos: initialising kernel...\n");
log.write(if (console.present())
"danos: framebuffer console online (bootstrap; graphics driver later)\n"
else
"danos: no framebuffer (headless) -> logging to serial/debugcon only\n");
log.write("danos: cpu tables online (GDT, IDT, TSS)\n");
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
log.print(" pitch : {d} bytes\n", .{fb.pitch});
log.print(" format : {s}\n", .{@tagName(fb.format)});
log.print(" framebuffer: 0x{x:0>16}\n", .{fb.base});
log.print (" footprint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
// Summarise the physical memory the loader handed us. The array is danos's
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
var usable_pages: u64 = 0;
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
for (regions) |r| {
switch (r.kind) {
.usable => usable_pages += r.pages,
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
.mmio => {},
}
}
const total_pages = usable_pages + reserved_pages;
const total_bytes = total_pages * danos.page_size;
const gib = 1 << 30;
log.write("\ndanos: physical memory\n");
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
log.print(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
// Bring up the physical frame allocator over that map, and prove it works:
// allocate three frames, then hand them back.
pmm.init(boot_information.memory_map);
// Claim the AP trampoline's low (<1 MiB) page *now*, before paging and the heap
// draw down sub-1 MiB frames (the allocator scans upward from frame 0). Held
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0;
const s1 = pmm.stats();
log.print("\ndanos: frame allocator online\n", .{});
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
const f0 = pmm.alloc();
const f1 = pmm.alloc();
const f2 = pmm.alloc();
log.print(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
if (f0) |p| pmm.free(p);
if (f1) |p| pmm.free(p);
if (f2) |p| pmm.free(p);
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
// Switch off the firmware's page tables onto our own (with real permissions).
architecture.enablePaging(pmm.alloc, pmm.free, boot_information);
log.checkpoint(cp_paging);
log.print("\ndanos: paging enabled\n", .{});
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
// Bring up the kernel heap (dynamic allocation), built on the VMM.
heap.init();
log.checkpoint(cp_heap);
log.write("\ndanos: kernel heap online\n");
// Measure the amount of resources the kernel is actually using
const s2 = pmm.stats();
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
// Enumerate hardware from the firmware tables (ACPI here) into a generic
// device tree, then list it. Discovery walks ACPI memory directly (identity-
// mapped) and maps PCIe configuration space on demand via the VMM. A failure here is
// not fatal yet — log it and carry on.
const hal = platform.Hal{
.mapMmio = architecture.mapMmio,
.pioRead = architecture.pioRead,
.pioWrite = architecture.pioWrite,
};
if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| {
var device_tree = devtree;
log.write("\ndanos: device discovery online\n");
device_tree.dump(log.write);
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
// mmio_map operate on this flat, id-indexed table + claim map).
device_service.init(&device_tree);
if (device_service.dropped > 0) {
// Otherwise entirely silent: drivers would just never see that hardware.
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{device_service.dropped});
}
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
// land on. Every line stays masked until something binds it (ioapic.init).
irq.init();
// Power register map extracted from the FADT + AML, for confidence it parsed.
const pw = platform.powerInformation();
log.write("danos: power\n");
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
if (pw.s5) |s| {
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
} else {
log.write(" S5 slp_typ : (not found)\n");
}
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
// AML namespace parse integrity: consumed should equal total.
const am = platform.amlStats();
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
const pinfo = platform.platformInformation();
const hpet_base: u64 = if (device_tree.firstOfClass(.timer)) |t|
(if (t.firstResource(.memory)) |r| r.start else 0)
else
0;
var ioapic_base: u64 = 0;
var ioapic_gsi: u32 = 0;
if (device_tree.firstOfClass(.interrupt_controller)) |ic| {
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
}
var isos: [16]architecture.IsoEntry = undefined;
const iso_n = @min(pinfo.override_count, isos.len);
for (0..iso_n) |i| isos[i] = .{
.source = pinfo.overrides[i].source,
.gsi = pinfo.overrides[i].gsi,
.flags = pinfo.overrides[i].flags,
};
const pm_timer: ?architecture.PmTimer = if (pinfo.pm_timer.present())
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
else
null;
architecture.configurePlatform(.{
.pic_present = pinfo.pic_present,
.hpet_base = hpet_base,
.pm_timer = pm_timer,
.ioapic_base = ioapic_base,
.ioapic_gsi_base = ioapic_gsi,
.overrides = isos[0..iso_n],
});
if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address);
log.write("danos: platform\n");
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
log.print(" hpet base : 0x{x}\n", .{hpet_base});
log.print(" pm timer : {s} 0x{x} ({s})\n", .{ if (pinfo.pm_timer.mmio) "mmio" else "io", pinfo.pm_timer.address, if (pinfo.pm_timer_32bit) "32-bit" else "24-bit" });
if (pinfo.spcr_uart) |u| {
log.print(" console UART: {s} 0x{x} (SPCR type {d})\n", .{ if (u.mmio) "mmio" else "io", u.address, pinfo.spcr_kind });
} else {
log.write(" console UART: none in SPCR -> legacy COM1\n");
}
log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, architecture.irqRouteCount(), architecture.irqRouteRaw(0) });
const cores = platform.cpus();
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
if (platform.cpusDropped() > 0)
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
} else |err| {
log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
}
log.checkpoint(cp_discovery);
// Install the system_call handler (int 0x80 gate + system_call stub) once, before any
// user code runs.
process.init();
// Register the current context as the first task before enabling preemption.
scheduler.init(4);
log.checkpoint(cp_scheduler);
log.write("\ndanos: scheduler online\n");
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
// the timer preempts among tasks.
architecture.startTimer();
architecture.enableInterrupts();
log.checkpoint(cp_timer);
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
// Wake the other cores (application processors). A no-op on a single-core
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
bringUpSecondaries();
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
// Normal builds fall through to the idle halt.
if (build_options.test_case) |case| {
tests.run(case, boot_information);
architecture.halt();
}
log.checkpoint(cp_running);
status("kernel initialised.\n");
// Hand over to user space: load /sbin/init (read off the boot volume by the
// loader) and spawn it as a real ring-3 process, PID 1. It runs on its own
// address space, preemptively, alongside the kernel — no cooperative
// borrowing. This boot context then becomes the BSP's idle loop.
if (boot_information.init_len != 0) {
status("starting /sbin/init...\n");
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
process.spawnProcess(image, 4) catch |err| {
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
};
} else {
status("no /sbin/init on the boot volume.\n");
}
// Spawn the extra user binaries the loader ferried in the initrd (the VFS
// server, and later device drivers). For now the kernel launches them all;
// once init is a real service supervisor it will spawn them itself (system_spawn).
startInitrdBinaries(boot_information);
// Become the idle task: drop below every real task and halt until an
// interrupt. The timer keeps preempting into init and any other work.
scheduler.setPriority(0);
status("\nkernel idle; /sbin/init is running.\n");
architecture.halt();
}
/// Spawn every program bundled in the initrd as its own ring-3 process. A bad
/// image or a program that fails to load is logged and skipped — the rest of the
/// system still runs.
fn startInitrdBinaries(boot_information: *const danos.BootInformation) void {
if (boot_information.initrd_len == 0) return;
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
const rd = initrd.Reader.init(image) orelse {
status("initrd: bad image, skipping\n");
return;
};
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
statusPrint("starting /sbin/{s} (from initrd)...\n", .{item.name});
process.spawnProcess(item.blob, 4) catch |err| {
statusPrint("initrd: {s} failed to load: {s}\n", .{ item.name, @errorName(err) });
};
}
}
/// Wake the application processors the firmware left parked. Allocates the low
/// trampoline page (and makes it executable), then wakes each non-boot core in turn,
/// handing it a fresh kernel stack and its per-CPU slot. Cores that don't report in
/// are left parked — the running system is unaffected. See docs/smp.md.
fn bringUpSecondaries() void {
const cores = platform.cpus();
if (cores.len <= 1) return;
// A low (<1 MiB) frame was reserved at boot for the real-mode trampoline (a SIPI
// vector addresses it). It's kept for the system's life — armed only during a
// wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later.
if (ap_trampoline_page == 0) {
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
return;
}
architecture.setTrampolinePage(ap_trampoline_page);
architecture.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop
// Test hook: the smp-retry case forces the first wake to fail, so the retry below
// must still bring every core online. Inert in a normal build (test_case is null).
if (build_options.test_case) |tc| {
if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1);
}
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
for (cores[1..], 1..) |core, index| {
const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch {
log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id});
continue;
};
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
// This core's dedicated fault stack — allocated only now that the core is
// real, rather than reserved statically for every possible core.
const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch {
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
continue;
};
architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
const pc = scheduler.prepareSecondary(index, core.apic_id);
var attempt: u32 = 1;
while (attempt <= maximum_wake_attempts) : (attempt += 1) {
if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
pc.online = true;
log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
break;
}
if (attempt == maximum_wake_attempts)
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
}
}
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
}
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
/// touches the framebuffer.
fn status(message: []const u8) void {
log.write(message);
console.write(message);
}
fn statusPrint(comptime fmt: []const u8, args: anytype) void {
var buffer: [256]u8 = undefined;
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
}
/// Frames (4 KiB pages) to whole MiB.
fn mib(pages: u64) u64 {
return pages * danos.page_size / (1024 * 1024);
}
fn kib(frames: u64) u64 {
return frames * danos.page_size / (1024);
}
/// Report a CPU exception and halt **this core**. There's no fault recovery yet, so
/// the faulting core is terminal — but the fault is *contained* to it: on an
/// application processor only that core stops, and the rest of the system keeps
/// running (full recovery — kill the task, keep the core — is the resilience track,
/// see docs/resilience.md). The report names the core so an AP fault is attributed,
/// and goes to every output sink plus a POST code and a persistent breadcrumb.
fn onException(state: *const architecture.CpuState) noreturn {
log.checkpoint(cp_exception);
const core = scheduler.currentCpuIndex();
// A fault is user-facing enough to paint on screen too (via statusPrint), on
// top of the diagnostic log.
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
statusPrint(" error code : 0x{x}\n", .{state.error_code});
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
var buffer: [128]u8 = undefined;
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
architecture.halt();
}
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
/// no live console), then halt. Assumes no console — the sinks self-guard.
pub const panic = std.debug.FullPanic(struct {
fn panic(message: []const u8, first_trace_address: ?usize) noreturn {
_ = first_trace_address;
log.checkpoint(cp_panic);
log.recordPanic(message);
status("\nKERNEL PANIC: ");
status(message);
status("\n");
architecture.halt();
}
}.panic);
+174
View File
@@ -0,0 +1,174 @@
//! Physical memory manager: a bitmap frame allocator. It hands out and reclaims
//! 4 KiB physical frames — the primitive every later memory feature (page
//! tables, the heap) is built on top of.
//!
//! This is generic kernel code: it works on the neutral `danos.MemoryRegion`
//! array the loader hands over (see docs/memory-map.md), so it carries no UEFI
//! and nothing architecture-specific beyond the 4 KiB page.
const std = @import("std");
const danos = @import("danos");
const page_size = danos.page_size;
/// One bit per frame, covering physical RAM from 0 up to the highest usable
/// address: 1 = used/unavailable, 0 = free. The bitmap itself lives in a frame
/// we carve out of usable memory during init.
var bitmap: []u8 = &.{};
var total_frames: usize = 0;
var used_frames: usize = 0;
/// Where the next allocation scan begins, so we don't rescan from frame 0 every
/// time. Pulled back on free() so reclaimed low frames get reused.
var next_hint: usize = 0;
pub const Stats = struct {
total_frames: usize,
used_frames: usize,
free_frames: usize,
};
pub fn stats() Stats {
return .{
.total_frames = total_frames,
.used_frames = used_frames,
.free_frames = total_frames - used_frames,
};
}
inline fn bit(frame: usize) u3 {
return @intCast(frame & 7);
}
inline fn isUsed(frame: usize) bool {
return (bitmap[frame >> 3] >> bit(frame)) & 1 != 0;
}
inline fn setUsed(frame: usize) void {
bitmap[frame >> 3] |= @as(u8, 1) << bit(frame);
}
inline fn setFree(frame: usize) void {
bitmap[frame >> 3] &= ~(@as(u8, 1) << bit(frame));
}
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(map.regions)))[0..map.len];
}
/// Build the allocator from the loader's memory map. Reaches physical memory
/// (the region array, the bitmap's own storage) through the physmap, which the
/// loader's bootstrap tables already provide — so this works before the kernel
/// installs its own tables. Invariant: the bitmap lands in the first usable
/// region (lowest address), which must sit under the bootstrap physmap's reach
/// (4 GiB); it always does, as both this and the page-table allocator scan from
/// low addresses up.
pub fn init(map: danos.MemoryMap) void {
const regs = regions(map);
// 1. Size the bitmap to cover every frame up to the highest RAM address —
// including reserved RAM, so those frames are trackable (e.g. to free the
// boot buffers later). Only MMIO (device address space) is excluded.
// Everything starts unallocatable; usable regions are freed below.
var highest: u64 = 0;
for (regs) |r| {
if (r.kind == .mmio) continue;
const end = r.base + r.pages * page_size;
if (end > highest) highest = end;
}
total_frames = @intCast(highest / page_size);
if (total_frames == 0) @panic("pmm: no usable memory");
const bitmap_bytes = (total_frames + 7) / 8;
const bitmap_pages = (bitmap_bytes + page_size - 1) / page_size;
// 2. Park the bitmap in the first usable region large enough to hold it.
// Start at least one page in, so we never place it on frame 0 (which is
// kept reserved as the "none" address, and is an awkward pointer besides).
var storage: ?u64 = null;
for (regs) |r| {
if (r.kind != .usable) continue;
const base = if (r.base == 0) page_size else r.base;
const skipped = (base - r.base) / page_size;
if (r.pages - skipped >= bitmap_pages) {
storage = base;
break;
}
}
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
bitmap = @as([*]u8, @ptrFromInt(danos.physicalToVirtual(bitmap_base)))[0..bitmap_bytes];
// 3. Start with everything marked used, then free the usable regions. Doing
// it this way means every gap, reserved span and MMIO hole is unallocatable
// by default — we only ever hand back memory the firmware called usable.
@memset(bitmap, 0xff);
used_frames = total_frames;
for (regs) |r| {
if (r.kind != .usable) continue;
var f: usize = @intCast(r.base / page_size);
const end = f + @as(usize, @intCast(r.pages));
while (f < end and f < total_frames) : (f += 1) {
setFree(f);
used_frames -= 1;
}
}
// 4. Take back the frames the bitmap occupies, and frame 0 — so a 0 result
// stays reserved to mean "no frame".
reserve(bitmap_base, bitmap_pages);
reserve(0, 1);
}
/// Mark `count` frames from physical `base` as used, counting only those that
/// were actually free.
fn reserve(base: u64, count: usize) void {
var f: usize = @intCast(base / page_size);
const end = f + count;
while (f < end and f < total_frames) : (f += 1) {
if (!isUsed(f)) {
setUsed(f);
used_frames += 1;
}
}
}
/// Allocate one physical frame, or null if none are free. The address is
/// page-aligned; the frame's contents are undefined.
pub fn alloc() ?u64 {
var scanned: usize = 0;
var f = next_hint;
while (scanned < total_frames) : (scanned += 1) {
if (f >= total_frames) f = 0;
if (!isUsed(f)) {
setUsed(f);
used_frames += 1;
next_hint = f + 1;
return @as(u64, f) * page_size;
}
f += 1;
}
return null; // out of physical memory
}
/// Allocate one free frame whose physical address is below `limit`, or null if
/// none is free down there. The AP trampoline needs this: an x86 STARTUP IPI vectors
/// a waking core to physical `vector << 12`, and `vector` is a byte — so the
/// trampoline must live under 1 MiB. A short linear scan of the low frames; only run
/// a handful of times at boot, so it needn't be fast.
pub fn allocBelow(limit: u64) ?u64 {
const cap = @min(total_frames, @as(usize, @intCast(limit / page_size)));
var f: usize = 1; // frame 0 stays reserved as the "none" address
while (f < cap) : (f += 1) {
if (!isUsed(f)) {
setUsed(f);
used_frames += 1;
return @as(u64, f) * page_size;
}
}
return null;
}
/// Return a frame obtained from alloc() to the pool. Bogus or double frees are
/// ignored rather than corrupting the count.
pub fn free(address: u64) void {
const f: usize = @intCast(address / page_size);
if (f >= total_frames or !isUsed(f)) return;
setFree(f);
used_frames -= 1;
if (f < next_hint) next_hint = f;
}
+579
View File
@@ -0,0 +1,579 @@
//! User-space processes: loading a user ELF and running it in ring 3. danos is a
//! microkernel, so this only ever loads *user* binaries — there is no kernel-space
//! loader; in-kernel code is linked into the kernel image, not loaded here.
//!
//! Two entry points:
//! - `spawnProcess` loads a user ELF (`/sbin/init`, and later servers/drivers)
//! into a fresh address space and schedules it as a real preemptive ring-3
//! process on its own page tables. This is the production path.
//! - `run` executes a raw code blob (the user-pf isolation test program) on the
//! *current* kernel context via the borrowed-thread path — a minimal probe of
//! the ring-transition mechanisms, kept for that test.
//! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program
//! talks to the kernel only through the system_call instruction (or the int 0x80
//! gate). The shared handler is installed once by `init`.
//!
//! Borrowed-path caveat (`run` only): it publishes TSS.rsp0 on the *current*
//! core and uses a single global unwind slot (`user_saved_rsp` in isr.s), so the
//! caller must disable preemption and only one core may be inside it at a time.
//! Real processes (`spawnProcess`) have none of these limits — the scheduler
//! maintains rsp0/CR3 per switch.
const std = @import("std");
const elf = std.elf;
const danos = @import("danos");
const architecture = @import("architecture");
const pmm = @import("pmm.zig");
const scheduler = @import("scheduler.zig");
const sync = @import("sync.zig");
const ipc = @import("ipc-synchronous.zig");
const device_service = @import("device-service.zig");
const irq = @import("irq.zig");
const log = @import("log.zig");
const page_size = danos.page_size;
const SystemCall = danos.SystemCall;
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
/// the identity map (low indices) and the vmm test address (index 128), so
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
/// An ELF image may occupy [code_virtual, stack_virtual); the stack page sits above.
pub const code_virtual: u64 = 0x0000_7000_0000_0000;
pub const stack_virtual: u64 = 0x0000_7000_0020_0000;
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
/// 1 GiB window is far more than any user heap needs today.
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
/// End of the user (low) canonical half. Any legitimate user pointer is below it;
/// used to bound the addresses a system_call will dereference on the caller's behalf.
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
/// `Task.device_map_next`.
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
/// chunks, so this bound is generous; it also caps the frame scratch array below.
const maximum_mmap_pages = 256;
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
/// The isolation-proof program: reads a kernel-only page, must #PF.
pub fn pfBlob() []const u8 {
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
}
/// What debug_write syscalls produced (accumulated), and the exit system_call's code.
pub var write_buffer: [256]u8 = undefined;
pub var write_len: usize = 0;
pub var write_from_user: bool = false;
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
pub var exit_code: u64 = 0;
/// The system_call surface, dispatched on the saved system_call number (`danos.SystemCall`).
/// This is the microkernel-minimal set — memory + scheduling only; file/device
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
/// written back into the trap frame, since the entry paths restore user registers
/// from it. One handler serves both the system_call/sysret and int-0x80 entry paths.
///
/// Install it once at boot (before any user code runs) via `init`.
pub fn init() void {
architecture.setSystemCallHandler(system_call);
}
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
fn fail(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, @bitCast(@as(i64, -1)));
}
fn system_call(state: *architecture.CpuState) void {
switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) {
.exit => {
exit_code = architecture.systemCallArg(state, 0);
// A scheduled process drops its endpoint references, frees its address
// space, and reschedules; a borrowed test thread unwinds back to the
// kernel that entered it.
if (scheduler.currentIsUserProcess()) {
// Unbind before closeHandles: dropping the last reference destroys the
// Endpoint, and a still-bound GSI would have an ISR call
// notifyFromIsr on freed memory the next time the device fired.
// unbindAll also leaves the line masked, so a dead driver's device
// goes quiet rather than storming.
releaseIrqs(scheduler.current());
ipc.closeHandles(scheduler.current());
scheduler.exitUser();
} else architecture.userExit();
},
.yield => {
scheduler.yield();
architecture.setSystemCallResult(state, 0);
},
.sleep => {
scheduler.sleep(architecture.systemCallArg(state, 0));
architecture.setSystemCallResult(state, 0);
},
.debug_write => systemDebugWrite(state),
.mmap => systemMmap(state),
.munmap => systemMunmap(state),
.create_endpoint => systemCreateEndpoint(state),
.ipc_register => systemIpcRegister(state),
.ipc_lookup => systemIpcLookup(state),
.ipc_call => systemIpcCall(state),
.ipc_reply_wait => systemIpcReplyWait(state),
.device_enumerate => systemDeviceEnumerate(state),
.device_claim => systemDeviceClaim(state),
.mmio_map => systemMmioMap(state),
.irq_bind => systemIrqBind(state),
.irq_ack => systemIrqAck(state),
.device_register => systemDeviceRegister(state),
_ => fail(state),
}
}
/// Return `-errno` in the system_call result register.
fn failErr(state: *architecture.CpuState, errno: i64) void {
architecture.setSystemCallResult(state, @bitCast(-errno));
}
/// create_endpoint() -> handle: allocate an endpoint and install it in the
/// caller's handle table.
fn systemCreateEndpoint(state: *architecture.CpuState) void {
const endpoint = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM);
const h = ipc.installHandle(scheduler.current(), endpoint);
if (h < 0) {
ipc.dropRef(endpoint);
return failErr(state, ipc.ENOSPC);
}
architecture.setSystemCallResult(state, @intCast(h));
}
/// ipc_register(service_id, handle): publish the caller's endpoint under a
/// well-known id so other processes can find it.
fn systemIpcRegister(state: *architecture.CpuState) void {
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
}
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
/// handle to it in the caller.
fn systemIpcLookup(state: *architecture.CpuState) void {
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
const h = ipc.installHandle(scheduler.current(), endpoint);
if (h < 0) {
ipc.dropRef(endpoint);
return failErr(state, ipc.ENOSPC);
}
architecture.setSystemCallResult(state, @intCast(h));
}
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
/// Blocks until the server replies; the trap frame lives on this task's kernel
/// stack, so it survives the block and receives the result on resume.
fn systemIpcCall(state: *architecture.CpuState) void {
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4));
architecture.setSystemCallResult(state, @bitCast(r));
}
/// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len,
/// with the sender's badge in the secondary result register (rdx).
fn systemIpcReplyWait(state: *architecture.CpuState) void {
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
var badge: u64 = 0;
const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), &badge);
architecture.setSystemCallResult(state, @bitCast(r));
architecture.setSystemCallResult2(state, badge);
}
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
/// buffer (up to `maximum` entries), returning the total device count.
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
const buffer_ptr = architecture.systemCallArg(state, 0);
const maximum = architecture.systemCallArg(state, 1);
const t = scheduler.current();
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
const sz = @sizeOf(danos.DeviceDescriptor);
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
const out: [*]danos.DeviceDescriptor = @ptrFromInt(buffer_ptr);
architecture.setSystemCallResult(state, device_service.enumerate(out[0..@intCast(cap)]));
}
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
fn systemDeviceClaim(state: *architecture.CpuState) void {
if (device_service.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
architecture.setSystemCallResult(state, 0)
else
fail(state);
}
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
/// this address space (strong-uncacheable) and return the register base address.
/// The claim is the capability — a process can only map hardware it owns.
fn systemMmioMap(state: *architecture.CpuState) void {
const device_id = architecture.systemCallArg(state, 0);
const resource_index = architecture.systemCallArg(state, 1);
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const owner = device_service.ownerOf(device_id) orelse return fail(state);
if (owner != t.id) return fail(state); // not claimed by this process
const r = device_service.resourceOf(device_id, resource_index) orelse return fail(state);
if (r.kind != @intFromEnum(danos.ResourceKind.memory)) return fail(state);
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
const first = r.start & ~@as(u64, page_size - 1);
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
const pages = (last - first) / page_size + 1;
const base_v = t.device_map_next;
if (base_v + pages * page_size > device_arena_end) return fail(state);
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len);
t.device_map_next = base_v + pages * page_size;
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
}
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
/// this process has claimed. The bus-driver primitive: a process that owns a bus
/// enumerates it and hands each device it finds to the table, where a class driver
/// can claim it.
///
/// The kernel copies the descriptor into a kernel local *once* (via the same
/// physmap-walking path as IPC, so an unmapped user page fails the call rather than
/// faulting the kernel), then validates and uses that copy — no second read of user
/// memory, so nothing it checked can change under it. It refuses any child resource
/// that escapes the parent's windows: a descriptor is a licence to map physical
/// memory, so a bus may only subdivide what it already holds. `id`/`parent` in the
/// supplied descriptor are ignored.
fn systemDeviceRegister(state: *architecture.CpuState) void {
const parent_id = architecture.systemCallArg(state, 0);
const descriptor_ptr = architecture.systemCallArg(state, 1);
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
var descriptor: danos.DeviceDescriptor = undefined;
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
const id = device_service.register(parent_id, t.id, &descriptor) catch return fail(state);
architecture.setSystemCallResult(state, id);
}
/// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed
/// (which is what frees the endpoints an ISR would otherwise notify into).
fn releaseIrqs(t: *scheduler.Task) void {
const flags = sync.enter();
defer sync.leave(flags);
irq.releaseOwner(t.id);
}
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
/// The two checks are the whole security story: the device must be *claimed* by the
/// caller, and the resource must be one of that device's `irq` resources as recorded
/// by discovery. Neither a raw GSI nor an unclaimed device can get through — which
/// is why irq_bind takes a resource index and not an interrupt number.
fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
const owner = device_service.ownerOf(device_id) orelse return null;
if (owner != t.id) return null;
const r = device_service.resourceOf(device_id, resource_index) orelse return null;
if (r.kind != @intFromEnum(danos.ResourceKind.irq)) return null;
if (r.start >= irq.maximum_gsi) return null;
return @intCast(r.start);
}
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
/// endpoint as an asynchronous IPC notification. The driver then blocks in
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
fn systemIrqBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
const flags = sync.enter();
defer sync.leave(flags);
irq.bind(gsi, endpoint, t.id) catch return fail(state);
architecture.setSystemCallResult(state, 0);
}
/// irq_ack(device_id, resource_index) -> 0/-1: re-arm a bound IRQ. The ISR left the line
/// masked (it could not quiet the device — that's this driver's job), so nothing
/// more arrives until the driver says it has serviced the hardware.
fn systemIrqAck(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
return fail(state);
const flags = sync.enter();
defer sync.leave(flags);
if (irq.ack(gsi)) architecture.setSystemCallResult(state, 0) else fail(state);
}
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
/// A bring-up diagnostic — real output goes through the VFS/console later.
///
/// The pointer must lie in the user (low) half, so kernel addresses and
/// non-canonical values fall outside it and the read below can't be steered at
/// kernel data. Length is checked first so the upper-bound add can't overflow.
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
/// the user half passes the check and the read #PFs -> on_fault halts — a
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
fn systemDebugWrite(state: *architecture.CpuState) void {
const ptr = architecture.systemCallArg(state, 0);
const len = architecture.systemCallArg(state, 1);
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
const source: [*]const u8 = @ptrFromInt(ptr);
@memcpy(write_buffer[0..len], source[0..len]); // keep the latest message
write_len = len;
write_from_user = architecture.fromUser(state);
write_count += 1;
log.write("DANOS-INIT: ");
log.write(source[0..len]);
architecture.setSystemCallResult(state, len);
} else {
fail(state);
}
}
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
/// base virtual address. `prot` is accepted but not yet honoured (grants are
/// always RW+NX; W^X for user code stays with the ELF loader). Failure returns
/// -1. The user-space allocator (lib `runtime`) carves these pages into malloc blocks.
fn systemMmap(state: *architecture.CpuState) void {
const len = architecture.systemCallArg(state, 0);
const t = scheduler.current();
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
const pages = (len + page_size - 1) / page_size;
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
const base = t.heap_next;
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
// (no partially-mapped grant leaks into the address space).
var frames: [maximum_mmap_pages]u64 = undefined;
var got: usize = 0;
while (got < pages) : (got += 1) {
frames[got] = pmm.alloc() orelse {
for (frames[0..got]) |f| pmm.free(f);
return fail(state);
};
}
for (frames[0..pages], 0..) |frame, i| {
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
@memset(destination[0..page_size], 0); // hand out zeroed memory
architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
}
t.heap_next = base + pages * page_size;
architecture.setSystemCallResult(state, base);
}
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
/// each page and frees its frame. The arena is a bump allocator, so the virtual
/// range is not recycled (the user-space allocator reuses freed *blocks* itself);
/// this just returns the physical frames to the kernel. Returns 0, or -1 if the
/// range is not page-aligned or lies outside the arena.
fn systemMunmap(state: *architecture.CpuState) void {
const base = architecture.systemCallArg(state, 0);
const len = architecture.systemCallArg(state, 1);
const t = scheduler.current();
if (t.aspace == 0 or base % page_size != 0) return fail(state);
const pages = (len + page_size - 1) / page_size;
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
for (0..pages) |i| {
const va = base + i * page_size;
if (architecture.translate(t.aspace, va)) |physical| {
architecture.unmapUserPageInto(t.aspace, va);
pmm.free(physical);
}
}
architecture.setSystemCallResult(state, 0);
}
/// Reset the recorded system_call evidence before a user-mode run.
fn resetRecords() void {
write_len = 0;
write_from_user = false;
write_count = 0;
exit_code = 0;
}
pub const RunError = error{ ProgramTooBig, OutOfMemory };
/// Map `blob` at code_virtual with a fresh user stack, drop to ring 3, and return
/// once the program exits via system_call 0. See the migration caveat in the module
/// doc. A program that faults instead never returns (on_fault halts the core).
pub fn run(blob: []const u8) RunError!void {
if (blob.len > page_size) return error.ProgramTooBig;
const code_frame = pmm.alloc() orelse return error.OutOfMemory;
const stack_frame = pmm.alloc() orelse {
pmm.free(code_frame);
return error.OutOfMemory;
};
// Fill the code frame through the physmap (supervisor RW): the user-facing
// mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is
// padded with int3 so a stray jump traps instead of sliding.
const code: [*]u8 = @ptrFromInt(danos.physicalToVirtual(code_frame));
@memcpy(code[0..blob.len], blob);
@memset(code[blob.len..page_size], 0xCC);
architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X
architecture.mapUserPage(stack_virtual, stack_frame, true, false); // RW + NX
resetRecords();
architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_virtual + page_size);
// Back via the exit system_call; the interrupt gate left IF clear.
architecture.enableInterrupts();
architecture.unmapPage(code_virtual);
architecture.unmapPage(stack_virtual);
pmm.free(code_frame);
pmm.free(stack_frame);
}
// --- user ELF loading (/sbin/init) ------------------------------------------
pub const InitError = error{
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
BadEntry, // e_entry not inside an executable segment
ProgramTooBig, // more pages than the loader's budget
OutOfMemory,
};
const maximum_segments = 16;
const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
const Segment = struct {
vaddr: u64,
memsz: u64,
filesz: u64,
off: u64,
writable: bool,
executable: bool,
fn pages(self: Segment) u64 {
return (self.memsz + page_size - 1) / page_size;
}
};
/// Parse and validate every PT_LOAD before touching memory. Bounds are checked
/// against the image and the user region; segments must be page-aligned,
/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only
/// segment, mapped RO+NX).
fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!struct { count: usize, entry: u64 } {
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf;
const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]);
if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf;
if (image[elf.EI_CLASS] != elf.ELFCLASS64) return error.BadElf;
if (ehdr.e_machine != .X86_64) return error.BadElf;
if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation
if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf;
if (ehdr.e_phnum > maximum_segments) return error.BadElf;
const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize;
if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf;
var count: usize = 0;
var total_pages: u64 = 0;
for (0..ehdr.e_phnum) |i| {
const off = ehdr.e_phoff + i * ehdr.e_phentsize;
const phdr = std.mem.bytesToValue(elf.Elf64_Phdr, image[off..][0..@sizeOf(elf.Elf64_Phdr)]);
if (phdr.p_type != elf.PT_LOAD) continue;
if (phdr.p_memsz == 0) continue;
if (phdr.p_vaddr % page_size != 0) return error.BadSegment;
if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment;
if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment;
// Inside the user image region, strictly below the stack page.
if (phdr.p_vaddr < code_virtual) return error.BadSegment;
if (phdr.p_memsz > stack_virtual - phdr.p_vaddr) return error.BadSegment;
const w = phdr.p_flags & elf.PF_W != 0;
const x = phdr.p_flags & elf.PF_X != 0;
if (w and x) return error.BadSegment; // W^X, even for init
const seg = Segment{
.vaddr = phdr.p_vaddr,
.memsz = phdr.p_memsz,
.filesz = phdr.p_filesz,
.off = phdr.p_offset,
.writable = w,
.executable = x,
};
// No overlap with any earlier segment (page-granular, since mapping is).
for (segs[0..count]) |other| {
const a_end = seg.vaddr + seg.pages() * page_size;
const b_end = other.vaddr + other.pages() * page_size;
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
}
total_pages += seg.pages();
if (total_pages > maximum_pages) return error.ProgramTooBig;
segs[count] = seg;
count += 1;
}
if (count == 0) return error.BadElf;
// The entry point must land inside an executable segment.
for (segs[0..count]) |seg| {
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
return .{ .count = count, .entry = ehdr.e_entry };
}
return error.BadEntry;
}
/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
/// On a later failure the whole address space is torn down, which frees every
/// frame mapped into it — so no per-page rollback list is needed here.
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
const frame = pmm.alloc() orelse return error.OutOfMemory;
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
@memset(destination[0..page_size], 0);
const page_off = page_index * page_size;
if (page_off < seg.filesz) {
const n = @min(page_size, seg.filesz - page_off);
@memcpy(destination[0..n], image[seg.off + page_off ..][0..n]);
}
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
}
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
/// ring-3 process at `priority`. Returns immediately — the process runs
/// preemptively on its own page tables alongside everything else, and its exit
/// is handled by the system_call layer. The whole build (address space + ELF load +
/// task) runs under the kernel lock so it appears atomically and can't race
/// pmm/heap on another core.
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
var segs: [maximum_segments]Segment = undefined;
const parsed = try parseSegments(image, &segs);
const flags = sync.enter();
defer sync.leave(flags);
const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory;
errdefer architecture.destroyAddressSpace(aspace);
for (segs[0..parsed.count]) |seg| {
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
}
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
architecture.mapUserPageInto(aspace, stack_virtual, stack_frame, true, false); // RW + NX
if (!scheduler.spawnUserLocked(aspace, parsed.entry, stack_virtual + page_size, priority))
return error.OutOfMemory;
}
+561
View File
@@ -0,0 +1,561 @@
//! The scheduler: fixed-priority preemptive multitasking.
//!
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
//! priority ready task always runs**; within a priority level, tasks round-robin.
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
//! docs/vision.md).
//!
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
//! makes those two paths coexist.
//!
//! Cross-core safety is the **big kernel lock** (`sync.zig`): every critical
//! section here runs under it, and it is held across a context switch and released
//! by the task that resumes (see sync.zig's hand-off rule). On a single core the
//! lock is never contended, so the behaviour is exactly the old interrupt-flag
//! model; it's what lets a second core enter `schedule()` without corrupting the
//! shared queues.
const std = @import("std");
const parameters = @import("parameters");
const architecture = @import("architecture");
const heap = @import("heap.zig");
const sync = @import("sync.zig");
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
pub const Priority = u3;
const number_priorities = 8;
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
const State = enum { free, ready, running, blocked };
pub const Task = struct {
id: u32 = 0,
state: State = .free,
priority: Priority = 0,
sp: usize = 0, // saved stack pointer, valid while not running
stack: []u8 = &.{},
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
// Physical root of this task's address space, or 0 for a kernel task (which
// runs on the shared kernel page tables). A user task carries its own.
aspace: u64 = 0,
user_ip: u64 = 0, // user-mode entry point (user task only)
user_sp: u64 = 0, // user-mode stack pointer (user task only)
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
// as the user heap grows; user task only.
heap_next: u64 = 0,
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
device_map_next: u64 = 0,
// --- synchronous IPC (ipc_sync.zig) ---
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
// opaque here so the scheduler and IPC modules don't import each other.
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
// A server holds the caller it currently owes a reply to (set by ReplyWait's
// receive, cleared when it replies). A client, while blocked in Call, records
// its message + reply buffers here and its result lands in `ipc_status`.
ipc_client: ?*Task = null,
ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS)
ipc_send_len: u64 = 0,
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
ipc_reply_cap: u64 = 0,
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
};
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
pub const ipc_maximum_handles = 16;
var tasks = [_]Task{.{}} ** maximum_tasks;
var next_id: u32 = 1;
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
/// queue of tasks **pinned** to it. One entry per core; the architecture layer stashes a
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
/// with a single read and no lock.
///
/// Most work stays in the **global** ready queue (below), which any idle core pulls
/// from — work-conserving. A task given an *affinity* instead goes to that core's
/// `pinned_*` queue and is only ever run there (no surprise migration — the more
/// real-time-predictable model, docs/smp.md). The two queues are merged at selection
/// time. Both are still mutated only under the big kernel lock, so one core enqueuing
/// into another core's pinned queue is safe.
pub const PerCpu = struct {
current: *Task = undefined, // the task running on this core
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
index: u32 = 0, // dense 0-based core index
online: bool = false, // has this core finished bring-up?
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
pinned_bitmap: u8 = 0,
};
const maximum_cpus = parameters.maximum_cpus;
var cpus = [_]PerCpu{.{}} ** maximum_cpus;
/// This core's per-CPU state, via the architecture layer's GS-base pointer. Valid only once
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
inline fn thisCpu() *PerCpu {
return @ptrFromInt(architecture.cpuLocal());
}
/// The task running on this core — the per-CPU replacement for the old global
/// `current`. A convenience reader; writes go through `thisCpu().current`.
pub inline fn current() *Task {
return thisCpu().current;
}
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
// are shared across all cores and mutated only under the big kernel lock.
var ready_head: [number_priorities]?*Task = .{null} ** number_priorities;
var ready_tail: [number_priorities]?*Task = .{null} ** number_priorities;
var ready_bitmap: u8 = 0;
var preemption_enabled = true;
/// Bring up scheduling on the bootstrap processor: register the currently-running
/// kernel context as task 0, publish this core's per-CPU state (via the GS base),
/// give the core an idle task, and hook the timer for preemption. Runs once, at
/// boot, before interrupts are enabled — so no lock is needed here.
pub fn init(boot_priority: Priority) void {
const pc = &cpus[0];
pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() };
architecture.setCpuLocal(0, @intFromPtr(pc));
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
pc.current = &tasks[0];
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
architecture.setTickHook(tick);
}
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
/// for the next interrupt at near-zero power (see docs/halting.md).
fn idle() void {
while (true) asm volatile ("hlt");
}
/// Reserve and initialise the per-CPU slot for an application processor at dense
/// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a
/// pointer the architecture bring-up hands to the core (it publishes it in its GS base).
/// Called on the BSP before waking each AP; the AP marks itself `online`.
pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu {
const pc = &cpus[index];
pc.* = .{ .index = @intCast(index), .hw_id = hw_id, .online = false };
return pc;
}
/// Entry for an application processor once the architecture layer has set up its per-CPU
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
/// interrupts enabled the timer preempts this idle context into whatever the global
/// ready queue offers, so the core runs real work in parallel with the others. The
/// `.c` calling convention lets the architecture trampoline path jump here. Never returns.
pub fn secondaryMain() callconv(.c) noreturn {
const flags = sync.enter();
const pc = thisCpu();
const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");
t.* = .{ .id = next_id, .state = .running, .priority = 0 };
next_id += 1;
pc.current = t;
pc.idle = t;
pc.online = true;
pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
sync.leave(flags);
architecture.enableInterrupts(); // the timer now preempts this idle context into work
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
}
/// Number of cores that have finished bring-up (the BSP plus every online AP).
pub fn onlineCount() usize {
var n: usize = 0;
for (&cpus) |*pc| {
if (pc.online) n += 1;
}
return n;
}
/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue;
/// everything else goes to the shared global queue.
fn enqueue(t: *Task) void {
if (t.affinity) |cpu| {
const pc = &cpus[cpu];
enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
} else {
enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t);
}
}
fn enqueueTo(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, t: *Task) void {
t.next = null;
const p: usize = t.priority;
if (tail[p]) |tl| tl.next = t else head[p] = t;
tail[p] = t;
bitmap.* |= @as(u8, 1) << t.priority;
}
/// The highest non-empty priority level in a bitmap, or -1 if empty.
fn topLevel(bitmap: u8) i32 {
if (bitmap == 0) return -1;
return @as(i32, number_priorities - 1) - @as(i32, @clz(bitmap));
}
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task
/// wins an equal-priority tie, so it can't be starved by global work at its level.
fn dequeueHighest(pc: *PerCpu) ?*Task {
const g = topLevel(ready_bitmap);
const p = topLevel(pc.pinned_bitmap);
if (g < 0 and p < 0) return null;
if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p));
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
}
fn dequeueFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
const t = head[level].?;
head[level] = t.next;
if (head[level] == null) {
tail[level] = null;
bitmap.* &= ~(@as(u8, 1) << @intCast(level));
}
t.next = null;
return t;
}
/// Create a task that runs `entry` at `priority`, runnable on any core. It becomes
/// ready immediately. Takes the kernel lock: it mutates the shared task table and
/// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be
/// serialised.
pub fn spawn(entry: *const fn () void, priority: Priority) void {
const flags = sync.enter();
_ = create(entry, priority, null);
sync.leave(flags);
}
/// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there.
/// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case
/// the task is still created but left unpinned (so it runs *somewhere* rather than
/// stranding in a queue no core services). Callers that require the pin (e.g. tests)
/// should check the result.
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
const flags = sync.enter();
defer sync.leave(flags);
const ok = cpu < maximum_cpus and cpus[cpu].online;
_ = create(entry, priority, if (ok) cpu else null);
return ok;
}
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
/// in user mode at `entry` on `user_sp`. It gets a fresh kernel stack for
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
/// Returns false (creating nothing) if the table is full or out of memory.
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
/// across the whole spawn, so the address space and the task appear atomically).
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority) bool {
const t = freeSlot() orelse return false;
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
t.* = .{
.id = next_id,
.state = .ready,
.priority = priority,
.stack = stack,
.aspace = aspace,
.user_ip = entry,
.user_sp = user_sp,
};
next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len;
t.kstack_top = top;
// First switch-in lands in startUserTask (no register smuggling — it reads
// the user entry/stack from the Task itself).
t.sp = architecture.initTaskStack(top, @intFromPtr(&startUserTask));
enqueue(t);
return true;
}
/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It
/// drops to ring 3 at the task's recorded entry/stack. Reading them from the
/// Task avoids smuggling values through callee-saved registers across the
/// context switch and lock release.
fn startUserTask() void {
const t = current();
var buffer: [96]u8 = undefined;
architecture.serialWrite(std.fmt.bufPrint(&buffer, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
}
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
/// Returns the new task so a core can keep a handle to its idle task.
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
const t = freeSlot() orelse @panic("sched: task table full");
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len;
t.kstack_top = top;
t.sp = architecture.initTaskStack(top, @intFromPtr(entry));
enqueue(t);
return t;
}
fn freeSlot() ?*Task {
for (&tasks) |*t| {
if (t.state == .free) return t;
}
return null;
}
/// Pick the highest-priority ready task and switch this core to it. The big kernel
/// lock must be held by the caller (which also keeps local interrupts disabled);
/// it serialises every core's scheduling, so no other core can touch the shared
/// queues while we requeue `previous` and dequeue `next`. A dequeued task is `.ready`,
/// never running elsewhere, so two cores never run the same task.
fn schedule() void {
const pc = thisCpu();
const previous = pc.current;
if (previous.state == .running) {
previous.state = .ready;
enqueue(previous); // back of its level's queue (round-robin)
}
const next = dequeueHighest(pc) orelse {
previous.state = .running; // nothing else ready — keep running
return;
};
next.state = .running;
pc.current = next;
if (next != previous) switchTo(pc, &previous.sp, next);
}
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
/// user-mode interrupt lands on a good stack) and its address space (only when
/// it differs from what's loaded — every page-table switch is a full TLB flush),
/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
/// workload. The big kernel lock is held and interrupts are off throughout, so no
/// interrupt can observe a half-updated (kernel stack, address space) pair.
/// `save_sp` receives the outgoing task's stack pointer.
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable();
if (want != pc.loaded_aspace) {
architecture.loadPageTable(want);
pc.loaded_aspace = want;
}
architecture.switchContext(save_sp, next.sp);
}
/// Voluntarily give up the CPU to the next ready task.
pub fn yield() void {
const flags = sync.enter();
schedule();
sync.leave(flags);
}
/// Block the current task for `ms` milliseconds, then let it become runnable
/// again. The idle task (or other work) runs in the meantime.
pub fn sleep(ms: u64) void {
const flags = sync.enter();
const t = current();
t.wake_at = architecture.millis() + ms;
t.state = .blocked;
schedule(); // current is blocked, so schedule() won't re-enqueue it
sync.leave(flags);
}
// --- event-based blocking -------------------------------------------------
//
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
// message). Tasks link into it through the same `next` field the ready queues
// use — a task is in exactly one queue at a time. These are the primitive locks,
// semaphores and IPC channels are built on.
pub const WaitQueue = struct {
head: ?*Task = null,
};
/// Block the current task on `wait_queue` and switch away. Precondition: the big kernel
/// lock is held (so a condition can be checked and the block committed atomically;
/// it also keeps local interrupts disabled). On return — when woken — the lock is
/// still held.
pub fn waitLocked(wait_queue: *WaitQueue) void {
const t = current();
t.state = .blocked;
t.next = wait_queue.head;
wait_queue.head = t;
schedule();
}
/// Move the highest-priority waiter on `wait_queue` (if any) to the ready queue.
/// Precondition: the big kernel lock is held. Does not preempt — the caller decides.
pub fn wakeLocked(wait_queue: *WaitQueue) void {
// Find the highest-priority waiter (bounded scan) and unlink it.
var best_previous: ?*Task = null;
var best: ?*Task = null;
var previous: ?*Task = null;
var node = wait_queue.head;
while (node) |t| : ({
previous = t;
node = t.next;
}) {
if (best == null or t.priority > best.?.priority) {
best = t;
best_previous = previous;
}
}
const t = best orelse return;
if (best_previous) |p| p.next = t.next else wait_queue.head = t.next;
t.state = .ready;
enqueue(t);
}
/// Block the current task and switch away, without putting it on any wait queue —
/// the caller has already linked it wherever it belongs (e.g. an endpoint's sender
/// FIFO). Precondition: the big kernel lock is held; still held on return (when the
/// task is made ready again). The IPC layer's counterpart to `waitLocked`.
pub fn blockCurrentLocked() void {
current().state = .blocked;
schedule();
}
/// Make a specific (currently blocked) task ready to run again. Precondition: the
/// big kernel lock is held. Used by the IPC layer to wake a specific caller/server
/// rather than "some waiter on a queue".
pub fn readyLocked(t: *Task) void {
t.state = .ready;
enqueue(t);
}
/// Block on `wait_queue` (a self-contained critical section).
pub fn wait(wait_queue: *WaitQueue) void {
const flags = sync.enter();
waitLocked(wait_queue);
sync.leave(flags);
}
/// Wake the highest-priority waiter on `wait_queue`, preempting if it outranks us.
pub fn wake(wait_queue: *WaitQueue) void {
const flags = sync.enter();
const pc = thisCpu();
wakeLocked(wait_queue);
// If a task this core would now pick outranks the running one, run it at once.
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
// next tick; this core doesn't preempt for work it can't run.)
if (highestReadyPriority(pc)) |p| {
if (p > pc.current.priority) schedule();
}
sync.leave(flags);
}
/// The highest-priority task core `pc` could run right now — the better of the global
/// queue and this core's pinned queue — or null if it would fall back to idle.
fn highestReadyPriority(pc: *PerCpu) ?Priority {
const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap));
if (top < 0) return null;
return @intCast(top);
}
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
fn wakeExpired() void {
const now = architecture.millis();
for (&tasks) |*t| {
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
t.wake_at = 0;
t.state = .ready;
enqueue(t);
}
}
}
/// Called from the timer interrupt (interrupts already disabled): wake due
/// sleepers, then preempt. Takes the kernel lock like any other critical section,
/// but releases it *without* touching the interrupt flag — the handler's `iretq`
/// restores the interrupted context's flags, so re-enabling here would open a
/// nested-interrupt window before the return.
pub fn tick() void {
_ = sync.enter();
wakeExpired();
if (preemption_enabled) schedule();
sync.leaveIsr();
}
/// Enable or disable timer-driven preemption (cooperative-only when off).
pub fn setPreemption(enabled: bool) void {
preemption_enabled = enabled;
}
/// End the current task and switch away for good; never returns. The task's stack
/// is leaked for now (no reaper yet). Acquires the kernel lock and hands it off to
/// the task we switch into (which releases it) — this frame never returns to leave.
pub fn exit() noreturn {
_ = sync.enter();
const pc = thisCpu();
pc.current.state = .free;
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
next.state = .running;
pc.current = next;
var discard: usize = 0;
switchTo(pc, &discard, next);
unreachable;
}
/// End the current **user** task: free its address space, then exit. Runs on the
/// dying task's kernel stack (in the shared kernel half, so it survives the CR3
/// switch to the kernel tables that must happen before we free the process's own
/// tables — we can't free the page tables we're standing on). The kernel stack
/// itself is leaked, as in `exit` (no reaper yet). Never returns.
pub fn exitUser() noreturn {
_ = sync.enter();
const pc = thisCpu();
const dying = pc.current;
const as = dying.aspace;
if (as != 0) {
const kroot = architecture.kernelPageTable();
architecture.loadPageTable(kroot); // off the process tables before freeing them
pc.loaded_aspace = kroot;
architecture.destroyAddressSpace(as);
}
dying.state = .free;
dying.aspace = 0;
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
next.state = .running;
pc.current = next;
var discard: usize = 0;
switchTo(pc, &discard, next);
unreachable;
}
/// Whether the running task is a user process (has its own address space).
pub fn currentIsUserProcess() bool {
return current().aspace != 0;
}
pub fn currentId() u32 {
return current().id;
}
/// The dense index of the core this task is currently running on (0 = BSP). Reads
/// per-CPU state, so a task calling it on different cores sees different values —
/// which is how a test can prove work is running in parallel. Returns 0 if the GS
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
/// reporter can call it unconditionally without a second fault.
pub fn currentCpuIndex() u32 {
if (architecture.cpuLocal() == 0) return 0;
return thisCpu().index;
}
/// Change the running task's priority (takes effect next time it's enqueued).
pub fn setPriority(p: Priority) void {
current().priority = p;
}
+80
View File
@@ -0,0 +1,80 @@
//! The big kernel lock (BKL) — the coarse mutual exclusion that lets more than one
//! CPU run kernel code safely.
//!
//! Until SMP, the kernel's mutual exclusion *was* the interrupt flag: a critical
//! section did `cli`, and since only one core existed, nothing else could touch
//! kernel state (the discipline in docs/scheduling.md). That invariant dies the
//! instant a second core runs kernel code — `cli` on one core does nothing to
//! another. So the kernel's shared state (the scheduler queues, IPC channels) is
//! guarded by a spinlock, and the lock is **always held with local interrupts
//! disabled**, so a core's own timer interrupt can't re-enter the kernel and
//! deadlock against the lock it already holds.
//!
//! This is deliberately *one coarse lock*, not many fine ones: it's philosophically
//! aligned with a tiny kernel and it keeps the single-core correctness model
//! (docs/scheduling.md) largely intact — one lock around kernel entry instead of
//! rethinking every critical section. It's the first-design choice seL4 makes and
//! docs/smp.md endorses; per-core run queues + fine-grained locking come later, if
//! contention ever bites. Because the kernel does little, the lock is held briefly.
//!
//! **The hand-off rule.** The lock is held *across* a context switch and released
//! by whichever task resumes, not by the one that switched away. A task that blocks
//! or yields calls `enter`, mutates the queues, `schedule()`s — switching to another
//! task *with the lock still held* — and only calls `leave` once it is eventually
//! resumed and its critical section runs to the end. So every call into `schedule()`
//! (and thus `switch_context`) happens with the lock held, and every task resumes
//! from a switch holding it. A freshly-spawned task has no `enter`/`leave` frame to
//! resume into, so `task_trampoline` releases the lock explicitly on its behalf via
//! `releaseForFreshTask` before running the task body.
const std = @import("std");
const architecture = @import("architecture");
/// 0 = free, 1 = held. A single global lock for the whole kernel.
var held = std.atomic.Value(u32).init(0);
/// Enter the kernel: disable interrupts on this core, then spin until we own the
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
/// Interrupts stay off for the whole critical section so this core's timer tick
/// can't try to re-acquire the lock we're holding.
pub fn enter() u64 {
const flags = architecture.saveInterrupts();
acquire();
return flags;
}
/// Release the lock and restore the interrupt flags `enter` returned (re-enabling
/// interrupts only if they were on beforehand). The normal exit for a critical
/// section reached from task context (`yield`, `sleep`, `wait`, `wake`, IPC).
pub fn leave(flags: u64) void {
release();
architecture.restoreInterrupts(flags);
}
/// Release the lock but leave interrupts as they are. The exit for a critical
/// section running inside an interrupt handler (the timer `tick`): the handler's
/// `iretq` is what restores the interrupted context's flags, so restoring them
/// here too would open a nested-interrupt window before the return. Release only.
pub fn leaveIsr() void {
release();
}
/// Release the lock on behalf of a freshly-spawned task. Such a task is switched to
/// (with the lock held) but has no `enter`/`leave` frame of its own to release
/// through — `task_trampoline` calls this before running the task body. Interrupts
/// are enabled separately by the trampoline. Exported for the assembly trampoline.
export fn releaseForFreshTask() callconv(.c) void {
release();
}
fn acquire() void {
// Test-and-test-and-set: try once, then spin read-only until the lock looks
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
while (held.swap(1, .acquire) != 0) {
while (held.load(.monotonic) != 0) architecture.cpuRelax();
}
}
fn release() void {
held.store(0, .release);
}
File diff suppressed because it is too large Load Diff