Re-organize the source tree as a monorepo mirroring the FHS
The source layout now mirrors the runtime filesystem hierarchy
(docs/danos-file-system-hierarchy-FSH.md): what lives under system/ in the
source is what a running danos represents under /system. Each service and
driver is a sub-project directory that is its own Zig module — cross-project
references go by module name, never by a path into another project's files.
Moves (all git mv, history preserved):
- src/ -> system/ (danos internals; the self-representation)
root.zig -> danos.zig (the kernel<->user contract module)
kernel/arch/ -> kernel/architecture/ (arch -> architecture)
device/ -> devices/ (what /system/devices reflects)
boot/ -> /boot (the loaders, top level)
- sbin/ -> split by role:
init, vfs -> system/services/<name>/<name>.zig
hpetd, busd -> system/drivers/<name>/<name>.zig
vfs-test -> system/services/vfs/vfs-test.zig (inside the vfs project)
- lib/ -> library/runtime/ (room for other libraries beside runtime)
The VFS wire protocol becomes its own module, system/services/vfs/protocol.zig
("vfs-protocol"): the vfs sub-project exposes its interface, and the runtime's
file layer imports it by name. First instance of the "protocol module" pattern
(docs/driver-model.md); usb/block will expose theirs the same way.
Also: fix a naming-standard violation in the protocol — Op -> Operation (and
req -> request, _pad -> _padding). Docs updated: /system/services added to the
FHS doc, a repository-layout section added to the docs index, and stale source
paths swept across comments and docs.
Runtime boot paths are unchanged (the bootloader still loads /sbin/init);
aligning the runtime filesystem to the FHS is a separate follow-up. Suite 35/35
plus host tests green.
This commit is contained in:
@@ -0,0 +1,413 @@
|
||||
//! Local APIC and its timer — the source of device interrupts.
|
||||
//!
|
||||
//! Modern x86 routes interrupts through the per-CPU Local APIC (the legacy 8259
|
||||
//! PIC is remapped out of the way and masked). The LAPIC also has a built-in
|
||||
//! timer, which is the simplest device interrupt to bring up: it needs no
|
||||
//! external routing, just a vector and a count. We use it as danos's heartbeat.
|
||||
//!
|
||||
//! The LAPIC is memory-mapped (default physical 0xFEE00000, inside our identity
|
||||
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
|
||||
//! the LAPIC won't deliver the next one.
|
||||
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// The ACPI PM timer, as a calibration reference: an I/O port or MMIO counter.
|
||||
pub const PmTimer = struct { mmio: bool, address: u64, is_32bit: bool };
|
||||
|
||||
// Platform facts from discovery (set by `configure` before bring-up). Defaults are
|
||||
// the legacy-safe assumptions so the code still works if discovery never ran.
|
||||
var configuration_pic_present: bool = true;
|
||||
var configuration_hpet_base: u64 = 0; // 0 = no HPET discovered
|
||||
var configuration_pm_timer: ?PmTimer = null;
|
||||
/// Which reference the last calibration used, for logging.
|
||||
var cal_source: []const u8 = "none";
|
||||
|
||||
/// Hand the LAPIC bring-up the discovered platform facts. Call before `init`.
|
||||
pub fn configure(pic_present: bool, hpet_base: u64, pm_timer: ?PmTimer) void {
|
||||
configuration_pic_present = pic_present;
|
||||
configuration_hpet_base = hpet_base;
|
||||
configuration_pm_timer = pm_timer;
|
||||
}
|
||||
|
||||
/// The calibration reference the timer was measured against ("cpuid"/"hpet"/…).
|
||||
pub fn calibrationSource() []const u8 {
|
||||
return cal_source;
|
||||
}
|
||||
|
||||
/// IDT vector the timer fires on (in the device range, >= 32).
|
||||
pub const timer_vector = 32;
|
||||
/// Spurious-interrupt vector. Low nibble 0xF by convention; also in our gate
|
||||
/// range so a stray spurious interrupt lands on a valid (no-op) handler.
|
||||
const spurious_vector = 47;
|
||||
|
||||
// LAPIC register offsets.
|
||||
const register_spurious = 0x0F0;
|
||||
const register_eoi = 0x0B0;
|
||||
const register_id = 0x020; // this core's LAPIC id, in bits 24-31
|
||||
const register_icr_low = 0x300; // interrupt command register, low dword (writing it sends)
|
||||
const register_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31)
|
||||
const register_lvt_timer = 0x320;
|
||||
const register_timer_initial = 0x380;
|
||||
const register_timer_current = 0x390;
|
||||
const register_timer_divide = 0x3E0;
|
||||
|
||||
const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight
|
||||
|
||||
const lvt_masked = 1 << 16;
|
||||
const lvt_periodic = 1 << 17;
|
||||
const timer_divide_16 = 0x3;
|
||||
|
||||
const ia32_apic_base_msr = 0x1B;
|
||||
|
||||
/// LAPIC MMIO base. A runtime var (not a constant) both because we read it from
|
||||
/// the MSR and so register writes compile to normal stores rather than a
|
||||
/// `mov moffs`, which the self-hosted backend can't encode.
|
||||
var base: usize = 0xFEE00000;
|
||||
|
||||
var tick_count: u64 = 0;
|
||||
|
||||
/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate).
|
||||
/// At divide-by-16, this is the effective counting rate.
|
||||
var ticks_per_ms: u32 = 0;
|
||||
/// The periodic-interrupt frequency the timer is armed at, once initTimer runs.
|
||||
var timer_hz: u32 = 0;
|
||||
|
||||
/// TSC (Time Stamp Counter) calibration: cycles per second, and the count at boot.
|
||||
/// The TSC is a per-core cycle counter, giving a ~nanosecond high-resolution
|
||||
/// monotonic clock — far finer than the millisecond timer tick.
|
||||
var tsc_hz: u64 = 0;
|
||||
var tsc_base: u64 = 0;
|
||||
|
||||
/// Read the 64-bit Time Stamp Counter.
|
||||
fn rdtsc() u64 {
|
||||
var low: u32 = undefined;
|
||||
var high: u32 = undefined;
|
||||
asm volatile ("rdtsc"
|
||||
: [low] "={eax}" (low),
|
||||
[high] "={edx}" (high),
|
||||
);
|
||||
return (@as(u64, high) << 32) | low;
|
||||
}
|
||||
|
||||
fn read(register: u32) u32 {
|
||||
return @as(*volatile u32, @ptrFromInt(base + register)).*;
|
||||
}
|
||||
fn write(register: u32, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(base + register)).* = value;
|
||||
}
|
||||
|
||||
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
|
||||
/// vectors) and mask every line, so it can't deliver interrupts behind the APIC.
|
||||
fn remapAndMaskPic() void {
|
||||
io.outb(0x20, 0x11); // start init (cascade mode)
|
||||
io.outb(0xA0, 0x11);
|
||||
io.outb(0x21, 0x20); // master offset 0x20
|
||||
io.outb(0xA1, 0x28); // slave offset 0x28
|
||||
io.outb(0x21, 0x04); // tell master about slave on IRQ2
|
||||
io.outb(0xA1, 0x02);
|
||||
io.outb(0x21, 0x01); // 8086 mode
|
||||
io.outb(0xA1, 0x01);
|
||||
io.outb(0x21, 0xFF); // mask all
|
||||
io.outb(0xA1, 0xFF);
|
||||
}
|
||||
|
||||
/// Enable the Local APIC: mask the PIC (only if one is present — a legacy-free
|
||||
/// UEFI Class 3 machine may have none), set the global-enable MSR bit, and
|
||||
/// software-enable the APIC via its spurious-vector register.
|
||||
pub fn init() void {
|
||||
if (configuration_pic_present) remapAndMaskPic();
|
||||
|
||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||
// Reach the LAPIC through the physmap (paging.init maps its page there).
|
||||
base = @intCast(danos.physicalToVirtual(msr & 0xFFFFF000)); // physical base is bits 12+
|
||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||
|
||||
write(register_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
||||
}
|
||||
|
||||
/// Software-enable *this* core's Local APIC — the application-processor counterpart
|
||||
/// of `init`, minus the one-time PIC remap (the BSP already masked it) and minus
|
||||
/// calibration (the timer rate is a shared hardware constant, measured once). Each
|
||||
/// core has its own LAPIC at the same MMIO address, so no per-core base is needed.
|
||||
pub fn initSecondary() void {
|
||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||
write(register_spurious, 0x100 | spurious_vector); // software enable
|
||||
}
|
||||
|
||||
// --- application-processor wakeup (INIT–SIPI–SIPI) --------------------------
|
||||
|
||||
/// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of
|
||||
/// the wake sequence. Blocks until the LAPIC reports the IPI was delivered.
|
||||
pub fn sendInit(apic_id: u32) void {
|
||||
write(register_icr_high, apic_id << 24);
|
||||
write(register_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered
|
||||
waitIcrIdle();
|
||||
}
|
||||
|
||||
/// Send a STARTUP IPI (SIPI) telling the target core to begin executing at physical
|
||||
/// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is
|
||||
/// sent twice after the INIT; both calls block until delivery completes.
|
||||
pub fn sendStartup(apic_id: u32, vector: u8) void {
|
||||
write(register_icr_high, apic_id << 24);
|
||||
write(register_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector
|
||||
waitIcrIdle();
|
||||
}
|
||||
|
||||
fn waitIcrIdle() void {
|
||||
while (read(register_icr_low) & icr_delivery_pending != 0) {}
|
||||
}
|
||||
|
||||
/// The calibration window: we time everything against a 10 ms reference interval.
|
||||
const calib_ms = 10;
|
||||
|
||||
/// Measure the LAPIC timer's and the TSC's rates. The PIT (legacy 8254) can be
|
||||
/// absent on UEFI Class 3 firmware — and polling it would hang — so we pick a
|
||||
/// reference clock in order of preference: the CPU's own TSC frequency (CPUID leaf
|
||||
/// 0x15, no external timer needed), then the discovered HPET, then the ACPI PM
|
||||
/// timer, and only the PIT as a last resort. Each path yields the same two rates.
|
||||
pub fn calibrate() void {
|
||||
var done = false;
|
||||
|
||||
// 1. CPUID leaf 0x15 gives the TSC frequency directly — measure the LAPIC
|
||||
// against the TSC itself, needing no external timer at all.
|
||||
if (cpuidTscHz()) |hz| {
|
||||
measure(hz, ~@as(u64, 0), rdtsc);
|
||||
tsc_hz = hz; // keep the exact enumerated value
|
||||
cal_source = "cpuid";
|
||||
done = true;
|
||||
}
|
||||
|
||||
// 2. The discovered HPET.
|
||||
if (!done and configuration_hpet_base != 0) {
|
||||
if (hpetHz()) |hpet_hz| {
|
||||
measure(hpet_hz, hpetMask(), readHpet);
|
||||
cal_source = "hpet";
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 3. The ACPI PM timer (fixed 3.579545 MHz).
|
||||
if (!done) {
|
||||
if (configuration_pm_timer) |pt| {
|
||||
measure(3_579_545, if (pt.is_32bit) 0xFFFF_FFFF else 0xFF_FFFF, readPmTimer);
|
||||
cal_source = "pm-timer";
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 4. The legacy PIT, last resort.
|
||||
if (!done) {
|
||||
calibratePit();
|
||||
cal_source = "pit";
|
||||
}
|
||||
|
||||
// A bad measurement (no reference actually ticked) leaves nonsense; fall back.
|
||||
if (ticks_per_ms == 0 or tsc_hz == 0) {
|
||||
calibratePit();
|
||||
cal_source = "pit";
|
||||
}
|
||||
|
||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||
}
|
||||
|
||||
/// Run the LAPIC timer one-shot from its maximum count while a monotonic reference
|
||||
/// clock (frequency `ref_hz`, counter width `ref_mask`) counts out `calib_ms`, and
|
||||
/// snapshot the TSC across the same window. Yields `ticks_per_ms` and `tsc_hz`.
|
||||
fn measure(ref_hz: u64, ref_mask: u64, refNow: *const fn () u64) void {
|
||||
const calib_ticks = ref_hz / (1000 / calib_ms); // reference ticks in calib_ms
|
||||
|
||||
write(register_timer_divide, timer_divide_16);
|
||||
write(register_lvt_timer, lvt_masked);
|
||||
write(register_timer_initial, 0xFFFFFFFF);
|
||||
|
||||
const ref0 = refNow();
|
||||
const tsc0 = rdtsc();
|
||||
while (((refNow() -% ref0) & ref_mask) < calib_ticks) {}
|
||||
const tsc1 = rdtsc();
|
||||
|
||||
const elapsed = 0xFFFFFFFF - read(register_timer_current);
|
||||
write(register_timer_initial, 0);
|
||||
|
||||
ticks_per_ms = elapsed / calib_ms;
|
||||
tsc_hz = (tsc1 -% tsc0) * (1000 / calib_ms);
|
||||
}
|
||||
|
||||
/// The PIT fallback (legacy 8254 channel 2, polled). Only reached when no better
|
||||
/// reference exists — on a legacy-free machine this path isn't taken.
|
||||
fn calibratePit() void {
|
||||
const pit_hz = 1_193_182;
|
||||
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
|
||||
|
||||
write(register_timer_divide, timer_divide_16);
|
||||
write(register_lvt_timer, lvt_masked);
|
||||
write(register_timer_initial, 0xFFFFFFFF);
|
||||
|
||||
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
|
||||
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
|
||||
io.outb(0x42, @truncate(pit_count));
|
||||
io.outb(0x42, @truncate(pit_count >> 8));
|
||||
|
||||
const tsc_start = rdtsc();
|
||||
io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start
|
||||
var guard: u64 = 0;
|
||||
while (io.inb(0x61) & 0x20 == 0 and guard < 100_000_000) : (guard += 1) {} // bounded
|
||||
const tsc_end = rdtsc();
|
||||
|
||||
const elapsed = 0xFFFFFFFF - read(register_timer_current);
|
||||
write(register_timer_initial, 0);
|
||||
|
||||
ticks_per_ms = elapsed / calib_ms;
|
||||
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms);
|
||||
}
|
||||
|
||||
// --- reference clocks ------------------------------------------------------
|
||||
|
||||
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
||||
/// null if the CPU doesn't enumerate it (common under QEMU).
|
||||
fn cpuidTscHz() ?u64 {
|
||||
if (cpuid(0).eax < 0x15) return null;
|
||||
const r = cpuid(0x15);
|
||||
if (r.eax == 0 or r.ebx == 0 or r.ecx == 0) return null; // ratio/crystal not given
|
||||
return @as(u64, r.ecx) * r.ebx / r.eax;
|
||||
}
|
||||
|
||||
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
||||
|
||||
fn cpuid(leaf: u32) CpuidRegs {
|
||||
var a: u32 = undefined;
|
||||
var b: u32 = undefined;
|
||||
var c: u32 = undefined;
|
||||
var d: u32 = undefined;
|
||||
asm volatile ("cpuid"
|
||||
: [a] "={eax}" (a),
|
||||
[b] "={ebx}" (b),
|
||||
[c] "={ecx}" (c),
|
||||
[d] "={edx}" (d),
|
||||
: [leaf] "{eax}" (leaf),
|
||||
[sub] "{ecx}" (@as(u32, 0)),
|
||||
);
|
||||
return .{ .eax = a, .ebx = b, .ecx = c, .edx = d };
|
||||
}
|
||||
|
||||
// HPET registers: capabilities at +0x00 (period in the high dword, in fs; bit 13 =
|
||||
// 64-bit-counter capable), general configuration at +0x10, main counter at +0xF0.
|
||||
fn hpetRead64(off: usize) u64 {
|
||||
return @as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).*;
|
||||
}
|
||||
fn hpetWrite64(off: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(configuration_hpet_base + off)).* = value;
|
||||
}
|
||||
|
||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||
/// Maps the HPET into the physmap and switches configuration_hpet_base to that virtual
|
||||
/// address, so the register accessors reach it without the identity map.
|
||||
fn hpetHz() ?u64 {
|
||||
configuration_hpet_base = paging.mapMmio(configuration_hpet_base, 0x400, true);
|
||||
const caps = hpetRead64(0x00);
|
||||
const period_fs = caps >> 32; // femtoseconds per tick
|
||||
if (period_fs == 0) return null;
|
||||
hpetWrite64(0x10, hpetRead64(0x10) | 1); // ENABLE_CNF: start the main counter
|
||||
return 1_000_000_000_000_000 / period_fs; // 1e15 fs/s ÷ fs/tick
|
||||
}
|
||||
|
||||
/// The HPET counter width mask (64- or 32-bit, per caps bit 13).
|
||||
fn hpetMask() u64 {
|
||||
return if (hpetRead64(0x00) & (1 << 13) != 0) ~@as(u64, 0) else 0xFFFF_FFFF;
|
||||
}
|
||||
|
||||
fn readHpet() u64 {
|
||||
return hpetRead64(0xF0);
|
||||
}
|
||||
|
||||
fn readPmTimer() u64 {
|
||||
const pt = configuration_pm_timer.?;
|
||||
// MMIO PM timer via the physmap (mapMmio is idempotent); the common case is
|
||||
// a legacy I/O port.
|
||||
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(paging.mapMmio(pt.address, 4, false))).*;
|
||||
return io.inl(@intCast(pt.address));
|
||||
}
|
||||
|
||||
/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires
|
||||
/// calibrate() to have run.
|
||||
pub fn initTimer(hz: u32) void {
|
||||
timer_hz = hz;
|
||||
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
|
||||
write(register_timer_divide, timer_divide_16);
|
||||
write(register_lvt_timer, timer_vector | lvt_periodic);
|
||||
write(register_timer_initial, @intCast(count));
|
||||
}
|
||||
|
||||
/// Configured periodic-interrupt frequency (Hz).
|
||||
pub fn frequencyHz() u32 {
|
||||
return timer_hz;
|
||||
}
|
||||
|
||||
/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks.
|
||||
pub fn lapicHz() u64 {
|
||||
return @as(u64, ticks_per_ms) * 1000;
|
||||
}
|
||||
|
||||
/// Measured TSC frequency (Hz).
|
||||
pub fn tscHz() u64 {
|
||||
return tsc_hz;
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
||||
// the scheduler uses for sleep deadlines.
|
||||
|
||||
pub fn nanos() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
||||
}
|
||||
|
||||
pub fn micros() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
||||
}
|
||||
|
||||
pub fn millis() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
||||
}
|
||||
|
||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||
pub fn eoi() void {
|
||||
write(register_eoi, 0);
|
||||
}
|
||||
|
||||
/// Optional callback run each tick (the scheduler registers it for preemption).
|
||||
var on_tick: ?*const fn () void = null;
|
||||
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
on_tick = hook;
|
||||
}
|
||||
|
||||
/// The timer interrupt handler: advance the monotonic tick count, then run the
|
||||
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
|
||||
/// the dispatcher before we get here, so a task switch here doesn't stall it.
|
||||
pub fn timerTick() void {
|
||||
// Acknowledge before the tick hook: `on_tick` is the scheduler, which may switch
|
||||
// tasks and not return promptly, and the LAPIC mustn't wait on it to deliver the
|
||||
// next interrupt. (Each device handler now owns its own EOI — see
|
||||
// `idt.interruptDispatch` — because a *routed* interrupt must be masked at the
|
||||
// I/O APIC before it is acknowledged, an ordering the dispatcher can't impose.)
|
||||
eoi();
|
||||
tick_count +%= 1;
|
||||
if (on_tick) |hook| hook();
|
||||
}
|
||||
|
||||
/// This core's Local APIC id — the interrupt destination for `routeGsi`.
|
||||
pub fn localId() u8 {
|
||||
return @truncate(read(register_id) >> 24);
|
||||
}
|
||||
|
||||
/// Number of timer ticks so far. Volatile load: the count is bumped
|
||||
/// asynchronously by the interrupt handler, so callers must re-read memory.
|
||||
pub fn ticks() u64 {
|
||||
return @as(*const volatile u64, &tick_count).*;
|
||||
}
|
||||
@@ -0,0 +1,578 @@
|
||||
//! x86_64 CPU operations. This is the "architecture" module: the generic kernel imports
|
||||
//! it as `@import("architecture")` and never names x86_64 directly, so a second
|
||||
//! architecture is added by pointing that module at a different directory in
|
||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
||||
|
||||
const danos = @import("danos");
|
||||
const parameters = @import("parameters");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const paging = @import("paging.zig");
|
||||
const serial = @import("serial.zig");
|
||||
const apic = @import("apic.zig");
|
||||
const ioapic = @import("ioapic.zig");
|
||||
const io = @import("io.zig");
|
||||
const smp = @import("smp.zig");
|
||||
const pcpu = @import("per-cpu.zig");
|
||||
|
||||
/// The saved register/trap frame passed to a fault handler.
|
||||
pub const CpuState = idt.CpuState;
|
||||
|
||||
// --- trap-frame accessors ---------------------------------------------------
|
||||
// The frame's fields are x86_64 registers; the generic kernel reads it through
|
||||
// these accessors so it never names one.
|
||||
|
||||
/// The interrupted/faulting instruction address (RIP here; ELR_EL1 on aarch64,
|
||||
/// sepc on riscv64).
|
||||
pub fn instructionPointer(state: *const CpuState) u64 {
|
||||
return state.rip;
|
||||
}
|
||||
|
||||
/// The interrupted stack pointer (RSP here).
|
||||
pub fn stackPointer(state: *const CpuState) u64 {
|
||||
return state.rsp;
|
||||
}
|
||||
|
||||
/// Whether the trap came from user mode (CPL 3 here; EL0 on aarch64, U-mode on
|
||||
/// riscv64).
|
||||
pub fn fromUser(state: *const CpuState) bool {
|
||||
return state.cs & 3 == 3;
|
||||
}
|
||||
|
||||
/// The faulting virtual address, if this trap is a page fault (CR2 here;
|
||||
/// FAR_EL1 on aarch64, stval on riscv64). Null for any other exception.
|
||||
pub fn faultAddress(state: *const CpuState) ?u64 {
|
||||
if (state.vector != 14) return null;
|
||||
return asm volatile ("mov %%cr2, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
// --- system_call ABI ------------------------------------------------------------
|
||||
// The System V-style register convention (number in rax, arguments in
|
||||
// rdi/rsi/rdx/r10/r8/r9, result in rax), exposed positionally so the generic
|
||||
// dispatcher never names a register.
|
||||
|
||||
/// The system_call number the user program passed.
|
||||
pub fn systemCallNumber(state: *const CpuState) u64 {
|
||||
return state.rax;
|
||||
}
|
||||
|
||||
/// Positional system_call argument `n`.
|
||||
pub fn systemCallArg(state: *const CpuState, n: u8) u64 {
|
||||
return switch (n) {
|
||||
0 => state.rdi,
|
||||
1 => state.rsi,
|
||||
2 => state.rdx,
|
||||
3 => state.r10,
|
||||
4 => state.r8,
|
||||
5 => state.r9,
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Write the system_call's return value into the frame — the entry paths restore
|
||||
/// user registers from it.
|
||||
pub fn setSystemCallResult(state: *CpuState, value: u64) void {
|
||||
state.rax = value;
|
||||
}
|
||||
|
||||
/// Write a *second* system_call return value (rdx here — restored by both the
|
||||
/// system_call/sysret and int-0x80 entry paths; unlike rcx/r11 it is not consumed by
|
||||
/// sysretq). Used by IPC_ReplyWait to hand back the sender's badge alongside the
|
||||
/// message length in rax.
|
||||
pub fn setSystemCallResult2(state: *CpuState, value: u64) void {
|
||||
state.rdx = value;
|
||||
}
|
||||
|
||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
||||
/// so it can be the very first thing called.
|
||||
pub fn serialInit() void {
|
||||
serial.init();
|
||||
}
|
||||
|
||||
/// Write bytes to the serial port.
|
||||
pub fn serialWrite(bytes: []const u8) void {
|
||||
serial.write(bytes);
|
||||
}
|
||||
|
||||
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||
/// displays. The last-resort progress signal when there's no text output at all.
|
||||
/// Writing 0x80 is universally safe (it's the legacy I/O-delay port).
|
||||
pub fn checkpoint(code: u8) void {
|
||||
io.outb(0x80, code);
|
||||
}
|
||||
|
||||
/// Whether a Bochs/QEMU-style debug console is on port 0xE9 (it returns 0xE9 when
|
||||
/// read). On real hardware the port reads back 0xFF, so this stays false — a safe
|
||||
/// probe before we write to it.
|
||||
pub fn debugconPresent() bool {
|
||||
return io.inb(0xE9) == 0xE9;
|
||||
}
|
||||
|
||||
/// Output sink: write bytes to the 0xE9 debug console (see `debugconPresent`).
|
||||
pub fn debugconWrite(bytes: []const u8) void {
|
||||
for (bytes) |b| io.outb(0xE9, b);
|
||||
}
|
||||
|
||||
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
|
||||
/// stack for double faults), then the IDT with exception handlers. After this a
|
||||
/// CPU fault is reported instead of triple-faulting. Install the fault handler
|
||||
/// (setFaultHandler) first so early faults are caught.
|
||||
pub fn init() void {
|
||||
gdt.init();
|
||||
tss.init();
|
||||
idt.init();
|
||||
pcpu.initSystemCall();
|
||||
}
|
||||
|
||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||
/// kernel's segment layout). Call once the frame allocator is up.
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
|
||||
paging.init(allocFrame, freeFrame, boot_information);
|
||||
}
|
||||
|
||||
/// Create a new address space (returns the physical address of its root table —
|
||||
/// the PML4 here — or null). Shares the kernel's higher half; the user (low)
|
||||
/// half starts empty.
|
||||
pub fn createAddressSpace() ?u64 {
|
||||
return paging.createAddressSpace();
|
||||
}
|
||||
|
||||
/// Free an address space and everything mapped in its user half. Caller must not
|
||||
/// be running on it.
|
||||
pub fn destroyAddressSpace(root: u64) void {
|
||||
paging.destroyAddressSpace(root);
|
||||
}
|
||||
|
||||
/// Map a user page into address space `root` (W^X is the caller's contract).
|
||||
pub fn mapUserPageInto(root: u64, virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||
paging.mapUserInto(root, virtual, physical, writable, executable);
|
||||
}
|
||||
|
||||
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
|
||||
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
|
||||
pub fn mapUserDeviceInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
paging.mapUserDeviceInto(root, virtual, physical, len);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
||||
paging.map(virtual, physical, writable);
|
||||
}
|
||||
|
||||
/// Map a device MMIO range and return the virtual address to reach it at. This
|
||||
/// is the device layer's `Hal.mapMmio` — it hands back a physmap pointer and
|
||||
/// never exposes how the mapping is placed.
|
||||
pub fn mapMmio(physical: u64, len: u64, writable: bool) u64 {
|
||||
return paging.mapMmio(physical, len, writable);
|
||||
}
|
||||
|
||||
/// The kernel's page-table root (physical), shared into every address space.
|
||||
pub fn kernelPageTable() u64 {
|
||||
return paging.kernelPml4();
|
||||
}
|
||||
|
||||
/// Switch the active address space (load CR3 with a physical root table).
|
||||
pub fn loadPageTable(root: u64) void {
|
||||
paging.loadCr3(root);
|
||||
}
|
||||
|
||||
/// The physical root of the currently active page tables (CR3 here; TTBR0/satp
|
||||
/// elsewhere).
|
||||
pub fn activePageTable() u64 {
|
||||
return asm volatile ("mov %%cr3, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
|
||||
/// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the
|
||||
/// per-CPU `kernel_rsp` (for the system_call stub, which switches by hand). Updated
|
||||
/// by the scheduler when it switches to a user task.
|
||||
pub fn setKernelStack(cpu: usize, top: usize) void {
|
||||
tss.rsp0Ptr(cpu).* = top;
|
||||
pcpu.setKernelRsp(cpu, top);
|
||||
}
|
||||
|
||||
/// Remove a kernel mapping.
|
||||
pub fn unmapPage(virtual: u64) void {
|
||||
paging.unmap(virtual);
|
||||
}
|
||||
|
||||
/// Remove a page mapping from address space `root` (for munmap of user pages).
|
||||
/// Clears the leaf entry only; freeing the underlying frame is the caller's job.
|
||||
pub fn unmapUserPageInto(root: u64, virtual: u64) void {
|
||||
paging.unmapInto(root, virtual);
|
||||
}
|
||||
|
||||
/// Resolve `virtual` to its physical address in the address space rooted at `root`
|
||||
/// (any address space, not just the live one), or null if unmapped. Used to find
|
||||
/// the frame behind a user page for munmap, and for cross-address-space copies.
|
||||
pub fn translate(root: u64, virtual: u64) ?u64 {
|
||||
return paging.translateIn(root, virtual);
|
||||
}
|
||||
|
||||
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
||||
/// W^X: code read-only + executable, data writable + no-execute.
|
||||
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||
paging.mapUser(virtual, physical, writable, executable);
|
||||
}
|
||||
|
||||
// --- ring 3 entry/exit -----------------------------------------------------
|
||||
|
||||
/// Drop to ring 3 at `rip` on `rsp` (defined in isr.s). Saves the kernel context,
|
||||
/// publishes the kernel stack pointer through `rsp0_slot` (this core's TSS.rsp0,
|
||||
/// so ring-3 interrupts land on a good stack), builds an iretq frame with the
|
||||
/// user selectors, and iretq's. "Returns" only when the user program triggers
|
||||
/// the exit path (user_exit_to_kernel).
|
||||
extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void;
|
||||
|
||||
/// Abandon the in-flight ring-3 trap context and resume the kernel as if
|
||||
/// `enter_user` had returned (defined in isr.s). Called by the exit system_call.
|
||||
extern fn user_exit_to_kernel() callconv(.c) noreturn;
|
||||
|
||||
/// Run user code at `entry` with stack `stack_top` on this core (`cpu` = the
|
||||
/// caller's CPU index; the architecture layer can't ask the scheduler). Returns after the
|
||||
/// user program exits via system_call. Interrupts are disabled on return (the exit
|
||||
/// arrives through an interrupt gate) — the caller re-enables.
|
||||
pub fn enterUser(cpu: usize, entry: u64, stack_top: u64) void {
|
||||
enter_user(entry, stack_top, tss.rsp0Ptr(cpu));
|
||||
}
|
||||
|
||||
/// Never returns to the user program: unwind to the kernel context that called
|
||||
/// `enterUser`. For the exit system_call's handler.
|
||||
pub fn userExit() noreturn {
|
||||
user_exit_to_kernel();
|
||||
}
|
||||
|
||||
/// Register the handler for the user system_call gate (int 0x80, vector 128). The
|
||||
/// handler may write the trap frame (see `setSystemCallResult`).
|
||||
pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void {
|
||||
idt.setSystemCallHandler(handler);
|
||||
}
|
||||
|
||||
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
|
||||
/// core calls this once, after its GDT is in place (a GS *selector* reload would
|
||||
/// clobber the base). See percpu.zig for the swapgs discipline.
|
||||
pub fn setCpuLocal(cpu: usize, ptr: usize) void {
|
||||
pcpu.setLocal(cpu, ptr);
|
||||
}
|
||||
|
||||
/// This core's scheduler pointer (via the GS base) — a per-core register, so each
|
||||
/// core sees its own without locking. Valid in any ring-0 context.
|
||||
pub fn cpuLocal() usize {
|
||||
return pcpu.scheduler();
|
||||
}
|
||||
|
||||
// --- SMP: application-processor bring-up ----------------------------------
|
||||
|
||||
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
|
||||
/// The frame stays inert (zeroed, non-executable) between wakes and is armed only
|
||||
/// while a core is climbing — so a core can be (re)woken at any time (retry, or a
|
||||
/// future power manager) without leaving an executable page resident. See smp.zig.
|
||||
pub fn setTrampolinePage(physical: u64) void {
|
||||
smp.setTrampolinePage(physical);
|
||||
}
|
||||
|
||||
/// Wake the core with hardware id `hw_id` (its Local APIC id here; MPIDR on
|
||||
/// aarch64, hart id on riscv64) as dense CPU `index`, giving it `stack_top` and
|
||||
/// its per-CPU pointer `percpu`; it adopts the kernel page tables. Returns false
|
||||
/// if it doesn't come online within the timeout. Blocks until the core reports in.
|
||||
pub fn startSecondary(hw_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
|
||||
// The AP adopts the kernel page tables explicitly — never the caller's live
|
||||
// CR3, which a future re-wake from a core running a process would make a
|
||||
// process address space.
|
||||
return smp.startAp(hw_id, stack_top, percpu, index, paging.kernelPml4());
|
||||
}
|
||||
|
||||
/// Register the generic entry a woken AP jumps to once its architecture state is up (its own
|
||||
/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here.
|
||||
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
||||
smp.setSecondaryEntry(entry);
|
||||
}
|
||||
|
||||
/// Bytes the kernel should allocate for a secondary core's dedicated fault stack
|
||||
/// (the IST double-fault stack here), and where to record its top before waking
|
||||
/// the core. The stack is heap-allocated per online core (the boot CPU's is
|
||||
/// static — it's needed before the allocator exists). See tss.zig.
|
||||
pub const fault_stack_size = tss.ist_stack_size;
|
||||
pub fn setFaultStack(cpu: usize, top: usize) void {
|
||||
tss.setApIstStack(cpu, top);
|
||||
}
|
||||
|
||||
/// Test hook: force the next `n` AP wake attempts to fail, so the retry path can be
|
||||
/// exercised deterministically (see the smp-retry test). No effect when `n` is 0.
|
||||
pub fn testFailNextWakes(n: u32) void {
|
||||
smp.testFailNextWakes(n);
|
||||
}
|
||||
|
||||
/// The reserved AP-trampoline frame (0 if none). For tests that check it's inert.
|
||||
pub fn trampolinePage() u64 {
|
||||
return smp.trampolinePage();
|
||||
}
|
||||
|
||||
/// Whether the page at `virtual` is currently mapped executable (present, NX clear).
|
||||
pub fn pageExecutable(virtual: u64) bool {
|
||||
return paging.isExecutable(virtual);
|
||||
}
|
||||
|
||||
/// Kernel tick rate (the scheduler's time quantum), from configuration.
|
||||
pub const timer_hz = parameters.timer_hz;
|
||||
|
||||
/// The ACPI PM timer, as a calibration reference (re-exported for the configuration).
|
||||
pub const PmTimer = apic.PmTimer;
|
||||
/// A MADT interrupt-source override (re-exported for the configuration).
|
||||
pub const IsoEntry = ioapic.IsoEntry;
|
||||
|
||||
/// Discovered platform facts the architecture layer needs so it makes no legacy
|
||||
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
||||
pub const PlatformConfiguration = struct {
|
||||
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
||||
pic_present: bool = true,
|
||||
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
||||
hpet_base: u64 = 0,
|
||||
/// The ACPI PM timer, another calibration reference.
|
||||
pm_timer: ?PmTimer = null,
|
||||
/// I/O APIC MMIO base + its first global system interrupt (0 = none).
|
||||
ioapic_base: u64 = 0,
|
||||
ioapic_gsi_base: u32 = 0,
|
||||
/// MADT ISA-IRQ overrides, for I/O APIC routing.
|
||||
overrides: []const IsoEntry = &.{},
|
||||
};
|
||||
|
||||
/// Apply the discovered platform configuration. Must run before `startTimer` (the timer
|
||||
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
||||
/// Maps + masks the I/O APIC immediately.
|
||||
pub fn configurePlatform(configuration: PlatformConfiguration) void {
|
||||
apic.configure(configuration.pic_present, configuration.hpet_base, configuration.pm_timer);
|
||||
ioapic.configure(configuration.ioapic_base, configuration.ioapic_gsi_base, configuration.overrides);
|
||||
ioapic.init();
|
||||
}
|
||||
|
||||
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
||||
pub fn serialReconfigure(is_mmio: bool, address: u64) void {
|
||||
serial.reconfigure(is_mmio, address);
|
||||
}
|
||||
|
||||
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
||||
pub fn timerCalibrationSource() []const u8 {
|
||||
return apic.calibrationSource();
|
||||
}
|
||||
|
||||
/// External-interrupt-router diagnostics, for boot logging / verification (the
|
||||
/// I/O APIC's redirection entries here; a GIC distributor or PLIC elsewhere).
|
||||
pub fn irqRouteCount() u32 {
|
||||
return ioapic.entryCount();
|
||||
}
|
||||
pub fn irqRouteRaw(n: u32) u32 {
|
||||
return ioapic.entryLow(n);
|
||||
}
|
||||
|
||||
// --- device-IRQ plumbing, for system/kernel/irq.zig -----------------------------
|
||||
//
|
||||
// The generic IRQ layer speaks GSIs and vectors; everything below hides the fact
|
||||
// that on x86_64 those mean "I/O APIC redirection entry" and "IDT gate". The
|
||||
// vector window is bounded by the stubs isr.s actually emits: `gate_count` = 48,
|
||||
// vector 32 is the LAPIC timer and 47 is the spurious vector, leaving 33..46.
|
||||
|
||||
pub const irq_vector_base: u8 = 33;
|
||||
pub const irq_vector_count: u8 = 14; // 33..46 inclusive
|
||||
|
||||
/// True if `gsi` is one this machine's interrupt router can deliver.
|
||||
pub fn irqOwnsGsi(gsi: u32) bool {
|
||||
return ioapic.ownsGsi(gsi);
|
||||
}
|
||||
|
||||
/// Install `handler` on `vector` (an absolute IDT gate index).
|
||||
pub fn irqSetHandler(vector: u8, handler: *const fn () void) void {
|
||||
idt.setHandler(vector, handler);
|
||||
}
|
||||
|
||||
/// Route `gsi` to `vector` on *this* core, masked. Unmask with `irqUnmask` once bound.
|
||||
pub fn irqRoute(gsi: u32, vector: u8, level: bool, active_low: bool) void {
|
||||
ioapic.routeGsi(gsi, vector, apic.localId(), level, active_low);
|
||||
}
|
||||
|
||||
pub fn irqMask(gsi: u32) void {
|
||||
ioapic.maskGsi(gsi);
|
||||
}
|
||||
pub fn irqUnmask(gsi: u32) void {
|
||||
ioapic.unmaskGsi(gsi);
|
||||
}
|
||||
|
||||
/// Acknowledge the interrupt currently in service on this core's LAPIC.
|
||||
pub fn irqEoi() void {
|
||||
apic.eoi();
|
||||
}
|
||||
|
||||
/// Enable the Local APIC, calibrate its timer against the best available reference
|
||||
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
||||
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
||||
/// unmasked with enableInterrupts() to be delivered. Run `configurePlatform` first.
|
||||
pub fn startTimer() void {
|
||||
apic.init();
|
||||
apic.calibrate();
|
||||
idt.setHandler(apic.timer_vector, apic.timerTick);
|
||||
apic.initTimer(timer_hz);
|
||||
}
|
||||
|
||||
/// Number of timer ticks since startTimer().
|
||||
pub fn ticks() u64 {
|
||||
return apic.ticks();
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock (from the TSC), one function per resolution.
|
||||
pub fn nanos() u64 {
|
||||
return apic.nanos();
|
||||
}
|
||||
pub fn micros() u64 {
|
||||
return apic.micros();
|
||||
}
|
||||
pub fn millis() u64 {
|
||||
return apic.millis();
|
||||
}
|
||||
|
||||
/// Measured frequency of the tick timer's input clock (the LAPIC timer here), in
|
||||
/// Hz, from calibration.
|
||||
pub fn timerClockHz() u64 {
|
||||
return apic.lapicHz();
|
||||
}
|
||||
|
||||
/// Measured frequency of the monotonic clock's underlying counter (the TSC here;
|
||||
/// CNTVCT on aarch64, `time` on riscv64), in Hz.
|
||||
pub fn clockHz() u64 {
|
||||
return apic.tscHz();
|
||||
}
|
||||
|
||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||
pub fn enableInterrupts() void {
|
||||
asm volatile ("sti");
|
||||
}
|
||||
|
||||
/// Mask maskable interrupts (`cli`).
|
||||
pub fn disableInterrupts() void {
|
||||
asm volatile ("cli");
|
||||
}
|
||||
|
||||
/// Disable interrupts and return the previous flags, so a nested critical section
|
||||
/// can restore the caller's state rather than blindly re-enabling. Pairs with
|
||||
/// restoreInterrupts.
|
||||
pub fn saveInterrupts() u64 {
|
||||
var flags: u64 = undefined;
|
||||
asm volatile (
|
||||
\\pushfq
|
||||
\\pop %[f]
|
||||
\\cli
|
||||
: [f] "=r" (flags),
|
||||
:
|
||||
: .{ .memory = true }
|
||||
);
|
||||
return flags;
|
||||
}
|
||||
|
||||
/// Re-enable interrupts only if they were enabled when `flags` was captured.
|
||||
pub fn restoreInterrupts(flags: u64) void {
|
||||
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
|
||||
}
|
||||
|
||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
apic.setTickHook(hook);
|
||||
}
|
||||
|
||||
// --- context switching (for the scheduler) -------------------------------
|
||||
|
||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
||||
|
||||
pub fn switchContext(old_sp: *usize, new_sp: usize) void {
|
||||
switch_context(old_sp, new_sp);
|
||||
}
|
||||
|
||||
/// Build the initial stack for a new task so that switching to it lands in
|
||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
||||
/// The layout must match switch_context's push order (callee-saved, then the
|
||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
||||
var sp = stack_top;
|
||||
const push = struct {
|
||||
fn f(p: *usize, value: usize) void {
|
||||
p.* -= @sizeOf(usize);
|
||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
||||
}
|
||||
}.f;
|
||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
||||
push(&sp, 0); // rbx
|
||||
push(&sp, 0); // rbp
|
||||
push(&sp, 0); // r12
|
||||
push(&sp, 0); // r13
|
||||
push(&sp, 0); // r14
|
||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
||||
return sp;
|
||||
}
|
||||
|
||||
/// Drop the current (kernel-context) task to ring 3 at `rip` on `rsp`, never
|
||||
/// returning (defined in isr.s). Used by the scheduler's user-task trampoline
|
||||
/// once it has switched onto the task and read its entry/stack. Interrupts are
|
||||
/// disabled across the swapgs+iretq so no interrupt observes the user GS base in
|
||||
/// ring 0; the pushed RFLAGS re-enables them in ring 3.
|
||||
extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn;
|
||||
|
||||
pub fn jumpToUser(entry: u64, stack_top: u64) noreturn {
|
||||
jump_to_user(entry, stack_top);
|
||||
}
|
||||
|
||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||
/// return. Until set, faults just halt the core.
|
||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||
idt.on_fault = handler;
|
||||
}
|
||||
|
||||
/// A human-readable name for a CPU exception vector.
|
||||
pub fn exceptionName(vector: u64) []const u8 {
|
||||
return idt.vectorName(vector);
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1/2/4) from an I/O port. The generic device layer drives
|
||||
/// ACPI registers through this rather than naming x86 port instructions; on an
|
||||
/// MMIO-only architecture this would be implemented differently.
|
||||
pub fn pioRead(width: u8, port: u16) u32 {
|
||||
return switch (width) {
|
||||
1 => io.inb(port),
|
||||
2 => io.inw(port),
|
||||
4 => io.inl(port),
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Write `width` bytes (1/2/4) to an I/O port.
|
||||
pub fn pioWrite(width: u8, port: u16, value: u32) void {
|
||||
switch (width) {
|
||||
1 => io.outb(port, @truncate(value)),
|
||||
2 => io.outw(port, @truncate(value)),
|
||||
4 => io.outl(port, value),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
||||
/// docs/halting.md for the full reasoning.
|
||||
pub fn halt() noreturn {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
/// Spin-wait hint (`pause`). Emitted in the body of a spinlock's busy-wait: it
|
||||
/// relaxes the core while it polls a contended lock — yielding pipeline resources
|
||||
/// to a hyperthread sibling and easing the cache-coherency traffic on the lock
|
||||
/// line. Purely a performance/power hint; correct to omit, but kinder on the bus.
|
||||
pub fn cpuRelax() void {
|
||||
asm volatile ("pause");
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
|
||||
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
|
||||
//! reference a code selector — so we install our own flat GDT with known
|
||||
//! selectors (0x08/0x10 kernel code/data, 0x18/0x20 user data/code for ring 3)
|
||||
//! rather than trusting whatever the firmware left in place.
|
||||
//!
|
||||
//! The code/data descriptors are identical on every core, but the **TSS descriptor
|
||||
//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see
|
||||
//! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its
|
||||
//! own copy of the table with its own TSS descriptor. Slot 0 is the BSP.
|
||||
|
||||
const parameters = @import("parameters");
|
||||
|
||||
/// Selectors into the table (index * 8). Same on every core's GDT.
|
||||
pub const kernel_code = 0x08;
|
||||
pub const kernel_data = 0x10;
|
||||
pub const user_data = 0x18;
|
||||
pub const user_code = 0x20;
|
||||
pub const tss_selector = 0x28;
|
||||
|
||||
/// Ring-3 selectors as loaded from user mode: RPL 3 or'd in. The user *data*
|
||||
/// descriptor is load-bearing even in long mode — iretq to CPL 3 with a null SS
|
||||
/// raises #GP(0).
|
||||
pub const user_code_rpl3 = user_code | 3;
|
||||
pub const user_data_rpl3 = user_data | 3;
|
||||
|
||||
const maximum_cpus = parameters.maximum_cpus;
|
||||
const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high
|
||||
|
||||
/// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor,
|
||||
/// filled in per core by `setTssFor`.
|
||||
/// kernel code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
||||
/// kernel data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
||||
/// user data: present, ring 3, writable -> 0x00CFF200_0000FFFF
|
||||
/// user code: present, ring 3, executable, readable, L=1 -> 0x00AFFA00_0000FFFF
|
||||
/// User data sits below user code so a future SYSRET works unchanged: it loads
|
||||
/// CS = STAR.SYSRET_CS + 16 and SS = STAR.SYSRET_CS + 8, so with SYSRET_CS = 0x10
|
||||
/// those land on 0x20 (user code) and 0x18 (user data).
|
||||
const template = [entries]u64{
|
||||
0, // null descriptor (required)
|
||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
||||
0x00CF92000000FFFF, // kernel data (0x10)
|
||||
0x00CFF2000000FFFF, // user data (0x18)
|
||||
0x00AFFA000000FFFF, // user code (0x20)
|
||||
0, // TSS descriptor low (0x28)
|
||||
0, // TSS descriptor high
|
||||
};
|
||||
|
||||
/// One GDT per core (each a copy of the template, differing only in its TSS slot).
|
||||
var gdts = [_][entries]u64{template} ** maximum_cpus;
|
||||
|
||||
/// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task
|
||||
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
|
||||
/// TSS. Write it into that core's GDT before it loads the TSS selector.
|
||||
pub fn setTssFor(cpu: usize, base: u64, limit: u64) void {
|
||||
gdts[cpu][5] = (limit & 0xFFFF) |
|
||||
((base & 0xFFFF) << 16) |
|
||||
(((base >> 16) & 0xFF) << 32) |
|
||||
(@as(u64, 0x89) << 40) |
|
||||
(((limit >> 16) & 0xF) << 48) |
|
||||
(((base >> 24) & 0xFF) << 56);
|
||||
gdts[cpu][6] = (base >> 32) & 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
||||
const Descriptor = packed struct {
|
||||
limit: u16,
|
||||
base: u64,
|
||||
};
|
||||
|
||||
/// Loads the GDT and reloads the segment registers (including CS). Defined in
|
||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match the table.
|
||||
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
||||
|
||||
/// Load core `cpu`'s GDT and switch onto its segments. Note this reloads the segment
|
||||
/// registers, which zeroes the GS base — so a core must publish its per-CPU pointer
|
||||
/// (setCpuLocal) *after* calling this.
|
||||
pub fn loadOnThisCpu(cpu: usize) void {
|
||||
const descriptor = Descriptor{
|
||||
.limit = @sizeOf([entries]u64) - 1,
|
||||
.base = @intFromPtr(&gdts[cpu]),
|
||||
};
|
||||
gdt_flush(&descriptor);
|
||||
}
|
||||
|
||||
/// Install the bootstrap processor's GDT (slot 0) and switch onto its segments.
|
||||
pub fn init() void {
|
||||
loadOnThisCpu(0);
|
||||
}
|
||||
@@ -0,0 +1,186 @@
|
||||
//! Interrupt Descriptor Table, CPU-exception handlers, and device-interrupt
|
||||
//! dispatch. Without this, any fault (a stray pointer, a bad page-table entry)
|
||||
//! triple-faults and silently resets the machine. With it, the CPU vectors into
|
||||
//! our stubs, which capture the register state and hand it to a dispatcher.
|
||||
//!
|
||||
//! Vectors split in two: 0-31 are CPU exceptions (terminal — reported and
|
||||
//! halted); 32+ are device interrupts (a registered handler runs, the APIC is
|
||||
//! acknowledged, and we return to the interrupted code).
|
||||
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
|
||||
/// Highest vector we install a gate/stub for (exceptions 0-31 plus the device
|
||||
/// range 32-47, which covers the timer and the spurious vector).
|
||||
const gate_count = 48;
|
||||
|
||||
/// A device-interrupt handler. It doesn't get the trap frame (a timer or keyboard
|
||||
/// handler doesn't need the interrupted registers); add that if one ever does.
|
||||
pub const Handler = *const fn () void;
|
||||
|
||||
var handlers = [_]?Handler{null} ** 256;
|
||||
|
||||
/// Register `handler` for a device-interrupt `vector` (>= 32).
|
||||
pub fn setHandler(vector: usize, handler: Handler) void {
|
||||
handlers[vector] = handler;
|
||||
}
|
||||
|
||||
/// The ring-3 system_call gate's vector (`int $0x80`, the classic choice — well away
|
||||
/// from the device range) and its handler. Unlike device handlers, a system_call
|
||||
/// handler gets the (mutable) trap frame: it reads its arguments from the saved
|
||||
/// user registers and writes rax as the return value, which isr_common then
|
||||
/// restores into the user context.
|
||||
pub const system_call_vector = 128;
|
||||
|
||||
var system_call_handler: ?*const fn (*CpuState) void = null;
|
||||
|
||||
pub fn setSystemCallHandler(handler: *const fn (*CpuState) void) void {
|
||||
system_call_handler = handler;
|
||||
}
|
||||
|
||||
/// The register + trap frame the ISR stubs build on the stack, laid out so the
|
||||
/// lowest address (where RSP points when we call the handler) is the first field.
|
||||
/// See the push order in `isrCommon` below.
|
||||
pub const CpuState = extern struct {
|
||||
r15: u64,
|
||||
r14: u64,
|
||||
r13: u64,
|
||||
r12: u64,
|
||||
r11: u64,
|
||||
r10: u64,
|
||||
r9: u64,
|
||||
r8: u64,
|
||||
rbp: u64,
|
||||
rdi: u64,
|
||||
rsi: u64,
|
||||
rdx: u64,
|
||||
rcx: u64,
|
||||
rbx: u64,
|
||||
rax: u64,
|
||||
vector: u64, // pushed by the per-vector stub
|
||||
error_code: u64, // real one from the CPU, or 0 pushed by the stub
|
||||
rip: u64, // from here down: pushed by the CPU on entry
|
||||
cs: u64,
|
||||
rflags: u64,
|
||||
rsp: u64,
|
||||
ss: u64,
|
||||
};
|
||||
|
||||
/// Where a fault is reported. The kernel overrides this (see setFaultHandler) with
|
||||
/// something that prints to the console; until then, just stop.
|
||||
pub var on_fault: *const fn (*const CpuState) noreturn = defaultFault;
|
||||
|
||||
fn defaultFault(_: *const CpuState) noreturn {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
/// Names for the 32 defined exception vectors, for readable output.
|
||||
const names = [_][]const u8{
|
||||
"divide error", "debug",
|
||||
"NMI", "breakpoint",
|
||||
"overflow", "bound range exceeded",
|
||||
"invalid opcode", "device not available",
|
||||
"double fault", "coprocessor segment overrun",
|
||||
"invalid TSS", "segment not present",
|
||||
"stack-segment fault", "general protection fault",
|
||||
"page fault", "reserved (15)",
|
||||
"x87 floating-point", "alignment check",
|
||||
"machine check", "SIMD floating-point",
|
||||
"virtualization", "control protection",
|
||||
"reserved (22)", "reserved (23)",
|
||||
"reserved (24)", "reserved (25)",
|
||||
"reserved (26)", "reserved (27)",
|
||||
"hypervisor injection", "VMM communication",
|
||||
"security exception", "reserved (31)",
|
||||
};
|
||||
|
||||
pub fn vectorName(vector: u64) []const u8 {
|
||||
return if (vector < names.len) names[vector] else "unknown";
|
||||
}
|
||||
|
||||
/// A 64-bit IDT gate descriptor (16 bytes).
|
||||
const Gate = packed struct {
|
||||
offset_low: u16,
|
||||
selector: u16,
|
||||
ist: u8, // interrupt-stack-table index; 0 = use the current stack
|
||||
flags: u8, // present, DPL, gate type
|
||||
offset_mid: u16,
|
||||
offset_high: u32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
var idt = [_]Gate{std.mem.zeroes(Gate)} ** 256;
|
||||
|
||||
const Descriptor = packed struct {
|
||||
limit: u16,
|
||||
base: u64,
|
||||
};
|
||||
|
||||
/// Loads the IDT (`lidt`). Defined in isr.s.
|
||||
extern fn idt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
||||
|
||||
fn setGate(vector: usize, handler: u64) void {
|
||||
idt[vector] = .{
|
||||
.offset_low = @truncate(handler),
|
||||
.selector = gdt.kernel_code,
|
||||
.ist = 0,
|
||||
.flags = 0x8E, // present, ring 0, 64-bit interrupt gate
|
||||
.offset_mid = @truncate(handler >> 16),
|
||||
.offset_high = @truncate(handler >> 32),
|
||||
};
|
||||
}
|
||||
|
||||
/// Point every installed vector at its stub (isr.s) and load the IDT.
|
||||
pub fn init() void {
|
||||
@setEvalBranchQuota(20000); // comptimePrint across all the gates adds up
|
||||
inline for (0..gate_count) |vector| {
|
||||
const stub = @extern(*const anyopaque, .{ .name = std.fmt.comptimePrint("isr{d}", .{vector}) });
|
||||
setGate(vector, @intFromPtr(stub));
|
||||
}
|
||||
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
|
||||
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
|
||||
idt[8].ist = tss.double_fault_ist;
|
||||
// The system_call gate. Installed outside the 0..gate_count loop (stubs 48-127
|
||||
// don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a
|
||||
// #GP. An interrupt gate (not trap): IF is cleared for the handler, which
|
||||
// the ring-3 exit path relies on.
|
||||
const system_call_stub = @extern(*const anyopaque, .{ .name = "isr128" });
|
||||
setGate(system_call_vector, @intFromPtr(system_call_stub));
|
||||
idt[system_call_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate
|
||||
loadOnThisCpu();
|
||||
}
|
||||
|
||||
/// Load the (shared, already-populated) IDT on the current core. The gate table is
|
||||
/// read-only after `init`, so every core points its IDTR at the same one. Called by
|
||||
/// the BSP via `init` and by each AP during bring-up.
|
||||
pub fn loadOnThisCpu() void {
|
||||
const descriptor = Descriptor{
|
||||
.limit = @sizeOf(@TypeOf(idt)) - 1,
|
||||
.base = @intFromPtr(&idt),
|
||||
};
|
||||
idt_flush(&descriptor);
|
||||
}
|
||||
|
||||
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
|
||||
/// assembly stubs can `call` it by name. Exceptions are terminal; device
|
||||
/// interrupts run their handler, get acknowledged, and return.
|
||||
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
|
||||
if (state.vector < 32) {
|
||||
on_fault(state); // CPU exception — never returns
|
||||
} else if (state.vector == system_call_vector) {
|
||||
// Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI.
|
||||
if (system_call_handler) |handler| handler(state);
|
||||
} else if (handlers[state.vector]) |handler| {
|
||||
// The handler owns its EOI. It used to be issued here, before the call —
|
||||
// correct for the LAPIC timer, but impossible to reconcile with a
|
||||
// level-triggered device line, which must be **masked at the I/O APIC
|
||||
// before** it is acknowledged or it redelivers instantly and storms
|
||||
// (the driver that would quiet it lives in ring 3 and hasn't run yet).
|
||||
// Only the handler knows which discipline its source needs, so only the
|
||||
// handler can sequence it. See apic.timerTick and irq.dispatch.
|
||||
handler();
|
||||
}
|
||||
// else: spurious/unhandled device interrupt — don't acknowledge it
|
||||
}
|
||||
|
||||
const std = @import("std");
|
||||
@@ -0,0 +1,68 @@
|
||||
//! x86 port I/O and model-specific registers — the low-level primitives the
|
||||
//! serial port and the APIC talk to hardware through.
|
||||
|
||||
pub fn outb(port: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[port]"
|
||||
:
|
||||
: [value] "{al}" (value),
|
||||
[port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn inb(port: u16) u8 {
|
||||
return asm volatile ("inb %[port], %[value]"
|
||||
: [value] "={al}" (-> u8),
|
||||
: [port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn outw(port: u16, value: u16) void {
|
||||
asm volatile ("outw %[value], %[port]"
|
||||
:
|
||||
: [value] "{ax}" (value),
|
||||
[port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn inw(port: u16) u16 {
|
||||
return asm volatile ("inw %[port], %[value]"
|
||||
: [value] "={ax}" (-> u16),
|
||||
: [port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn outl(port: u16, value: u32) void {
|
||||
asm volatile ("outl %[value], %[port]"
|
||||
:
|
||||
: [value] "{eax}" (value),
|
||||
[port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn inl(port: u16) u32 {
|
||||
return asm volatile ("inl %[port], %[value]"
|
||||
: [value] "={eax}" (-> u32),
|
||||
: [port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
/// Read a model-specific register (returns edx:eax combined).
|
||||
pub fn rdmsr(msr: u32) u64 {
|
||||
var low: u32 = undefined;
|
||||
var high: u32 = undefined;
|
||||
asm volatile ("rdmsr"
|
||||
: [low] "={eax}" (low),
|
||||
[high] "={edx}" (high),
|
||||
: [msr] "{ecx}" (msr),
|
||||
);
|
||||
return (@as(u64, high) << 32) | low;
|
||||
}
|
||||
|
||||
pub fn wrmsr(msr: u32, value: u64) void {
|
||||
asm volatile ("wrmsr"
|
||||
:
|
||||
: [msr] "{ecx}" (msr),
|
||||
[low] "{eax}" (@as(u32, @truncate(value))),
|
||||
[high] "{edx}" (@as(u32, @truncate(value >> 32))),
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,152 @@
|
||||
//! I/O APIC — routes external device interrupts (a device's line) to a LAPIC
|
||||
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
||||
//! ACPI's MADT (via discovery), never assumed.
|
||||
//!
|
||||
//! `init` maps the I/O APIC and **masks every input** — the correct quiescent state
|
||||
//! on a legacy-free machine. Lines are then unmasked one at a time, as user-space
|
||||
//! drivers bind them (`routeGsi`/`unmaskGsi`, driven by system/kernel/irq.zig).
|
||||
//!
|
||||
//! Two entry points, for two kinds of caller. `routeIrq` takes a legacy **ISA IRQ**
|
||||
//! and resolves it through the MADT overrides — for in-kernel use, and still without
|
||||
//! a caller. `routeGsi` takes a **GSI** directly, which is what a device's own
|
||||
//! routing capability names (e.g. the HPET's `Tn_INT_ROUTE_CAP`), and is the path a
|
||||
//! bound driver interrupt takes.
|
||||
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// A MADT Interrupt Source Override: an ISA IRQ that appears at a different global
|
||||
/// system interrupt, with its own polarity/trigger (MPS INTI `flags`).
|
||||
pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
|
||||
|
||||
var base: u64 = 0; // 0 = no I/O APIC discovered
|
||||
var gsi_base: u32 = 0;
|
||||
var maximum_entries: u32 = 0;
|
||||
var overrides: [16]IsoEntry = undefined;
|
||||
var override_count: usize = 0;
|
||||
|
||||
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
||||
const register_ioregsel = 0x00;
|
||||
const register_iowin = 0x10;
|
||||
const register_version = 0x01;
|
||||
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
||||
const redir_mask = 1 << 16; // mask bit in the low dword
|
||||
|
||||
/// Supply the discovered I/O APIC location + the MADT IRQ overrides. Call before `init`.
|
||||
pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) void {
|
||||
base = ioapic_base;
|
||||
gsi_base = ioapic_gsi_base;
|
||||
override_count = @min(isos.len, overrides.len);
|
||||
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
||||
}
|
||||
|
||||
fn registerRead(index: u32) u32 {
|
||||
@as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index;
|
||||
return @as(*volatile u32, @ptrFromInt(base + register_iowin)).*;
|
||||
}
|
||||
fn registerWrite(index: u32, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(base + register_ioregsel)).* = index;
|
||||
@as(*volatile u32, @ptrFromInt(base + register_iowin)).* = value;
|
||||
}
|
||||
|
||||
fn writeEntry(n: u32, low: u32, high: u32) void {
|
||||
registerWrite(redir_base + 2 * n, low);
|
||||
registerWrite(redir_base + 2 * n + 1, high);
|
||||
}
|
||||
|
||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
||||
pub fn init() void {
|
||||
if (base == 0) return;
|
||||
// Reach the I/O APIC through the physmap; switch `base` to that virtual
|
||||
// address so the register accessors work without the identity map.
|
||||
base = paging.mapMmio(base, 0x1000, true);
|
||||
maximum_entries = ((registerRead(register_version) >> 16) & 0xFF) + 1;
|
||||
var n: u32 = 0;
|
||||
while (n < maximum_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
||||
}
|
||||
|
||||
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
||||
/// for its GSI/polarity/trigger, and unmask it. No caller yet — groundwork for the
|
||||
/// first device driver.
|
||||
pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
||||
if (base == 0) return;
|
||||
|
||||
var gsi: u32 = irq;
|
||||
var flags: u16 = 0;
|
||||
for (overrides[0..override_count]) |o| {
|
||||
if (o.source == irq) {
|
||||
gsi = o.gsi;
|
||||
flags = o.flags;
|
||||
}
|
||||
}
|
||||
if (gsi < gsi_base) return;
|
||||
const n = gsi - gsi_base;
|
||||
if (n >= maximum_entries) return;
|
||||
|
||||
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
||||
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
||||
var low: u32 = vector;
|
||||
if (flags & 0x3 == 3) low |= (1 << 13);
|
||||
if ((flags >> 2) & 0x3 == 3) low |= (1 << 15);
|
||||
const high: u32 = @as(u32, apic_id) << 24; // destination APIC ID
|
||||
writeEntry(n, low, high);
|
||||
}
|
||||
|
||||
// --- GSI-level control (the user-space driver path) --------------------------
|
||||
//
|
||||
// `routeIrq` above takes an *ISA IRQ* and resolves it through the MADT overrides.
|
||||
// A driver-bound interrupt is already a **GSI** (the device told us so, e.g. the
|
||||
// HPET's `Tn_INT_ROUTE_CAP`), so it needs no override lookup — just the redirection
|
||||
// entry. These three are what `system/kernel/irq.zig` drives.
|
||||
//
|
||||
// Callers must serialise: the I/O APIC is reached through an index/data register
|
||||
// pair, so two cores interleaving `registerWrite` would corrupt each other. The kernel
|
||||
// holds the big lock across these.
|
||||
|
||||
/// Redirection-entry index for `gsi`, or null if this I/O APIC doesn't own it.
|
||||
fn entryFor(gsi: u32) ?u32 {
|
||||
if (base == 0 or gsi < gsi_base) return null;
|
||||
const n = gsi - gsi_base;
|
||||
return if (n < maximum_entries) n else null;
|
||||
}
|
||||
|
||||
/// True if `gsi` lands on this I/O APIC — the kernel's validity check before binding.
|
||||
pub fn ownsGsi(gsi: u32) bool {
|
||||
return entryFor(gsi) != null;
|
||||
}
|
||||
|
||||
/// Point `gsi` at `vector` on the LAPIC `apic_id`, with explicit polarity/trigger,
|
||||
/// and leave it **masked**. The caller unmasks once a handler is bound — otherwise a
|
||||
/// device asserting between route and bind would fire into a null handler.
|
||||
pub fn routeGsi(gsi: u32, vector: u8, apic_id: u8, level: bool, active_low: bool) void {
|
||||
const n = entryFor(gsi) orelse return;
|
||||
var low: u32 = @as(u32, vector) | redir_mask; // masked until bound
|
||||
if (active_low) low |= (1 << 13);
|
||||
if (level) low |= (1 << 15);
|
||||
writeEntry(n, low, @as(u32, apic_id) << 24);
|
||||
}
|
||||
|
||||
/// Stop `gsi` reaching any CPU. Called from the ISR *before* the LAPIC EOI: a
|
||||
/// level-triggered line is still asserted at that point, so an unmasked entry would
|
||||
/// redeliver immediately and storm before the user-space driver ever runs.
|
||||
pub fn maskGsi(gsi: u32) void {
|
||||
const n = entryFor(gsi) orelse return;
|
||||
registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) | redir_mask);
|
||||
}
|
||||
|
||||
/// Let `gsi` through again — the tail of `irq_ack`, once the driver has quieted the
|
||||
/// device (so the line is deasserted and this can't immediately refire).
|
||||
pub fn unmaskGsi(gsi: u32) void {
|
||||
const n = entryFor(gsi) orelse return;
|
||||
registerWrite(redir_base + 2 * n, registerRead(redir_base + 2 * n) & ~@as(u32, redir_mask));
|
||||
}
|
||||
|
||||
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
||||
pub fn entryCount() u32 {
|
||||
return maximum_entries;
|
||||
}
|
||||
|
||||
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
||||
pub fn entryLow(n: u32) u32 {
|
||||
if (base == 0) return 0;
|
||||
return registerRead(redir_base + 2 * n);
|
||||
}
|
||||
@@ -0,0 +1,382 @@
|
||||
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
||||
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
||||
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
|
||||
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
||||
#
|
||||
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
||||
# error code where the CPU pushes none, then the vector number — and jumps to the
|
||||
# shared tail, which saves the general registers and calls the Zig handler with a
|
||||
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
||||
|
||||
.text
|
||||
|
||||
# _start: the kernel entry. The loader jumps here (higher-half address) with
|
||||
# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned
|
||||
# stack in .bss (the loader stack is a low address that goes away once the low
|
||||
# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never
|
||||
# returns; the hlt loop is a belt-and-braces backstop.
|
||||
.global _start
|
||||
_start:
|
||||
leaq bootstrap_stack_top(%rip), %rsp
|
||||
call kmainEntry
|
||||
1: hlt
|
||||
jmp 1b
|
||||
|
||||
# The kernel's initial stack (used until the scheduler hands each task its own).
|
||||
# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it
|
||||
# needs more than a token stack. Lives in .bss (zeroed, higher-half).
|
||||
.section .bss
|
||||
.balign 16
|
||||
bootstrap_stack:
|
||||
.skip 65536
|
||||
bootstrap_stack_top:
|
||||
.text
|
||||
|
||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
||||
# set with mov, so we far-return through the caller's own return address.
|
||||
.global gdt_flush
|
||||
gdt_flush:
|
||||
lgdt (%rdi)
|
||||
mov $0x10, %ax # kernel data selector
|
||||
mov %ax, %ds
|
||||
mov %ax, %es
|
||||
mov %ax, %ss
|
||||
mov %ax, %fs
|
||||
mov %ax, %gs
|
||||
pop %rax # caller's return address
|
||||
push $0x08 # kernel code selector (new CS)
|
||||
push %rax # return address (new RIP)
|
||||
lretq
|
||||
|
||||
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
||||
.global idt_flush
|
||||
idt_flush:
|
||||
lidt (%rdi)
|
||||
ret
|
||||
|
||||
# load_tr(di = TSS selector): load the task register.
|
||||
.global load_tr
|
||||
load_tr:
|
||||
ltr %di
|
||||
ret
|
||||
|
||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
||||
# Cooperative context switch: save the callee-saved registers on the current
|
||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
||||
# registers are the compiler's responsibility (this looks like a normal call).
|
||||
.global switch_context
|
||||
switch_context:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
||||
mov %rsi, %rsp # switch to the new task's stack
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret # return into the new task's saved instruction pointer
|
||||
|
||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
||||
# leaves its entry function in r15. A fresh task is switched to with the big kernel
|
||||
# lock held (the hand-off rule in sync.zig) but has no enter/leave frame of its own,
|
||||
# so it releases the lock here before running its body. r15 survives the call (it's
|
||||
# callee-saved). New tasks then start with interrupts enabled.
|
||||
.extern releaseForFreshTask
|
||||
.global task_trampoline
|
||||
task_trampoline:
|
||||
call releaseForFreshTask # drop the kernel lock we inherited across the switch
|
||||
sti
|
||||
call *%r15 # call the task entry (fn() void)
|
||||
1: hlt # if the entry returns, idle (still preemptible)
|
||||
jmp 1b
|
||||
|
||||
# user_task_trampoline: the first thing a freshly-spawned *user* task runs.
|
||||
# init_user_task_stack leaves the user entry in r15 and the user stack in r14
|
||||
# (both callee-saved, so they survive the lock-release call). Like task_trampoline
|
||||
# it drops the inherited kernel lock, then — instead of calling a kernel fn — it
|
||||
# builds an iretq frame and drops to ring 3. The scheduler's switchTo already
|
||||
# loaded this task's address space (CR3) and published its kernel stack
|
||||
# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls
|
||||
# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed).
|
||||
# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context
|
||||
# to ring 3, never returning. The scheduler calls this from a fresh user task's
|
||||
# trampoline (after the lock is released and the entry/stack read from the Task).
|
||||
# cli guards the swapgs..iretq window: an interrupt there would run in ring 0
|
||||
# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set)
|
||||
# re-enables interrupts on the drop to ring 3.
|
||||
.global jump_to_user
|
||||
jump_to_user:
|
||||
cli
|
||||
push $0x1B # user SS (0x18 | RPL 3)
|
||||
push %rsi # user RSP
|
||||
push $0x202 # RFLAGS: IF | reserved-1
|
||||
push $0x23 # user CS (0x20 | RPL 3)
|
||||
push %rdi # user RIP
|
||||
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||
iretq
|
||||
|
||||
# --- ring 3 entry/exit ------------------------------------------------------
|
||||
|
||||
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
||||
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
|
||||
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
|
||||
# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames
|
||||
# grow safely into it — then builds the 5-word iretq frame with the user
|
||||
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
|
||||
# timer keeps running in user mode.
|
||||
.global enter_user
|
||||
enter_user:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
|
||||
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
|
||||
mov %rsp, %gs:0 # kernel_rsp: the syscall stub stacks here too
|
||||
push $0x1B # user SS (0x18 | RPL 3)
|
||||
push %rsi # user RSP
|
||||
push $0x202 # RFLAGS: IF | reserved-1
|
||||
push $0x23 # user CS (0x20 | RPL 3)
|
||||
push %rdi # user RIP
|
||||
swapgs # user GS base for ring 3 (isr_common swaps back)
|
||||
iretq
|
||||
|
||||
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
|
||||
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
|
||||
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
|
||||
# and stay off — the Zig caller re-enables.
|
||||
.global user_exit_to_kernel
|
||||
user_exit_to_kernel:
|
||||
mov user_saved_rsp(%rip), %rsp
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret
|
||||
|
||||
# syscall_entry: the target of the SYSCALL instruction (LSTAR). The CPU does NOT
|
||||
# switch stacks — it puts the return RIP in RCX, the saved RFLAGS in R11, loads
|
||||
# CS/SS from STAR, masks RFLAGS with SFMASK (so IF is already clear), and jumps
|
||||
# here with RSP still the *user* stack. We swap in the kernel GS, switch to the
|
||||
# task's kernel stack via the per-CPU block, build a CpuState frame identical to
|
||||
# the interrupt path's, and reuse interruptDispatch (vector 128) — then SYSRET.
|
||||
#
|
||||
# Hazard (acceptable while init is the only, trusted, user program): SYSRETQ #GPs
|
||||
# in ring 0 if the return RIP (RCX) is non-canonical. A hostile user could arrange
|
||||
# that; hardening (canonical check / iretq fallback) is a later security-track item.
|
||||
.global syscall_entry
|
||||
syscall_entry:
|
||||
swapgs # kernel GS base
|
||||
movq %rsp, %gs:8 # stash user rsp in the scratch slot
|
||||
movq %gs:0, %rsp # switch to this task's kernel stack
|
||||
# Build the trap frame (same field order as isr_common), highest field first.
|
||||
pushq $0x1B # ss (user data | 3)
|
||||
pushq %gs:8 # rsp (user, from scratch)
|
||||
pushq %r11 # rflags (saved by syscall)
|
||||
pushq $0x23 # cs (user code | 3)
|
||||
pushq %rcx # rip (saved by syscall)
|
||||
pushq $0 # error_code (none for a syscall)
|
||||
pushq $128 # vector (same as the int 0x80 gate)
|
||||
push %rax
|
||||
push %rbx
|
||||
push %rcx
|
||||
push %rdx
|
||||
push %rsi
|
||||
push %rdi
|
||||
push %rbp
|
||||
push %r8
|
||||
push %r9
|
||||
push %r10
|
||||
push %r11
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # trap-frame pointer
|
||||
call interruptDispatch
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %r11
|
||||
pop %r10
|
||||
pop %r9
|
||||
pop %r8
|
||||
pop %rbp
|
||||
pop %rdi
|
||||
pop %rsi
|
||||
pop %rdx
|
||||
pop %rcx
|
||||
pop %rbx
|
||||
pop %rax
|
||||
add $16, %rsp # drop vector + error_code -> rsp at rip
|
||||
popq %rcx # rip -> RCX (SYSRETQ restores RIP from RCX)
|
||||
addq $8, %rsp # skip the cs slot (SYSRETQ loads CS from STAR)
|
||||
popq %r11 # rflags -> R11 (SYSRETQ restores RFLAGS from R11)
|
||||
popq %rsp # user rsp (the ss slot below is abandoned)
|
||||
swapgs # user GS base
|
||||
sysretq # -> ring 3: RIP=RCX, RFLAGS=R11, CS/SS from STAR
|
||||
|
||||
.section .bss
|
||||
.balign 8
|
||||
user_saved_rsp:
|
||||
.skip 8
|
||||
.text
|
||||
|
||||
# --- user-mode test program --------------------------------------------------
|
||||
# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and
|
||||
# entered via enter_user. Position-independent (immediates and short jumps only).
|
||||
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
|
||||
# from the user mapping. (The old hello/ping blob was retired once /sbin/init
|
||||
# became the real ring-3 exerciser; only the isolation proof remains.)
|
||||
.section .rodata
|
||||
|
||||
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
|
||||
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
|
||||
# page, so this must take a #PF with error code 0x5 (present | user) before any
|
||||
# access happens. movabs loads the full 64-bit higher-half address (a disp32
|
||||
# would sign-extend and miss).
|
||||
.global user_pf_start
|
||||
.global user_pf_end
|
||||
user_pf_start:
|
||||
movabs $0xFFFF8800FEE00000, %rcx
|
||||
mov (%rcx), %rax
|
||||
1: jmp 1b
|
||||
user_pf_end:
|
||||
|
||||
.text
|
||||
|
||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
||||
.macro STUB_NOERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $0
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
||||
.macro STUB_ERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
STUB_NOERR 0
|
||||
STUB_NOERR 1
|
||||
STUB_NOERR 2
|
||||
STUB_NOERR 3
|
||||
STUB_NOERR 4
|
||||
STUB_NOERR 5
|
||||
STUB_NOERR 6
|
||||
STUB_NOERR 7
|
||||
STUB_ERR 8
|
||||
STUB_NOERR 9
|
||||
STUB_ERR 10
|
||||
STUB_ERR 11
|
||||
STUB_ERR 12
|
||||
STUB_ERR 13
|
||||
STUB_ERR 14
|
||||
STUB_NOERR 15
|
||||
STUB_NOERR 16
|
||||
STUB_ERR 17
|
||||
STUB_NOERR 18
|
||||
STUB_NOERR 19
|
||||
STUB_NOERR 20
|
||||
STUB_ERR 21
|
||||
STUB_NOERR 22
|
||||
STUB_NOERR 23
|
||||
STUB_NOERR 24
|
||||
STUB_NOERR 25
|
||||
STUB_NOERR 26
|
||||
STUB_NOERR 27
|
||||
STUB_NOERR 28
|
||||
STUB_NOERR 29
|
||||
STUB_NOERR 30
|
||||
STUB_NOERR 31
|
||||
|
||||
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
||||
# code, so they all use the dummy-zero form.
|
||||
STUB_NOERR 32
|
||||
STUB_NOERR 33
|
||||
STUB_NOERR 34
|
||||
STUB_NOERR 35
|
||||
STUB_NOERR 36
|
||||
STUB_NOERR 37
|
||||
STUB_NOERR 38
|
||||
STUB_NOERR 39
|
||||
STUB_NOERR 40
|
||||
STUB_NOERR 41
|
||||
STUB_NOERR 42
|
||||
STUB_NOERR 43
|
||||
STUB_NOERR 44
|
||||
STUB_NOERR 45
|
||||
STUB_NOERR 46
|
||||
STUB_NOERR 47
|
||||
|
||||
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
|
||||
# vector; dispatched specially in interruptDispatch.
|
||||
STUB_NOERR 128
|
||||
|
||||
.extern interruptDispatch
|
||||
|
||||
# Shared tail. Register push order here defines the CpuState field order.
|
||||
# If the interrupt came from ring 3 the GS base holds the user's value, so swap
|
||||
# in the kernel's before anything reads per-CPU data (swapgs discipline; see
|
||||
# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24).
|
||||
isr_common:
|
||||
testb $3, 24(%rsp)
|
||||
jz 1f
|
||||
swapgs
|
||||
1: push %rax
|
||||
push %rbx
|
||||
push %rcx
|
||||
push %rdx
|
||||
push %rsi
|
||||
push %rdi
|
||||
push %rbp
|
||||
push %r8
|
||||
push %r9
|
||||
push %r10
|
||||
push %r11
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
||||
call interruptDispatch
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %r11
|
||||
pop %r10
|
||||
pop %r9
|
||||
pop %r8
|
||||
pop %rbp
|
||||
pop %rdi
|
||||
pop %rsi
|
||||
pop %rdx
|
||||
pop %rcx
|
||||
pop %rbx
|
||||
pop %rax
|
||||
add $16, %rsp # drop the vector and error code
|
||||
# Symmetric to entry: if returning to ring 3, restore the user GS base. CS is
|
||||
# now at offset 8 (RIP@0, CS@8).
|
||||
testb $3, 8(%rsp)
|
||||
jz 1f
|
||||
swapgs
|
||||
1: iretq
|
||||
@@ -0,0 +1,52 @@
|
||||
/* Kernel link layout — higher half.
|
||||
*
|
||||
* The kernel is linked to *run* in the higher half (virtual base
|
||||
* 0xFFFF_FFFF_8000_0000, matching danos.kernel_virt_base and build.zig's
|
||||
* image_base) but is *loaded* low. Each section's load address (LMA) is its
|
||||
* virtual address minus KERNEL_VIRT_BASE via AT(), so the ELF's p_paddr lands
|
||||
* at a low physical address (.text at 1 MiB) that the loader can allocate and
|
||||
* copy into. The loader maps p_vaddr (high) -> p_paddr (low) in its bootstrap
|
||||
* tables and jumps to the high entry; the kernel then builds its own tables
|
||||
* with the physmap and abandons the identity map. Requires LLD (build.zig pins
|
||||
* it) — the self-hosted linker ignores PHDRS/AT()/section order.
|
||||
*/
|
||||
|
||||
KERNEL_VIRT_BASE = 0xFFFFFFFF80000000;
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
||||
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
.text ALIGN(4K) : AT(ADDR(.text) - KERNEL_VIRT_BASE) {
|
||||
*(.text .text.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : AT(ADDR(.rodata) - KERNEL_VIRT_BASE) {
|
||||
*(.rodata .rodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : AT(ADDR(.data) - KERNEL_VIRT_BASE) {
|
||||
*(.data .data.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space. The loader zeroes it via the
|
||||
* gap between each PT_LOAD segment's file size and memory size, so no
|
||||
* boundary symbols are needed here. */
|
||||
.bss ALIGN(4K) : AT(ADDR(.bss) - KERNEL_VIRT_BASE) {
|
||||
*(.bss .bss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,389 @@
|
||||
//! The kernel's page tables and virtual memory manager.
|
||||
//!
|
||||
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
|
||||
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
|
||||
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
|
||||
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
|
||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||
//! which the kernel heap will build on.
|
||||
//!
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
// Page-table entry bits.
|
||||
const present: u64 = 1 << 0;
|
||||
const writable: u64 = 1 << 1;
|
||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||
const pwt: u64 = 1 << 3; // page write-through
|
||||
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
|
||||
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// ELF segment flags (p_flags).
|
||||
const pf_x: u32 = 1;
|
||||
const pf_w: u32 = 2;
|
||||
|
||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
||||
var kernel_pml4: u64 = 0;
|
||||
var alloc_frame: *const fn () ?u64 = undefined;
|
||||
var free_frame: *const fn (u64) void = undefined; // for tearing down address spaces
|
||||
|
||||
/// Set once the kernel is running on its own tables (past the CR3 load in
|
||||
/// `init`). Before that, the kernel reaches page-table frames through the
|
||||
/// *loader's* bootstrap physmap, which only covers the low 4 GiB — so every
|
||||
/// frame allocated for a table during that window must be below 4 GiB. Both the
|
||||
/// frame allocator and this code scan from low addresses up, so it holds
|
||||
/// naturally; the assertion in `allocTable` makes a violation loud rather than
|
||||
/// a silent fault. After the switch the kernel's own physmap covers all RAM.
|
||||
var on_own_tables = false;
|
||||
|
||||
/// Set at the end of `init`. Guards against a new *higher-half* PML4 entry being
|
||||
/// created afterward: the kernel half is pre-populated at init and then shared
|
||||
/// by copying PML4[256..512) into every process address space (M3), so a late
|
||||
/// top-half entry would be invisible to already-created address spaces.
|
||||
var init_done = false;
|
||||
|
||||
const bootstrap_physmap_limit: u64 = 4 << 30;
|
||||
|
||||
/// Dereference a page-table frame by its physical address, via the physmap.
|
||||
/// This is the single hinge for the higher-half move: page tables hold physical
|
||||
/// frame addresses (pmm gives out physical frames, and CR3/PTEs must be
|
||||
/// physical), but the kernel reaches them at `physmap_base + physical`. Valid under
|
||||
/// both the loader's bootstrap tables and the kernel's own, which share the
|
||||
/// physmap base.
|
||||
fn tableAt(physical: u64) *[512]u64 {
|
||||
return @ptrFromInt(danos.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
fn allocTable() u64 {
|
||||
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
|
||||
if (!on_own_tables and frame >= bootstrap_physmap_limit)
|
||||
@panic("paging: table frame above the 4 GiB bootstrap physmap");
|
||||
@memset(tableAt(frame)[0..], 0);
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Return the table an entry points at, creating it if empty. Intermediate
|
||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||
/// writable only if every level is; non-executable if any level is).
|
||||
fn descend(entry: *u64) u64 {
|
||||
if (entry.* & present != 0) return entry.* & address_mask;
|
||||
const frame = allocTable();
|
||||
entry.* = frame | present | writable;
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Map one 4 KiB page `virtual` -> `physical` with `flags` (present is added).
|
||||
fn mapPage(pml4: u64, virtual: u64, physical: u64, flags: u64) void {
|
||||
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
// The kernel half is fixed after init: every top-half PML4 entry is
|
||||
// pre-created so address spaces can share it by copying these slots. A new
|
||||
// one here would be invisible to address spaces already made.
|
||||
if (init_done and (virtual >> 63) == 1 and pml4e.* & present == 0)
|
||||
@panic("paging: new higher-half PML4 entry after init");
|
||||
const pdpt = descend(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte);
|
||||
const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF];
|
||||
const pt = descend(pde);
|
||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags | present;
|
||||
}
|
||||
|
||||
/// Map [physical_base, physical_base+len) into the physmap (at physicalToVirtual(physical)) with
|
||||
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||
/// window onto physical memory once the low identity map goes away.
|
||||
fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
||||
var address = physical_base & ~@as(u64, page_size - 1);
|
||||
const end = physical_base + len;
|
||||
while (address < end) : (address += page_size) {
|
||||
mapPage(pml4, danos.physicalToVirtual(address), address, flags);
|
||||
}
|
||||
}
|
||||
|
||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(mm.regions)))[0..mm.len];
|
||||
}
|
||||
|
||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
||||
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
|
||||
fn enableNx() void {
|
||||
const efer_msr = 0xC0000080;
|
||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
|
||||
alloc_frame = allocFrame;
|
||||
free_frame = freeFrame;
|
||||
enableNx();
|
||||
const pml4 = allocTable();
|
||||
|
||||
// 1. All RAM in the physmap (physicalToVirtual(physical)) RW + NX. No identity/low-half
|
||||
// mapping: the low half belongs to user space. MMIO is skipped here and
|
||||
// mapped on demand (mapMmio) or explicitly below.
|
||||
for (regions(boot_information.memory_map)) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||
}
|
||||
|
||||
// 2. Physmap windows for the framebuffer and the Local APIC (device memory
|
||||
// the kernel touches directly), RW + NX.
|
||||
const fb = boot_information.framebuffer;
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapPage(pml4, danos.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||
// to their low physical load addresses with real ELF permissions: code
|
||||
// R+X, rodata R, data R+W+NX. This is the W^X guarantee.
|
||||
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||
var flags: u64 = present;
|
||||
if (seg.flags & pf_w != 0) flags |= writable;
|
||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
||||
var off: u64 = 0;
|
||||
while (off < seg.pages * page_size) : (off += page_size) {
|
||||
mapPage(pml4, seg.virtual + off, seg.physical + off, flags);
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Pre-create every higher-half PML4 entry (an empty PDPT where none
|
||||
// exists yet), so the whole kernel half is a fixed set of top-level
|
||||
// slots. A process address space (M3) then shares the kernel half simply
|
||||
// by copying PML4[256..512) — growth beneath these slots (heap, on-demand
|
||||
// MMIO) propagates to every address space because they share the PDPTs.
|
||||
for (256..512) |i| {
|
||||
const e = &tableAt(pml4)[i];
|
||||
if (e.* & present == 0) e.* = allocTable() | present | writable;
|
||||
}
|
||||
|
||||
kernel_pml4 = pml4;
|
||||
asm volatile ("mov %[pml4], %%cr3"
|
||||
:
|
||||
: [pml4] "r" (pml4),
|
||||
: .{ .memory = true }
|
||||
);
|
||||
on_own_tables = true; // now on the kernel's physmap (covers all RAM)
|
||||
init_done = true; // the kernel half is fixed from here
|
||||
}
|
||||
|
||||
/// The kernel's own top-level page table (physical). Every kernel task and every
|
||||
/// per-process address space shares this table's higher half.
|
||||
pub fn kernelPml4() u64 {
|
||||
return kernel_pml4;
|
||||
}
|
||||
|
||||
/// Load CR3 (switch the active address space). `pml4` is a physical frame.
|
||||
pub fn loadCr3(pml4: u64) void {
|
||||
asm volatile ("mov %[pml4], %%cr3"
|
||||
:
|
||||
: [pml4] "r" (pml4),
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
||||
pub fn map(virtual: u64, physical: u64, writable_page: bool) void {
|
||||
var flags: u64 = present | no_execute;
|
||||
if (writable_page) flags |= writable;
|
||||
mapPage(kernel_pml4, virtual, physical, flags);
|
||||
invalidate(virtual);
|
||||
}
|
||||
|
||||
/// Map a device MMIO range into the physmap and return the virtual address to
|
||||
/// use for it (physicalToVirtual(physical)). The single way the kernel (and the device
|
||||
/// layer, via the HAL) reaches memory-mapped registers once the identity map is
|
||||
/// gone: physmap pages are RW + NX, so a driver never executes device memory.
|
||||
/// Idempotent for already-mapped ranges. `len` 0 maps one page.
|
||||
pub fn mapMmio(physical: u64, len: u64, writable_page: bool) u64 {
|
||||
var flags: u64 = present | no_execute;
|
||||
if (writable_page) flags |= writable;
|
||||
const first = physical & ~@as(u64, page_size - 1);
|
||||
const last = physical + (if (len == 0) 1 else len) - 1;
|
||||
var address = first;
|
||||
while (address <= (last & ~@as(u64, page_size - 1))) : (address += page_size) {
|
||||
const virtual = danos.physicalToVirtual(address);
|
||||
mapPage(kernel_pml4, virtual, address, flags);
|
||||
invalidate(virtual);
|
||||
}
|
||||
return danos.physicalToVirtual(physical);
|
||||
}
|
||||
|
||||
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
|
||||
/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves
|
||||
/// existing entries untouched. Only used under user-exclusive virtual ranges, so
|
||||
/// no kernel mapping's protection is widened (the leaf still governs).
|
||||
fn descendUser(entry: *u64) u64 {
|
||||
const table = descend(entry);
|
||||
entry.* |= user;
|
||||
return table;
|
||||
}
|
||||
|
||||
/// Map one 4 KiB page `virtual` -> `physical` accessible from ring 3. W^X is the
|
||||
/// caller's contract: code pages are read-only + executable, data pages are
|
||||
/// writable + no-execute. `virtual` must lie in a user-exclusive region (see
|
||||
/// `descendUser`).
|
||||
pub fn mapUser(virtual: u64, physical: u64, writable_page: bool, executable: bool) void {
|
||||
mapUserInto(kernel_pml4, virtual, physical, writable_page, executable);
|
||||
}
|
||||
|
||||
/// Map a ring-3-accessible page into the address space rooted at `pml4` (which
|
||||
/// may be a process's own table or the kernel's). W^X is the caller's contract.
|
||||
pub fn mapUserInto(pml4: u64, virtual: u64, physical: u64, writable_page: bool, executable: bool) void {
|
||||
var flags: u64 = present | user;
|
||||
if (writable_page) flags |= writable;
|
||||
if (!executable) flags |= no_execute;
|
||||
const pml4e = &tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
const pdpt = descendUser(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = descendUser(pdpte);
|
||||
const pde = &tableAt(pd)[(virtual >> 21) & 0x1FF];
|
||||
const pt = descendUser(pde);
|
||||
tableAt(pt)[(virtual >> 12) & 0x1FF] = (physical & address_mask) | flags;
|
||||
invalidate(virtual);
|
||||
}
|
||||
|
||||
/// Map a device MMIO window `[physical, physical+len)` into the user (low) half of the
|
||||
/// address space rooted at `pml4`, page by page. Unlike `mapUserInto` these pages
|
||||
/// are **strong-uncacheable** (PCD|PWT — device registers must not be cached) and
|
||||
/// carry the `device_grant` bit so teardown does not return the MMIO frames to the
|
||||
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virtual` in a
|
||||
/// user-exclusive range (PML4[225]). Both `virtual` and `physical` are page-aligned by
|
||||
/// the caller; a sub-page `physical` offset is the caller's to re-apply.
|
||||
pub fn mapUserDeviceInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
|
||||
const first = physical & ~@as(u64, page_size - 1);
|
||||
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||
var off: u64 = 0;
|
||||
while (first + off <= last) : (off += page_size) {
|
||||
const v = virtual + off;
|
||||
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
|
||||
const pdpt = descendUser(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
|
||||
const pd = descendUser(pdpte);
|
||||
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
|
||||
const pt = descendUser(pde);
|
||||
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
|
||||
invalidate(v);
|
||||
}
|
||||
}
|
||||
|
||||
/// Create a new address space: a fresh PML4 with an empty user half and the
|
||||
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
||||
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
||||
/// the kernel half propagates to every address space). Returns the physical
|
||||
/// PML4, or null if out of frames.
|
||||
pub fn createAddressSpace() ?u64 {
|
||||
const pml4 = alloc_frame() orelse return null;
|
||||
const t = tableAt(pml4);
|
||||
@memset(t[0..256], 0); // empty user half
|
||||
@memcpy(t[256..512], tableAt(kernel_pml4)[256..512]); // shared kernel half
|
||||
return pml4;
|
||||
}
|
||||
|
||||
/// Tear down an address space created by `createAddressSpace`: free every frame
|
||||
/// and table in the user half [0..256), then the PML4 itself. The shared kernel
|
||||
/// half [256..512) is never touched. The caller must not be running on `pml4`.
|
||||
pub fn destroyAddressSpace(pml4: u64) void {
|
||||
const t = tableAt(pml4);
|
||||
for (0..256) |i| {
|
||||
if (t[i] & present != 0) freeSubtree(t[i] & address_mask, 3); // PDPT level
|
||||
}
|
||||
free_frame(pml4);
|
||||
}
|
||||
|
||||
/// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At
|
||||
/// level 1 the entries are leaf data frames; above, they are child tables.
|
||||
fn freeSubtree(physical: u64, level: u32) void {
|
||||
const t = tableAt(physical);
|
||||
for (t) |e| {
|
||||
if (e & present == 0) continue;
|
||||
if (level > 1) {
|
||||
freeSubtree(e & address_mask, level - 1);
|
||||
} else if (e & device_grant == 0) {
|
||||
// A device-grant leaf points at MMIO, not RAM — returning it to the
|
||||
// frame allocator would corrupt the pool. Only reclaim real RAM.
|
||||
free_frame(e & address_mask);
|
||||
}
|
||||
}
|
||||
free_frame(physical); // page-table frames are always real RAM
|
||||
}
|
||||
|
||||
/// Whether `virtual` is currently mapped **executable** — present with the NX bit
|
||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
||||
pub fn isExecutable(virtual: u64) bool {
|
||||
const pml4e = tableAt(kernel_pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return false;
|
||||
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return false;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return false;
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return false;
|
||||
return pte & no_execute == 0;
|
||||
}
|
||||
|
||||
/// Make an already-identity-mapped RAM page **executable** (clear its NX bit),
|
||||
/// leaving it present and writable. The blanket RAM mapping is NX for W^X, but the
|
||||
/// application processors fetch the AP trampoline from a low RAM page under paging —
|
||||
/// so that one page must be executable. A deliberate, temporary W^X exception for a
|
||||
/// single bring-up page; the caller frees it once every AP is up.
|
||||
pub fn setExecutable(physical: u64) void {
|
||||
mapPage(kernel_pml4, physical, physical, present | writable); // note: no no_execute
|
||||
invalidate(physical);
|
||||
}
|
||||
|
||||
/// Remove a mapping and flush it from the TLB.
|
||||
pub fn unmap(virtual: u64) void {
|
||||
unmapInto(kernel_pml4, virtual);
|
||||
}
|
||||
|
||||
/// Remove a mapping from the address space rooted at `pml4` (a process's own
|
||||
/// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE —
|
||||
/// the intermediate tables and any frame the PTE pointed at are left to the
|
||||
/// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables).
|
||||
pub fn unmapInto(pml4: u64, virtual: u64) void {
|
||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return;
|
||||
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return;
|
||||
tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF] = 0;
|
||||
invalidate(virtual);
|
||||
}
|
||||
|
||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
||||
/// needs the frame behind a user vaddr to free it).
|
||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return null;
|
||||
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return null;
|
||||
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return null;
|
||||
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return null;
|
||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||
}
|
||||
|
||||
fn invalidate(virtual: u64) void {
|
||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||
// inline asm won't form directly, so stage the address in a register first.
|
||||
asm volatile (
|
||||
\\mov %[v], %%rax
|
||||
\\invlpg (%%rax)
|
||||
:
|
||||
: [v] "r" (virtual),
|
||||
: .{ .rax = true, .memory = true }
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
//! Per-CPU data reached through the GS segment base. The GS base holds a pointer
|
||||
//! to this core's `ArchitecturePerCpu`, so kernel code gets the running core's block with
|
||||
//! a single MSR read (`scheduler()`) and the system_call entry stub gets its kernel stack
|
||||
//! with a `%gs`-relative load (no usable stack yet at that point).
|
||||
//!
|
||||
//! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds
|
||||
//! the user's own GS (which ring 3 may set freely), and this pointer lives in the
|
||||
//! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the
|
||||
//! system_call stub and the conditional swapgs in isr_common) brings it back, and
|
||||
//! every ring-0 -> ring-3 exit swaps it away. Because the very first ring
|
||||
//! transition is always an exit (the kernel starts in ring 0), the swap pairs
|
||||
//! keep the invariant without seeding KERNEL_GS_BASE. `scheduler()` is therefore
|
||||
//! valid in any ring-0 context and never sees a user-controlled base.
|
||||
|
||||
const std = @import("std");
|
||||
const io = @import("io.zig");
|
||||
const parameters = @import("parameters");
|
||||
|
||||
const ia32_gs_base = 0xC000_0101;
|
||||
|
||||
/// Layout is load-bearing: the system_call entry stub in isr.s reaches `kernel_rsp`
|
||||
/// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below
|
||||
/// pin the offsets.
|
||||
pub const ArchitecturePerCpu = extern struct {
|
||||
kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for system_call entry (== TSS.rsp0)
|
||||
scratch: u64 = 0, // %gs:8 — stashes the user rsp during system_call entry
|
||||
scheduler: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns)
|
||||
};
|
||||
|
||||
comptime {
|
||||
std.debug.assert(@offsetOf(ArchitecturePerCpu, "kernel_rsp") == 0);
|
||||
std.debug.assert(@offsetOf(ArchitecturePerCpu, "scratch") == 8);
|
||||
}
|
||||
|
||||
var blocks = [_]ArchitecturePerCpu{.{}} ** parameters.maximum_cpus;
|
||||
|
||||
/// Publish core `index`'s per-CPU block: record the scheduler pointer and point
|
||||
/// the GS base at the block. Called once per core during bring-up, after the GDT
|
||||
/// is loaded (a GS *selector* reload would clobber the base).
|
||||
pub fn setLocal(index: usize, scheduler_ptr: usize) void {
|
||||
blocks[index].scheduler = scheduler_ptr;
|
||||
io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index]));
|
||||
}
|
||||
|
||||
/// The scheduler pointer for the running core (via the GS base). Valid in any
|
||||
/// ring-0 context under the swapgs discipline.
|
||||
pub fn scheduler() usize {
|
||||
return @as(*const ArchitecturePerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).scheduler;
|
||||
}
|
||||
|
||||
/// Record core `index`'s kernel stack top, used by the system_call entry stub to
|
||||
/// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it
|
||||
/// switches to a user task.
|
||||
pub fn setKernelRsp(index: usize, top: usize) void {
|
||||
blocks[index].kernel_rsp = top;
|
||||
}
|
||||
|
||||
// Fast-system_call MSRs.
|
||||
const ia32_efer = 0xC000_0080;
|
||||
const ia32_star = 0xC000_0081;
|
||||
const ia32_lstar = 0xC000_0082;
|
||||
const ia32_sfmask = 0xC000_0084;
|
||||
|
||||
/// Enable the `system_call`/`sysret` fast path on this core (BSP and each AP). EFER.SCE
|
||||
/// turns the instructions on; STAR sets the selectors system_call/sysret load; LSTAR
|
||||
/// is the entry stub (isr.s); SFMASK clears RFLAGS bits on entry (notably IF —
|
||||
/// the handler runs with interrupts off, like the int-gate path). The GDT is laid
|
||||
/// out (kernel code 0x08, then user data 0x18 / code 0x20) precisely so these line
|
||||
/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8
|
||||
/// with RPL forced to 3, so base 0x10 gives CS 0x23 (user code|3) and SS 0x1B.
|
||||
pub fn initSystemCall() void {
|
||||
io.wrmsr(ia32_efer, io.rdmsr(ia32_efer) | 1); // SCE
|
||||
io.wrmsr(ia32_star, (@as(u64, 0x08) << 32) | (@as(u64, 0x10) << 48));
|
||||
const entry = @extern(*const anyopaque, .{ .name = "syscall_entry" });
|
||||
io.wrmsr(ia32_lstar, @intFromPtr(entry));
|
||||
io.wrmsr(ia32_sfmask, 0x4_0700); // clear IF, TF, DF, AC on entry
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
//! Serial console (16550-compatible UART) — the kernel's machine-readable output
|
||||
//! channel. Unlike the framebuffer console, serial text can be captured to a file
|
||||
//! by QEMU (`-serial file:...`), which is what the test harness asserts on.
|
||||
//!
|
||||
//! The UART defaults to the legacy PC COM1 at I/O port `0x3F8`, but a UEFI Class 3
|
||||
//! (legacy-free) machine may have no COM1 — or its debug UART somewhere else, and
|
||||
//! reachable via MMIO rather than port I/O. So the location is a runtime value:
|
||||
//! `reconfigure` repoints it once ACPI's SPCR table has been read. Early boot logs
|
||||
//! optimistically to COM1 (harmless if absent); the framebuffer console is the
|
||||
//! always-present log.
|
||||
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// How the UART registers are reached: legacy I/O ports or memory-mapped.
|
||||
const Access = enum { port, mmio };
|
||||
|
||||
var access: Access = .port;
|
||||
var base: u64 = 0x3F8; // COM1
|
||||
|
||||
fn portOut(p: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[p]"
|
||||
:
|
||||
: [value] "{al}" (value),
|
||||
[p] "{dx}" (p),
|
||||
);
|
||||
}
|
||||
|
||||
fn portIn(p: u16) u8 {
|
||||
return asm volatile ("inb %[p], %[value]"
|
||||
: [value] "={al}" (-> u8),
|
||||
: [p] "{dx}" (p),
|
||||
);
|
||||
}
|
||||
|
||||
/// Read UART register `off` through the active access method.
|
||||
fn register(off: u64) u8 {
|
||||
if (access == .mmio) return @as(*volatile u8, @ptrFromInt(base + off)).*;
|
||||
return portIn(@intCast(base + off));
|
||||
}
|
||||
|
||||
/// Write UART register `off` through the active access method.
|
||||
fn setRegister(off: u64, value: u8) void {
|
||||
if (access == .mmio) {
|
||||
@as(*volatile u8, @ptrFromInt(base + off)).* = value;
|
||||
} else {
|
||||
portOut(@intCast(base + off), value);
|
||||
}
|
||||
}
|
||||
|
||||
/// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything
|
||||
/// else; it has no dependencies, and is a harmless no-op if the port is absent.
|
||||
pub fn init() void {
|
||||
setRegister(1, 0x00); // disable interrupts
|
||||
setRegister(3, 0x80); // enable DLAB (set baud divisor)
|
||||
setRegister(0, 0x03); // divisor low: 38400 baud
|
||||
setRegister(1, 0x00); // divisor high
|
||||
setRegister(3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||
setRegister(2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||
setRegister(4, 0x0B); // RTS/DSR set
|
||||
}
|
||||
|
||||
/// Point the console at the UART ACPI's SPCR table names (MMIO or I/O port) and
|
||||
/// re-run the UART setup there. Called after discovery when an SPCR entry exists.
|
||||
pub fn reconfigure(is_mmio: bool, address: u64) void {
|
||||
access = if (is_mmio) .mmio else .port;
|
||||
// An MMIO UART is reached through the physmap; an I/O-port UART keeps its
|
||||
// port number unchanged.
|
||||
base = if (is_mmio) paging.mapMmio(address, 0x100, true) else address;
|
||||
init();
|
||||
}
|
||||
|
||||
fn writeByte(c: u8) void {
|
||||
// Wait for the transmit-holding register to empty — but bounded, so an absent
|
||||
// UART (whose line-status register reads back as 0x00) can't hang the kernel.
|
||||
var guard: u32 = 0;
|
||||
while (register(5) & 0x20 == 0 and guard < 100_000) : (guard += 1) {}
|
||||
setRegister(0, c);
|
||||
}
|
||||
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
for (bytes) |c| {
|
||||
if (c == '\n') writeByte('\r');
|
||||
writeByte(c);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
//! Application-processor (AP) bring-up: waking the cores the firmware left parked.
|
||||
//!
|
||||
//! The firmware starts only the bootstrap processor (BSP); the others sit idle until
|
||||
//! the kernel wakes them with an INIT–SIPI–SIPI sequence (Intel SDM Vol.3, "MP
|
||||
//! Initialization"). A woken core begins in 16-bit real mode at a low physical page,
|
||||
//! runs the [trampoline](trampoline.s) up into 64-bit long mode, and lands in
|
||||
//! `apEntry` here. This module copies the trampoline into place, patches its
|
||||
//! per-AP parameters, drives the wake IPIs, and waits for each core to report in.
|
||||
//!
|
||||
//! Cores are brought up **one at a time**: a single trampoline page and parameter
|
||||
//! block are reused, so the BSP patches, wakes, and waits for one AP before the
|
||||
//! next. That also lets `apEntry` pick up its dense CPU index from a plain global.
|
||||
//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic
|
||||
//! scheduler entry and joins the run loop — mechanism here, policy there.
|
||||
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const apic = @import("apic.zig");
|
||||
const paging = @import("paging.zig");
|
||||
const pcpu = @import("per-cpu.zig");
|
||||
|
||||
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
|
||||
/// path doesn't depend on cpu.zig and risk an import cycle).
|
||||
const ia32_gs_base = 0xC000_0101;
|
||||
const page_size = 0x1000;
|
||||
|
||||
/// Physical address of the low (<1 MiB) frame reserved for the trampoline. Held for
|
||||
/// the life of the system so any core can be (re)woken on demand — a retry, or a
|
||||
/// future power manager bringing a core back online. The frame is kept **inert**
|
||||
/// between wakes (zeroed and non-executable) and only armed for the brief moment a
|
||||
/// core is actually climbing. Its low 20 bits are zero, so `physical >> 12` is the SIPI
|
||||
/// vector.
|
||||
var tramp_physical: u64 = 0;
|
||||
|
||||
/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own
|
||||
/// bring-up. The BSP clears it before each wake and polls it afterwards — a simple
|
||||
/// one-at-a-time handshake (only one AP is being started at any moment).
|
||||
var ap_alive: u32 = 0;
|
||||
|
||||
/// The dense CPU index of the AP currently being started. Set by the BSP before the
|
||||
/// wake, read by `apEntry` (safe because bring-up is strictly one core at a time).
|
||||
var boot_index: usize = 0;
|
||||
|
||||
/// The generic scheduler entry a woken core jumps to once its architecture state is up. Set
|
||||
/// by the kernel via `setSecondaryEntry`; never returns.
|
||||
var secondary_entry: ?*const fn () callconv(.c) noreturn = null;
|
||||
|
||||
/// Register the generic entry an AP calls once its per-CPU tables/LAPIC/timer are up.
|
||||
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
||||
secondary_entry = entry;
|
||||
}
|
||||
|
||||
/// Test hook: force the next `n` wake attempts to fail (skipping the actual
|
||||
/// INIT-SIPI-SIPI), so the retry path can be exercised deterministically. Zero in
|
||||
/// normal operation — the smp-retry test arms it via `architecture.testFailNextWakes`.
|
||||
var fail_next_wakes: u32 = 0;
|
||||
pub fn testFailNextWakes(n: u32) void {
|
||||
fail_next_wakes = n;
|
||||
}
|
||||
|
||||
/// Record the reserved low frame the trampoline uses. Call once at boot. The frame
|
||||
/// starts inert (identity-mapped RW+NX like all RAM); each wake arms it and disarms
|
||||
/// it again, so it's only ever executable while a core is climbing.
|
||||
pub fn setTrampolinePage(physical: u64) void {
|
||||
tramp_physical = physical;
|
||||
}
|
||||
|
||||
/// The reserved trampoline frame (0 if SMP bring-up never ran). Exposed so a test
|
||||
/// can verify it's inert — zeroed and non-executable — when dormant.
|
||||
pub fn trampolinePage() u64 {
|
||||
return tramp_physical;
|
||||
}
|
||||
|
||||
/// Arm the trampoline for a wake: make its page executable (W^X exception for the
|
||||
/// duration of the climb) and copy the blob in.
|
||||
fn arm() void {
|
||||
// The AP executes this page at its physical address (identity) while it
|
||||
// climbs from real to long mode, so it needs a low identity mapping that is
|
||||
// executable — the one deliberate, transient W^X exception. The BSP writes
|
||||
// the blob into the frame through the physmap.
|
||||
paging.setExecutable(tramp_physical);
|
||||
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
|
||||
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
|
||||
const len = @intFromPtr(end) - @intFromPtr(start);
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
|
||||
@memcpy(destination[0..len], start[0..len]);
|
||||
}
|
||||
|
||||
/// Disarm after a wake: wipe the page through the physmap and remove its low
|
||||
/// identity mapping, so no executable code (nor any stale bytes, nor any
|
||||
/// low-half mapping) lingers between wakes. Safe once the woken core has
|
||||
/// reported in — it's long past the trampoline by then, in the kernel image; a
|
||||
/// core that never answered is dead and can't be mid-climb.
|
||||
fn disarm() void {
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
|
||||
@memset(destination[0..page_size], 0);
|
||||
paging.unmap(tramp_physical); // drop the transient low identity mapping
|
||||
}
|
||||
|
||||
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
|
||||
/// base plus the field's offset within it (a same-section symbol difference). The
|
||||
/// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86
|
||||
/// tolerates unaligned stores, so we don't force layout constraints on the asm.
|
||||
fn param(comptime name: []const u8) *align(1) volatile u64 {
|
||||
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
|
||||
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
|
||||
return @ptrFromInt(danos.physicalToVirtual(tramp_physical + (sym - start)));
|
||||
}
|
||||
|
||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
|
||||
/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive. This
|
||||
/// is one self-contained attempt: it arms the trampoline, drives INIT–SIPI–SIPI, and
|
||||
/// disarms again before returning — so it's safe to call repeatedly (a retry, or a
|
||||
/// power manager re-waking a core; the INIT resets a core that was wedged). Returns
|
||||
/// false if the core doesn't report in within the timeout (left parked, no harm to
|
||||
/// the running system). `cr3` is the kernel page tables the AP adopts. Precondition:
|
||||
/// `setTrampolinePage` has run.
|
||||
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool {
|
||||
// The trampoline loads CR3 with a 32-bit `movl` before it reaches long mode,
|
||||
// so the page-table root must be addressable in 32 bits.
|
||||
if (cr3 >= (1 << 32)) @panic("smp: kernel page tables above 4 GiB");
|
||||
arm();
|
||||
defer disarm();
|
||||
|
||||
if (fail_next_wakes > 0) { // test hook: simulate a core missing this attempt
|
||||
fail_next_wakes -= 1;
|
||||
return false;
|
||||
}
|
||||
|
||||
boot_index = index;
|
||||
param("ap_tramp_cr3").* = cr3;
|
||||
param("ap_tramp_stack").* = stack_top;
|
||||
param("ap_tramp_entry").* = @intFromPtr(&apEntry);
|
||||
param("ap_tramp_percpu").* = percpu;
|
||||
|
||||
@atomicStore(u32, &ap_alive, 0, .seq_cst);
|
||||
|
||||
const vector: u8 = @intCast(tramp_physical >> 12);
|
||||
apic.sendInit(apic_id);
|
||||
delayMicros(10_000); // 10 ms INIT settle
|
||||
apic.sendStartup(apic_id, vector);
|
||||
delayMicros(200);
|
||||
apic.sendStartup(apic_id, vector);
|
||||
|
||||
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
|
||||
const deadline = apic.millis() + 100;
|
||||
while (apic.millis() < deadline) {
|
||||
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
|
||||
asm volatile ("pause");
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Busy-wait `us` microseconds against the calibrated TSC clock (the AP wake happens
|
||||
/// after the timer is up, so the clock is available).
|
||||
fn delayMicros(us: u64) void {
|
||||
const start = apic.micros();
|
||||
while (apic.micros() - start < us) asm volatile ("pause");
|
||||
}
|
||||
|
||||
/// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU
|
||||
/// pointer in RDI. Brings up this core's own descriptor tables, LAPIC and timer,
|
||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
const cpu = boot_index;
|
||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||
idt.loadOnThisCpu(); // the shared IDT
|
||||
pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload
|
||||
pcpu.initSystemCall(); // enable system_call/sysret on this core
|
||||
|
||||
apic.initSecondary(); // software-enable this core's LAPIC
|
||||
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
|
||||
|
||||
@atomicStore(u32, &ap_alive, 1, .release); // "architecture state up" — BSP is polling this
|
||||
|
||||
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
||||
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
# AP trampoline: brings a waking application processor from the 16-bit real mode it
|
||||
# starts in (after INIT-SIPI-SIPI) up through protected mode into 64-bit long mode,
|
||||
# then jumps to the Zig AP entry (arch/x86_64/smp.zig:apEntry).
|
||||
#
|
||||
# A STARTUP IPI vectors a core to physical address `vector << 12` in real mode, so
|
||||
# this blob is copied to a low (<1 MiB) page and started there; at entry CS = that
|
||||
# page >> 4 and IP = 0. It is fully **position-independent**: it derives its own
|
||||
# linear base (CS << 4) into EBX and addresses every internal datum as
|
||||
# `(label - ap_trampoline_start)(%ebx)` — a difference of two symbols in the same
|
||||
# section, which the assembler folds to a constant page offset no matter where the
|
||||
# blob was linked or copied to. The BSP patches the parameter block (CR3, stack,
|
||||
# entry, per-CPU pointer) before each wake; see arch/x86_64/smp.zig.
|
||||
#
|
||||
# It lives in .rodata (not .text): it is data to be copied out and executed
|
||||
# elsewhere, never run at its link address, so it must not be a normal code segment.
|
||||
|
||||
.section .rodata
|
||||
.balign 16
|
||||
.code16
|
||||
.global ap_trampoline_start
|
||||
ap_trampoline_start:
|
||||
cli
|
||||
cld
|
||||
|
||||
# Linear base of this page (CS << 4) into EBX; all data is addressed off it.
|
||||
xorl %eax, %eax
|
||||
mov %cs, %ax
|
||||
shll $4, %eax
|
||||
movl %eax, %ebx
|
||||
|
||||
mov %cs, %ax # DS = CS, so we address our data as DS:(label - start):
|
||||
mov %ax, %ds # the segment base (CS<<4) already supplies the page base,
|
||||
# so data operands use the page *offset*, not EBX.
|
||||
|
||||
# Relocate the pointers whose absolute (linear) targets depend on where we were
|
||||
# copied: the GDT base and the two far-jump targets = EBX + their page offsets.
|
||||
# EBX supplies the base for the *value* (via leal); the store address is DS-rel.
|
||||
leal (gdt32 - ap_trampoline_start)(%ebx), %eax
|
||||
movl %eax, gdtr32_base - ap_trampoline_start
|
||||
leal (prot_entry - ap_trampoline_start)(%ebx), %eax
|
||||
movl %eax, jmp32_off - ap_trampoline_start
|
||||
leal (long_entry - ap_trampoline_start)(%ebx), %eax
|
||||
movl %eax, jmp64_off - ap_trampoline_start
|
||||
|
||||
lgdtl gdtr32 - ap_trampoline_start
|
||||
|
||||
movl %cr0, %eax # enter protected mode (CR0.PE)
|
||||
orl $1, %eax
|
||||
movl %eax, %cr0
|
||||
|
||||
ljmpl *(jmp32_ptr - ap_trampoline_start) # -> prot_entry, CS = 0x08
|
||||
|
||||
.code32
|
||||
prot_entry:
|
||||
movw $0x10, %ax # flat 32-bit data segments
|
||||
movw %ax, %ds
|
||||
movw %ax, %es
|
||||
movw %ax, %ss
|
||||
movw %ax, %fs
|
||||
movw %ax, %gs
|
||||
|
||||
# CR4: PAE (required for long mode) + OSFXSR/OSXMMEXCPT. The kernel is built with
|
||||
# SSE (part of the x86_64 baseline), and the compiler emits SSE for things as
|
||||
# ordinary as a struct copy — without OSFXSR those instructions #UD. The BSP got
|
||||
# these bits from UEFI; an AP starts fresh, so we must set them ourselves.
|
||||
movl %cr4, %eax
|
||||
orl $((1 << 5) | (1 << 9) | (1 << 10)), %eax
|
||||
movl %eax, %cr4
|
||||
|
||||
# CR0: clear EM (no x87 emulation) and set MP, so SSE/x87 don't fault.
|
||||
movl %cr0, %eax
|
||||
andl $~(1 << 2), %eax # ~EM
|
||||
orl $(1 << 1), %eax # MP
|
||||
movl %eax, %cr0
|
||||
|
||||
movl (param_cr3 - ap_trampoline_start)(%ebx), %eax # kernel page tables
|
||||
movl %eax, %cr3
|
||||
|
||||
movl $0xC0000080, %ecx # EFER: long mode enable (LME) + NX enable (NXE, since
|
||||
rdmsr # the kernel's PTEs set the NX bit)
|
||||
orl $((1 << 8) | (1 << 11)), %eax
|
||||
wrmsr
|
||||
|
||||
movl %cr0, %eax # paging on (CR0.PG) — now in long mode (compat sub-mode)
|
||||
orl $(1 << 31), %eax
|
||||
movl %eax, %cr0
|
||||
|
||||
ljmpl *(jmp64_ptr - ap_trampoline_start)(%ebx) # -> long_entry, CS = 0x18 (L=1)
|
||||
|
||||
.code64
|
||||
long_entry:
|
||||
movw $0x10, %ax # sane flat data segments
|
||||
movw %ax, %ds
|
||||
movw %ax, %es
|
||||
movw %ax, %ss
|
||||
|
||||
# RBX = EBX (zero-extended) = page base. Load our stack and per-CPU pointer, then
|
||||
# call the Zig entry — which runs from the kernel image and never returns.
|
||||
movq (param_stack - ap_trampoline_start)(%rbx), %rsp
|
||||
movq (param_percpu - ap_trampoline_start)(%rbx), %rdi # SysV arg 0
|
||||
movq (param_entry - ap_trampoline_start)(%rbx), %rax
|
||||
callq *%rax
|
||||
1: hlt # unreachable; guard against a stray return
|
||||
jmp 1b
|
||||
|
||||
# --- data: GDT, far pointers, and the BSP-patched parameter block -----------
|
||||
.balign 8
|
||||
gdt32:
|
||||
.quad 0x0000000000000000 # 0x00 null
|
||||
.quad 0x00CF9A000000FFFF # 0x08 32-bit code (G, D, present, exec/read)
|
||||
.quad 0x00CF92000000FFFF # 0x10 data (valid in 32- and 64-bit)
|
||||
.quad 0x00AF9A000000FFFF # 0x18 64-bit code (L=1)
|
||||
gdt32_end:
|
||||
|
||||
gdtr32:
|
||||
.word gdt32_end - gdt32 - 1
|
||||
gdtr32_base:
|
||||
.long 0 # patched (16-bit code): linear base of gdt32
|
||||
|
||||
jmp32_ptr: # indirect far-jump operand: offset then selector
|
||||
jmp32_off:
|
||||
.long 0 # patched: linear address of prot_entry
|
||||
.word 0x08 # 32-bit code selector
|
||||
|
||||
jmp64_ptr:
|
||||
jmp64_off:
|
||||
.long 0 # patched: linear address of long_entry
|
||||
.word 0x18 # 64-bit code selector
|
||||
|
||||
# The parameter block, filled in by the BSP (smp.zig) before each STARTUP IPI. Global
|
||||
# so the Zig side can locate each field as (symbol - ap_trampoline_start).
|
||||
.global ap_tramp_cr3
|
||||
.global ap_tramp_stack
|
||||
.global ap_tramp_entry
|
||||
.global ap_tramp_percpu
|
||||
param_cr3:
|
||||
ap_tramp_cr3:
|
||||
.quad 0 # kernel PML4 physical address (CR3)
|
||||
param_stack:
|
||||
ap_tramp_stack:
|
||||
.quad 0 # top of this AP's kernel stack
|
||||
param_entry:
|
||||
ap_tramp_entry:
|
||||
.quad 0 # address of apEntry (the Zig AP entry)
|
||||
param_percpu:
|
||||
ap_tramp_percpu:
|
||||
.quad 0 # this AP's per-CPU pointer (goes in GS base)
|
||||
|
||||
.global ap_trampoline_end
|
||||
ap_trampoline_end:
|
||||
@@ -0,0 +1,87 @@
|
||||
//! Task State Segment and its interrupt stacks. In long mode the TSS has two
|
||||
//! jobs. First, the Interrupt Stack Table: an IDT gate can name an IST entry,
|
||||
//! and the CPU switches to that stack when the exception fires — no matter how
|
||||
//! broken the interrupted stack was. We use IST1 for the double-fault handler,
|
||||
//! so a fault that happens *because* the current stack is unusable still lands
|
||||
//! on solid ground instead of triple-faulting. Second, rsp0: the kernel stack
|
||||
//! the CPU switches to when an interrupt arrives from ring 3 (published by the
|
||||
//! user-mode entry path via `rsp0Ptr`).
|
||||
//!
|
||||
//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at
|
||||
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
|
||||
//! indexed by CPU number; slot 0 is the BSP.
|
||||
|
||||
const parameters = @import("parameters");
|
||||
const gdt = @import("gdt.zig");
|
||||
|
||||
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
||||
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
|
||||
const Tss = packed struct {
|
||||
reserved0: u32 = 0,
|
||||
rsp0: u64 = 0,
|
||||
rsp1: u64 = 0,
|
||||
rsp2: u64 = 0,
|
||||
reserved1: u64 = 0,
|
||||
ist1: u64 = 0,
|
||||
ist2: u64 = 0,
|
||||
ist3: u64 = 0,
|
||||
ist4: u64 = 0,
|
||||
ist5: u64 = 0,
|
||||
ist6: u64 = 0,
|
||||
ist7: u64 = 0,
|
||||
reserved2: u64 = 0,
|
||||
reserved3: u16 = 0,
|
||||
iomap_base: u16 = 0,
|
||||
};
|
||||
|
||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
||||
pub const double_fault_ist = 1;
|
||||
|
||||
const maximum_cpus = parameters.maximum_cpus;
|
||||
pub const ist_stack_size = parameters.ist_stack_size;
|
||||
|
||||
/// One TSS per core (small — kept static). The IST stacks are 16 KiB each, so only
|
||||
/// the **BSP's** is static: it must exist before the frame allocator does, to catch a
|
||||
/// fault during early boot. Each **AP** gets a heap-allocated IST stack at bring-up
|
||||
/// (after the heap is up), the top of which the BSP records here before waking it —
|
||||
/// so we reserve big stacks only for cores that actually come online.
|
||||
var tss_table = [_]Tss{.{}} ** maximum_cpus;
|
||||
var bsp_ist_stack: [ist_stack_size]u8 align(16) = undefined;
|
||||
var ap_ist_top = [_]usize{0} ** maximum_cpus; // per-AP IST stack top (0 = BSP / not set)
|
||||
|
||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
||||
|
||||
/// Address of core `cpu`'s rsp0 slot — the kernel stack the CPU switches to on a
|
||||
/// ring-3 -> ring-0 interrupt. Computed as base + 4 (rsp0's architectural offset,
|
||||
/// which is why the pointer is only 4-aligned) rather than `&t.rsp0`, which on a
|
||||
/// packed struct would be an unaligned bit-pointer type. The ring-3 entry path
|
||||
/// (enter_user in isr.s) writes the current kernel stack pointer through this
|
||||
/// before dropping to user mode.
|
||||
pub fn rsp0Ptr(cpu: usize) *align(4) u64 {
|
||||
return @ptrFromInt(@intFromPtr(&tss_table[cpu]) + 4);
|
||||
}
|
||||
|
||||
/// Record the top of the IST stack the kernel allocated for AP `cpu`. Called on the
|
||||
/// BSP before waking that core; read by the core's own `setupThisCpu`.
|
||||
pub fn setApIstStack(cpu: usize, top: usize) void {
|
||||
ap_ist_top[cpu] = top;
|
||||
}
|
||||
|
||||
/// Set up core `cpu`'s TSS: point IST1 at its stack (the BSP's static one for core 0,
|
||||
/// the allocated one recorded via `setApIstStack` for an AP), install the TSS
|
||||
/// descriptor into that core's GDT, and load it into the task register. Requires the
|
||||
/// core's GDT to already be loaded (gdt.loadOnThisCpu first).
|
||||
pub fn setupThisCpu(cpu: usize) void {
|
||||
const t = &tss_table[cpu];
|
||||
t.* = .{};
|
||||
t.ist1 = if (cpu == 0) @intFromPtr(&bsp_ist_stack) + ist_stack_size else ap_ist_top[cpu];
|
||||
t.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
||||
gdt.setTssFor(cpu, @intFromPtr(t), @sizeOf(Tss) - 1);
|
||||
load_tr(gdt.tss_selector);
|
||||
}
|
||||
|
||||
/// Set up the bootstrap processor's TSS (slot 0). Requires gdt.init first.
|
||||
pub fn init() void {
|
||||
setupThisCpu(0);
|
||||
}
|
||||
@@ -0,0 +1,158 @@
|
||||
//! A framebuffer text console: draws glyphs from an embedded PSF2 font directly
|
||||
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
||||
//! — just pixels.
|
||||
//!
|
||||
//! This is a **bootstrap** console — a stop-gap so early boot has something on
|
||||
//! screen. The framebuffer is a general graphics surface, *not* inherently a text
|
||||
//! terminal; once the driver machinery exists it becomes a proper graphics device
|
||||
//! driver and this text-grid crutch goes away. It is therefore kept **separate
|
||||
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file,
|
||||
//! while this only paints the handful of user-facing status lines and panics.
|
||||
//!
|
||||
//! The module owns a single console and a `present` flag; `write` is a no-op when
|
||||
//! the firmware handed over no framebuffer (a headless machine), so the kernel
|
||||
//! never assumes a display exists.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
|
||||
/// The one framebuffer console, valid only when `con_present`.
|
||||
var con: Console = undefined;
|
||||
var con_present: bool = false;
|
||||
|
||||
/// Set up the console over `fb`, or mark it absent if there's no usable
|
||||
/// framebuffer. Clears the screen when present.
|
||||
pub fn init(fb: danos.Framebuffer) void {
|
||||
if (!fb.present()) {
|
||||
con_present = false;
|
||||
return;
|
||||
}
|
||||
con = Console.init(fb);
|
||||
if (con.cols == 0 or con.rows == 0) {
|
||||
con_present = false;
|
||||
return;
|
||||
}
|
||||
con.clear();
|
||||
con_present = true;
|
||||
}
|
||||
|
||||
/// Whether an on-screen console is available.
|
||||
pub fn present() bool {
|
||||
return con_present;
|
||||
}
|
||||
|
||||
/// Output sink: draw `bytes` on screen. A no-op when no framebuffer is present,
|
||||
/// so it's always safe to call.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
if (!con_present) return;
|
||||
for (bytes) |c| con.putChar(c);
|
||||
}
|
||||
|
||||
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
|
||||
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
|
||||
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
|
||||
const font = @embedFile("font.psf");
|
||||
const glyph_w = 8;
|
||||
const glyph_h = 16;
|
||||
const glyph_bytes = glyph_h; // 8 pixels wide => 1 byte per row
|
||||
const glyph_data = 32; // PSF2 header size
|
||||
|
||||
pub const Console = struct {
|
||||
fb: danos.Framebuffer,
|
||||
cols: u32,
|
||||
rows: u32,
|
||||
col: u32 = 0,
|
||||
row: u32 = 0,
|
||||
fg: u32 = 0x00c8_c8c8, // light grey
|
||||
bg: u32 = 0x0000_0000, // black
|
||||
|
||||
pub fn init(fb: danos.Framebuffer) Console {
|
||||
// Reach the framebuffer through the physmap, so the pointer stays valid
|
||||
// once the low identity map is gone. The base is mapped by both the
|
||||
// loader's bootstrap tables and paging.init.
|
||||
var mapped = fb;
|
||||
if (fb.base != 0) mapped.base = danos.physicalToVirtual(fb.base);
|
||||
return .{
|
||||
.fb = mapped,
|
||||
.cols = fb.width / glyph_w,
|
||||
.rows = fb.height / glyph_h,
|
||||
};
|
||||
}
|
||||
|
||||
/// Fill the whole screen with the background colour and home the cursor.
|
||||
pub fn clear(self: *Console) void {
|
||||
var y: u32 = 0;
|
||||
while (y < self.fb.height) : (y += 1) self.fillRow(y, self.bg);
|
||||
self.col = 0;
|
||||
self.row = 0;
|
||||
}
|
||||
|
||||
pub fn putChar(self: *Console, ch: u8) void {
|
||||
switch (ch) {
|
||||
'\n' => self.newline(),
|
||||
'\r' => self.col = 0,
|
||||
else => {
|
||||
if (self.col >= self.cols) self.newline();
|
||||
self.drawGlyph(ch, self.col * glyph_w, self.row * glyph_h);
|
||||
self.col += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn newline(self: *Console) void {
|
||||
self.col = 0;
|
||||
if (self.row + 1 >= self.rows) {
|
||||
self.scroll();
|
||||
} else {
|
||||
self.row += 1;
|
||||
}
|
||||
}
|
||||
|
||||
fn drawGlyph(self: *Console, ch: u8, px: u32, py: u32) void {
|
||||
const rows = font[glyph_data + @as(usize, ch) * glyph_bytes ..][0..glyph_bytes];
|
||||
var gy: u32 = 0;
|
||||
while (gy < glyph_h) : (gy += 1) {
|
||||
const bits = rows[gy];
|
||||
var gx: u32 = 0;
|
||||
while (gx < glyph_w) : (gx += 1) {
|
||||
// Leftmost pixel is the high bit.
|
||||
const on = (bits >> @as(u3, @intCast(7 - gx))) & 1 != 0;
|
||||
self.pixel(px + gx, py + gy, if (on) self.fg else self.bg);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Shift the visible text up one glyph row and clear the freed bottom row,
|
||||
/// leaving the cursor on that now-blank last line.
|
||||
fn scroll(self: *Console) void {
|
||||
const visible = self.rows * glyph_h;
|
||||
var y: u32 = 0;
|
||||
while (y + glyph_h < visible) : (y += 1) self.copyRow(y, y + glyph_h);
|
||||
while (y < visible) : (y += 1) self.fillRow(y, self.bg);
|
||||
self.row = self.rows - 1;
|
||||
}
|
||||
|
||||
inline fn rowPtr(self: *Console, y: u32) [*]volatile u32 {
|
||||
const base: [*]volatile u8 = @ptrFromInt(self.fb.base);
|
||||
return @ptrCast(@alignCast(base + y * self.fb.pitch));
|
||||
}
|
||||
|
||||
inline fn pixel(self: *Console, x: u32, y: u32, color: u32) void {
|
||||
self.rowPtr(y)[x] = color;
|
||||
}
|
||||
|
||||
fn fillRow(self: *Console, y: u32, color: u32) void {
|
||||
const row = self.rowPtr(y);
|
||||
var x: u32 = 0;
|
||||
while (x < self.fb.width) : (x += 1) row[x] = color;
|
||||
}
|
||||
|
||||
fn copyRow(self: *Console, destination_y: u32, source_y: u32) void {
|
||||
const destination = self.rowPtr(destination_y);
|
||||
const source = self.rowPtr(source_y);
|
||||
var x: u32 = 0;
|
||||
while (x < self.fb.width) : (x += 1) destination[x] = source[x];
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
//! Device service: the kernel side of user-space driver access. At boot it
|
||||
//! flattens the discovered device tree (src/device) into a stable, id-indexed
|
||||
//! snapshot and a per-device claim table. User drivers enumerate the snapshot,
|
||||
//! claim the device they own, and map its MMIO — the claim is the capability that
|
||||
//! gates `mmio_map`/`irq_bind`, so a process can only ever touch hardware the
|
||||
//! firmware-neutral device tree says it owns.
|
||||
//!
|
||||
//! The table is a **tree**: each entry carries its parent's id. Firmware discovery
|
||||
//! seeds it, and a **bus driver** grows it — a process that has claimed a bus can
|
||||
//! `register` children below it as it enumerates them (USB devices behind a hub, PCI
|
||||
//! functions behind a bridge, comparators inside a timer block).
|
||||
//!
|
||||
//! Registration is where the capability model earns its keep. A `DeviceDescriptor` is, in
|
||||
//! effect, a licence to map physical memory: whoever claims it may `mmio_map` its
|
||||
//! `.memory` resources and `irq_bind` its `.irq` resources. If a bus driver could
|
||||
//! invent arbitrary resources, it would invent one covering the kernel's RAM, claim
|
||||
//! it, and map it. So `register` enforces **containment**: every resource of a child
|
||||
//! must lie inside a resource of the same kind on its parent. A bus driver can only
|
||||
//! ever subdivide what it was already given.
|
||||
|
||||
const std = @import("std");
|
||||
const platform = @import("platform");
|
||||
const danos = @import("danos");
|
||||
|
||||
const maximum_devices = 64;
|
||||
|
||||
/// Cap on children a single parent may have. A zero-resource child (legal — a USB
|
||||
/// device is addressed through its controller, not by MMIO) sidesteps the containment
|
||||
/// check, so without a bound a process that claimed one device could loop
|
||||
/// `device_register` and exhaust the whole table, permanently denying it to every other
|
||||
/// driver. This bounds the blast radius of one claim; a real quota (and a
|
||||
/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md.
|
||||
const maximum_children_per_parent = 16;
|
||||
|
||||
var devices: [maximum_devices]danos.DeviceDescriptor = undefined;
|
||||
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
||||
var count: usize = 0;
|
||||
|
||||
/// Devices discovery found but the table had no room for. Non-zero means the machine
|
||||
/// is bigger than `maximum_devices` and some hardware is simply invisible to drivers —
|
||||
/// which would otherwise be an entirely silent failure. Logged at boot.
|
||||
pub var dropped: usize = 0;
|
||||
|
||||
/// Snapshot the device tree into the flat table. Run once, right after discovery.
|
||||
pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||
count = 0;
|
||||
dropped = 0;
|
||||
for (&claimed) |*c| c.* = null;
|
||||
walk(device_tree.root, danos.no_parent);
|
||||
}
|
||||
|
||||
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
||||
/// assigned it down to its children as their parent.
|
||||
fn walk(node: *platform.Device, parent_id: u64) void {
|
||||
const id = if (node.class == .root) danos.no_parent else record(node, parent_id);
|
||||
var child = node.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) walk(c, id);
|
||||
}
|
||||
|
||||
fn record(node: *platform.Device, parent_id: u64) u64 {
|
||||
if (count >= maximum_devices) {
|
||||
dropped += 1;
|
||||
return danos.no_parent; // children of a dropped node become roots, not orphans
|
||||
}
|
||||
var d = std.mem.zeroes(danos.DeviceDescriptor);
|
||||
d.id = count;
|
||||
d.parent = parent_id;
|
||||
d.class = @intFromEnum(node.class);
|
||||
const h = node.hid();
|
||||
d.hid_len = @min(h.len, d.hid.len);
|
||||
@memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]);
|
||||
const rc = @min(node.resource_count, danos.maximum_device_resources);
|
||||
d.resource_count = rc;
|
||||
for (0..rc) |i| {
|
||||
const r = node.resources[i];
|
||||
d.resources[i] = .{ .kind = @intFromEnum(r.kind), .start = r.start, .len = r.len };
|
||||
}
|
||||
devices[count] = d;
|
||||
count += 1;
|
||||
return d.id;
|
||||
}
|
||||
|
||||
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
||||
/// available (which may exceed `out.len`).
|
||||
pub fn enumerate(out: []danos.DeviceDescriptor) usize {
|
||||
const n = @min(count, out.len);
|
||||
@memcpy(out[0..n], devices[0..n]);
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
||||
/// out of range or already claimed.
|
||||
pub fn claim(id: u64, owner: u32) bool {
|
||||
if (id >= count) return false;
|
||||
if (claimed[@intCast(id)] != null) return false;
|
||||
claimed[@intCast(id)] = owner;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The task that owns device `id`, or null.
|
||||
pub fn ownerOf(id: u64) ?u32 {
|
||||
if (id >= count) return null;
|
||||
return claimed[@intCast(id)];
|
||||
}
|
||||
|
||||
/// Resource `index` of device `id`, or null if out of range.
|
||||
pub fn resourceOf(id: u64, index: u64) ?danos.ResourceDescriptor {
|
||||
if (id >= count) return null;
|
||||
const d = &devices[@intCast(id)];
|
||||
if (index >= d.resource_count) return null;
|
||||
return d.resources[@intCast(index)];
|
||||
}
|
||||
|
||||
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
|
||||
/// interval containment; for an irq it's equality, since an interrupt line is not
|
||||
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
||||
/// and would otherwise vacuously "fit" anywhere.
|
||||
fn contains(parent: danos.ResourceDescriptor, child: danos.ResourceDescriptor) bool {
|
||||
if (parent.kind != child.kind) return false;
|
||||
if (child.kind == @intFromEnum(danos.ResourceKind.irq)) return parent.start == child.start;
|
||||
if (child.len == 0 or parent.len == 0) return false;
|
||||
// No overflow: a resource that wraps the address space is not containable.
|
||||
const child_end = std.math.add(u64, child.start, child.len) catch return false;
|
||||
const parent_end = std.math.add(u64, parent.start, parent.len) catch return false;
|
||||
return child.start >= parent.start and child_end <= parent_end;
|
||||
}
|
||||
|
||||
pub const RegisterError = error{
|
||||
NoSpace, // the device table is full
|
||||
BadParent, // no such device, or not claimed by this task
|
||||
TooManyResources,
|
||||
TooManyChildren, // this parent is at maximum_children_per_parent
|
||||
NotContained, // a child resource escapes its parent's window
|
||||
};
|
||||
|
||||
/// Number of devices currently recorded with `parent_id` as their parent.
|
||||
fn childCount(parent_id: u64) usize {
|
||||
var n: usize = 0;
|
||||
for (devices[0..count]) |d| {
|
||||
if (d.parent == parent_id) n += 1;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, on behalf of `owner`. Returns the new
|
||||
/// device id. The child is left **unclaimed**, so another process (a class driver)
|
||||
/// can claim it — that is how a bus hands a device to its driver.
|
||||
///
|
||||
/// `owner` must have claimed `parent_id`, and every resource in `descriptor` must be
|
||||
/// contained in a parent resource of the same kind. A device with no resources is
|
||||
/// fine and common: a USB device is addressed through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, owner: u32, descriptor: *const danos.DeviceDescriptor) RegisterError!u64 {
|
||||
const parent_owner = ownerOf(parent_id) orelse return error.BadParent;
|
||||
if (parent_owner != owner) return error.BadParent;
|
||||
if (descriptor.resource_count > danos.maximum_device_resources) return error.TooManyResources;
|
||||
if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;
|
||||
if (count >= maximum_devices) return error.NoSpace;
|
||||
|
||||
const parent = &devices[@intCast(parent_id)];
|
||||
for (0..@intCast(descriptor.resource_count)) |i| {
|
||||
const r = descriptor.resources[i];
|
||||
var ok = false;
|
||||
for (0..@intCast(parent.resource_count)) |j| {
|
||||
if (contains(parent.resources[j], r)) ok = true;
|
||||
}
|
||||
if (!ok) return error.NotContained;
|
||||
}
|
||||
|
||||
var d = std.mem.zeroes(danos.DeviceDescriptor);
|
||||
d.id = count;
|
||||
d.parent = parent_id;
|
||||
d.class = descriptor.class;
|
||||
d.hid_len = @min(descriptor.hid_len, d.hid.len);
|
||||
@memcpy(d.hid[0..@intCast(d.hid_len)], descriptor.hid[0..@intCast(d.hid_len)]);
|
||||
d.resource_count = descriptor.resource_count;
|
||||
for (0..@intCast(descriptor.resource_count)) |i| d.resources[i] = descriptor.resources[i];
|
||||
|
||||
devices[count] = d;
|
||||
count += 1;
|
||||
return d.id;
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,174 @@
|
||||
//! The kernel heap: dynamic allocation for the kernel.
|
||||
//!
|
||||
//! Where the frame allocator ([pmm]) hands out fixed 4 KiB physical frames, the
|
||||
//! heap hands out arbitrary byte-sized blocks from a virtual region, growing on
|
||||
//! demand by mapping fresh frames into it (architecture.mapPage) — the first real user of
|
||||
//! the VMM (see docs/paging.md).
|
||||
//!
|
||||
//! The algorithm is a first-fit free list: an address-ordered singly linked list
|
||||
//! of free blocks, split on allocation and coalesced with neighbours on free. It
|
||||
//! is exposed as a std.mem.Allocator, so the kernel can use std containers.
|
||||
//!
|
||||
//! Not yet concurrency-safe: it assumes a single caller and no allocation from
|
||||
//! interrupt handlers (ours don't). A lock comes with threads/SMP.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const architecture = @import("architecture");
|
||||
const pmm = @import("pmm.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
/// Virtual base of the heap: the start of the higher half, which is unmapped and
|
||||
/// well clear of the identity-mapped low half. (Canonical on x86_64; an architecture that
|
||||
/// splits the address space differently would choose its own.)
|
||||
const heap_base: usize = 0xFFFF_8000_0000_0000;
|
||||
/// Cap on heap growth for now.
|
||||
const heap_maximum: usize = 64 * 1024 * 1024;
|
||||
|
||||
/// A block header, placed at the start of every block. While the block is free it
|
||||
/// also links into the free list via `next`.
|
||||
const Block = extern struct {
|
||||
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||
next: ?*Block, // free-list link (only meaningful while free)
|
||||
};
|
||||
|
||||
const header_size = @sizeOf(Block); // 16
|
||||
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||
|
||||
var free_list: ?*Block = null;
|
||||
var heap_end: usize = heap_base; // [heap_base, heap_end) is currently mapped
|
||||
|
||||
fn alignUp(value: usize, alignment: usize) usize {
|
||||
return (value + alignment - 1) & ~(alignment - 1);
|
||||
}
|
||||
|
||||
fn payloadOf(block: *Block) [*]u8 {
|
||||
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||
}
|
||||
|
||||
/// Bring the heap up with an initial mapped region.
|
||||
pub fn init() void {
|
||||
free_list = null;
|
||||
heap_end = heap_base;
|
||||
_ = grow(page_size);
|
||||
}
|
||||
|
||||
/// Map more pages onto the end of the heap and add them as a free block. Returns
|
||||
/// false if out of heap virtual space or out of physical frames.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const start = heap_end;
|
||||
const bytes = alignUp(minimum_bytes, page_size);
|
||||
if (start + bytes > heap_base + heap_maximum) return false;
|
||||
|
||||
var virtual = start;
|
||||
while (virtual < start + bytes) : (virtual += page_size) {
|
||||
const frame = pmm.alloc() orelse return false;
|
||||
architecture.mapPage(virtual, frame, true);
|
||||
}
|
||||
heap_end = start + bytes;
|
||||
|
||||
const block: *Block = @ptrFromInt(start);
|
||||
block.size = bytes;
|
||||
insertFree(block); // coalesces with the previous tail block if adjacent
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Insert a block into the address-ordered free list, coalescing with the
|
||||
/// physically adjacent free blocks on either side.
|
||||
fn insertFree(block: *Block) void {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |c| : (current = c.next) {
|
||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||
previous = c;
|
||||
}
|
||||
|
||||
block.next = current;
|
||||
if (previous) |p| p.next = block else free_list = block;
|
||||
|
||||
// Merge forward into `current` if they're contiguous.
|
||||
if (current) |c| {
|
||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||
block.size += c.size;
|
||||
block.next = c.next;
|
||||
}
|
||||
}
|
||||
// Merge `previous` forward into `block` if they're contiguous.
|
||||
if (previous) |p| {
|
||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||
p.size += block.size;
|
||||
p.next = block.next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < 2) : (attempts += 1) {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |block| : ({
|
||||
previous = block;
|
||||
current = block.next;
|
||||
}) {
|
||||
if (block.size < need) continue;
|
||||
|
||||
if (block.size >= need + minimum_block) {
|
||||
// Split: carve `need` off the front, leave the rest free.
|
||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||
rest.size = block.size - need;
|
||||
rest.next = block.next;
|
||||
if (previous) |p| p.next = rest else free_list = rest;
|
||||
block.size = need;
|
||||
} else {
|
||||
// Take the whole block.
|
||||
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||
}
|
||||
return payloadOf(block);
|
||||
}
|
||||
|
||||
// Nothing fit: grow and try once more.
|
||||
if (!grow(need)) return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
// --- std.mem.Allocator interface -----------------------------------------
|
||||
|
||||
pub fn allocator() std.mem.Allocator {
|
||||
return .{ .ptr = undefined, .vtable = &vtable };
|
||||
}
|
||||
|
||||
const vtable = std.mem.Allocator.VTable{
|
||||
.alloc = allocImpl,
|
||||
.resize = resizeImpl,
|
||||
.remap = remapImpl,
|
||||
.free = freeImpl,
|
||||
};
|
||||
|
||||
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||
// Blocks are 16-byte aligned; larger alignments aren't supported yet.
|
||||
if (alignment.toByteUnits() > 16) return null;
|
||||
return rawAlloc(len);
|
||||
}
|
||||
|
||||
fn resizeImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) bool {
|
||||
return false; // no in-place resize; the caller reallocates
|
||||
}
|
||||
|
||||
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||
rawFree(memory.ptr);
|
||||
}
|
||||
@@ -0,0 +1,314 @@
|
||||
//! Synchronous IPC: the microkernel message backbone. An `Endpoint` is a
|
||||
//! rendezvous point; a client `call`s it (send a message, block for a reply) and
|
||||
//! a server `replyWait`s on it (reply to the last client, then block for the next
|
||||
//! request). This is the substrate the user-space VFS server and device drivers
|
||||
//! are reached through — `open`/`read`/`write` become user-space wrappers that
|
||||
//! marshal a request into a `call`.
|
||||
//!
|
||||
//! Design (see docs/syscall.md, the plan):
|
||||
//! - **Copy method, no bounce buffer.** Payloads are copied frame-to-frame
|
||||
//! through the physmap (`copyAcross`), which is mapped in every address space's
|
||||
//! shared kernel half — so the kernel reads/writes either process's user memory
|
||||
//! without a CR3 switch, and an unmapped page fails the copy instead of #PF-ing.
|
||||
//! - **Reply routing on the server.** IPC is synchronous, so a server owes a reply
|
||||
//! to exactly one client at a time; that caller is held in `Task.ipc_client`.
|
||||
//! - **Sender FIFO on the endpoint.** A blocked caller must be *received without
|
||||
//! becoming runnable*, which a WaitQueue can't express, so callers queue on the
|
||||
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
||||
//! waiting for work use a normal WaitQueue.
|
||||
//!
|
||||
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
|
||||
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
|
||||
//! copy is a later security-track item, matching the existing debug_write gap.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const architecture = @import("architecture");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const Task = scheduler.Task;
|
||||
|
||||
/// Largest message a single call/reply may carry. Bumping it is trivial; kept
|
||||
/// small because the copy runs under the big kernel lock.
|
||||
pub const MESSAGE_MAXIMUM: usize = 256;
|
||||
|
||||
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
||||
pub const maximum_services = 8;
|
||||
|
||||
/// Errno-style failures, returned as `-value` in the system_call result register.
|
||||
pub const EBADF: i64 = 1; // bad handle
|
||||
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
|
||||
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
||||
pub const ENOENT: i64 = 4; // no such registered service
|
||||
pub const ENOSPC: i64 = 5; // handle table or registry full
|
||||
pub const ENOMEM: i64 = 6; // out of memory
|
||||
|
||||
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
|
||||
/// message from a client — there is no reply owed. The low bits carry the source
|
||||
/// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in system/kernel/irq.zig;
|
||||
/// the message path uses a plain task-id badge with this bit clear. Defined in the
|
||||
/// shared contract (system/danos.zig), because ring 3 has to test the same bit.
|
||||
pub const notify_badge_bit: u64 = danos.notify_badge_bit;
|
||||
|
||||
/// End of the user (low) canonical half — user buffers must lie below it.
|
||||
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
|
||||
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||
pub const Endpoint = struct {
|
||||
refcount: u32 = 1,
|
||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||
sender_head: ?*Task = null,
|
||||
sender_tail: ?*Task = null,
|
||||
// Servers blocked in `replyWait` awaiting a request.
|
||||
receive_wait_queue: scheduler.WaitQueue = .{},
|
||||
// Pending asynchronous notifications (badges), a small coalescing ring.
|
||||
notify_buffer: [8]u64 = undefined,
|
||||
notify_head: u8 = 0,
|
||||
notify_tail: u8 = 0,
|
||||
};
|
||||
|
||||
pub fn createEndpoint() ?*Endpoint {
|
||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||
endpoint.* = .{};
|
||||
return endpoint;
|
||||
}
|
||||
|
||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||
pub fn dropRef(endpoint: *Endpoint) void {
|
||||
if (endpoint.refcount > 1) {
|
||||
endpoint.refcount -= 1;
|
||||
} else {
|
||||
heap.allocator().destroy(endpoint);
|
||||
}
|
||||
}
|
||||
|
||||
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
||||
|
||||
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
||||
t.next = null;
|
||||
if (endpoint.sender_tail) |tail| tail.next = t else endpoint.sender_head = t;
|
||||
endpoint.sender_tail = t;
|
||||
}
|
||||
|
||||
fn dequeueSender(endpoint: *Endpoint) ?*Task {
|
||||
const t = endpoint.sender_head orelse return null;
|
||||
endpoint.sender_head = t.next;
|
||||
if (endpoint.sender_head == null) endpoint.sender_tail = null;
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
// --- cross-address-space copy ----------------------------------------------
|
||||
|
||||
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
|
||||
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
|
||||
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
||||
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
|
||||
/// unmapped or out of range. Handles page-straddling buffers.
|
||||
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
|
||||
const source_root = if (source_as != 0) source_as else architecture.kernelPageTable();
|
||||
const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable();
|
||||
if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false;
|
||||
if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false;
|
||||
|
||||
var off: usize = 0;
|
||||
while (off < len) {
|
||||
const s = architecture.translate(source_root, source_va + off) orelse return false;
|
||||
const d = architecture.translate(destination_root, destination_va + off) orelse return false;
|
||||
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
||||
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
||||
const n = @min(@min(s_left, d_left), len - off);
|
||||
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(d));
|
||||
@memcpy(destination[0..n], source[0..n]);
|
||||
off += n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel
|
||||
/// buffer `destination`, walking the user page tables through the physmap. Returns false if
|
||||
/// the range escapes the user half or any source page is unmapped — so a bad user
|
||||
/// pointer *fails the system_call* rather than faulting the kernel (danos has no
|
||||
/// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt
|
||||
/// the machine). The correct way to pull a fixed-size struct in from user space, and
|
||||
/// a single fetch: no TOCTOU against a hostile pointer.
|
||||
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||
if (user_as == 0) return false; // not a user address space
|
||||
if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false;
|
||||
var off: usize = 0;
|
||||
while (off < destination.len) {
|
||||
const s = architecture.translate(user_as, user_va + off) orelse return false;
|
||||
const s_left = page_size - ((user_va + off) & (page_size - 1));
|
||||
const n = @min(s_left, destination.len - off);
|
||||
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
|
||||
@memcpy(destination[off..][0..n], source[0..n]);
|
||||
off += n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- the two IPC operations -------------------------------------------------
|
||||
|
||||
/// Client side of IPC_Call: send `[message_ptr, message_len)` to `endpoint` and block until a
|
||||
/// server replies into `[reply_ptr, reply_cap)`. Returns the reply length, or a
|
||||
/// negative errno. Runs as the current task.
|
||||
pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr: u64, reply_cap: u64) i64 {
|
||||
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const me = scheduler.current();
|
||||
me.ipc_send_ptr = message_ptr;
|
||||
me.ipc_send_len = message_len;
|
||||
me.ipc_reply_ptr = reply_ptr;
|
||||
me.ipc_reply_cap = reply_cap;
|
||||
me.ipc_status = 0;
|
||||
|
||||
enqueueSender(endpoint, me); // join the FIFO, then...
|
||||
scheduler.wakeLocked(&endpoint.receive_wait_queue); // ...wake a waiting server (no-op if none)
|
||||
scheduler.blockCurrentLocked(); // block until the reply readies us again
|
||||
|
||||
return me.ipc_status; // reply length or -errno, written by the replier
|
||||
}
|
||||
|
||||
/// Server side of IPC_ReplyWait: deliver `[reply_ptr, reply_len)` to the client
|
||||
/// we currently owe (if any), then receive the next request into
|
||||
/// `[receive_ptr, receive_cap)`, blocking until one arrives. Writes the sender's badge
|
||||
/// to `out_badge` and returns the request length, or a negative errno. A pending
|
||||
/// notification is delivered ahead of client requests (length 0, badge with
|
||||
/// `notify_badge_bit` set, no reply owed).
|
||||
pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_ptr: u64, receive_cap: u64, out_badge: *u64) i64 {
|
||||
if (reply_len > MESSAGE_MAXIMUM or receive_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const me = scheduler.current();
|
||||
|
||||
// (1) Reply to the client we're still holding, if any.
|
||||
if (me.ipc_client) |client| {
|
||||
me.ipc_client = null;
|
||||
const n = @min(reply_len, client.ipc_reply_cap);
|
||||
if (copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
|
||||
client.ipc_status = @intCast(n);
|
||||
} else {
|
||||
client.ipc_status = -EFAULT;
|
||||
}
|
||||
scheduler.readyLocked(client); // its `call` now returns
|
||||
}
|
||||
|
||||
// (2) Receive the next request (or notification), blocking until one is ready.
|
||||
while (true) {
|
||||
if (popNotify(endpoint)) |badge| {
|
||||
out_badge.* = badge | notify_badge_bit;
|
||||
return 0; // notification: no payload, no reply owed
|
||||
}
|
||||
if (dequeueSender(endpoint)) |caller| {
|
||||
const n = @min(caller.ipc_send_len, receive_cap);
|
||||
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) {
|
||||
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
||||
scheduler.readyLocked(caller);
|
||||
continue;
|
||||
}
|
||||
me.ipc_client = caller; // remember who to reply to
|
||||
out_badge.* = caller.id;
|
||||
return @intCast(n);
|
||||
}
|
||||
scheduler.waitLocked(&endpoint.receive_wait_queue); // nothing yet — sleep until woken, then retry
|
||||
}
|
||||
}
|
||||
|
||||
// --- asynchronous notification (for IRQ-as-message, M10) --------------------
|
||||
|
||||
fn popNotify(endpoint: *Endpoint) ?u64 {
|
||||
if (endpoint.notify_head == endpoint.notify_tail) return null;
|
||||
const badge = endpoint.notify_buffer[endpoint.notify_head % endpoint.notify_buffer.len];
|
||||
endpoint.notify_head +%= 1;
|
||||
return badge;
|
||||
}
|
||||
|
||||
/// Post an asynchronous notification carrying `badge` to `endpoint` and wake a waiting
|
||||
/// receiver. Precondition: the big kernel lock is held.
|
||||
///
|
||||
/// The lock must already cover whatever produced `endpoint` — an ISR that looked the
|
||||
/// endpoint up in a table and *then* took the lock could be racing a process exit
|
||||
/// that unbinds and frees it in between. See irq.dispatch, which holds one lock
|
||||
/// region across the table read and this call.
|
||||
///
|
||||
/// A full ring drops the notification. That is the correct semantics, not a
|
||||
/// concession: a notification is a *level* ("this device wants attention"), and the
|
||||
/// driver re-reads device state on wake. It is never a count of events.
|
||||
pub fn notifyLocked(endpoint: *Endpoint, badge: u64) void {
|
||||
if (endpoint.notify_tail -% endpoint.notify_head < endpoint.notify_buffer.len) {
|
||||
endpoint.notify_buffer[endpoint.notify_tail % endpoint.notify_buffer.len] = badge;
|
||||
endpoint.notify_tail +%= 1;
|
||||
}
|
||||
scheduler.wakeLocked(&endpoint.receive_wait_queue);
|
||||
}
|
||||
|
||||
/// `notifyLocked` as a self-contained ISR critical section, for a caller that holds
|
||||
/// `endpoint` by some means other than a table the lock protects. Releases the lock without
|
||||
/// touching the interrupt flag (the ISR's iretq restores it), like the timer tick.
|
||||
pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
|
||||
_ = sync.enter();
|
||||
notifyLocked(endpoint, badge);
|
||||
sync.leaveIsr();
|
||||
}
|
||||
|
||||
// --- per-process handle table + name registry -------------------------------
|
||||
|
||||
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
|
||||
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
||||
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||
for (&t.handles, 0..) |*slot, i| {
|
||||
if (slot.* == null) {
|
||||
slot.* = @ptrCast(endpoint);
|
||||
return @intCast(i);
|
||||
}
|
||||
}
|
||||
return -ENOSPC;
|
||||
}
|
||||
|
||||
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
if (h >= t.handles.len) return null;
|
||||
const slot = t.handles[@intCast(h)] orelse return null;
|
||||
return @ptrCast(@alignCast(slot));
|
||||
}
|
||||
|
||||
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
||||
/// exit path so a dead server's endpoints don't linger referenced.
|
||||
pub fn closeHandles(t: *Task) void {
|
||||
for (&t.handles) |*slot| {
|
||||
if (slot.*) |p| {
|
||||
dropRef(@ptrCast(@alignCast(p)));
|
||||
slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
||||
|
||||
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||
pub fn register(id: u32, endpoint: *Endpoint) i64 {
|
||||
if (id >= maximum_services) return -ENOENT;
|
||||
if (registry[id]) |old| dropRef(old);
|
||||
endpoint.refcount += 1;
|
||||
registry[id] = endpoint;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, taking a reference for the caller to
|
||||
/// install in its handle table. Null if nothing is registered there.
|
||||
pub fn lookup(id: u32) ?*Endpoint {
|
||||
if (id >= maximum_services) return null;
|
||||
const endpoint = registry[id] orelse return null;
|
||||
endpoint.refcount += 1;
|
||||
return endpoint;
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
//! Inter-process communication: message-passing channels.
|
||||
//!
|
||||
//! IPC is the backbone of a microkernel ([vision](../docs/vision.md)): once
|
||||
//! drivers and services live in separate address spaces, a message is how they
|
||||
//! talk. This first form is a **bounded blocking channel** — a ring buffer of
|
||||
//! messages with a producer/consumer rendezvous, built on the scheduler's
|
||||
//! [wait queues](scheduling.md). `send` blocks when the channel is full, `receive`
|
||||
//! blocks when it's empty; neither busy-waits.
|
||||
//!
|
||||
//! For now both endpoints are kernel threads sharing the kernel address space.
|
||||
//! When user mode arrives, the same primitive carries messages across the
|
||||
//! isolation boundary (with the payload copied between address spaces).
|
||||
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
|
||||
/// A bounded blocking channel of `capacity` messages of type `T`.
|
||||
pub fn Channel(comptime T: type, comptime capacity: usize) type {
|
||||
return struct {
|
||||
const Self = @This();
|
||||
|
||||
buffer: [capacity]T = undefined,
|
||||
head: usize = 0, // next slot to read
|
||||
tail: usize = 0, // next slot to write
|
||||
count: usize = 0,
|
||||
not_full: scheduler.WaitQueue = .{}, // senders wait here
|
||||
not_empty: scheduler.WaitQueue = .{}, // receivers wait here
|
||||
|
||||
/// Send a message, blocking while the channel is full.
|
||||
pub fn send(self: *Self, message: T) void {
|
||||
const flags = sync.enter();
|
||||
// Recheck the condition in a loop: a wakeup only means "try again"
|
||||
// (another waiter may have taken the slot first).
|
||||
while (self.count == capacity) scheduler.waitLocked(&self.not_full);
|
||||
self.buffer[self.tail] = message;
|
||||
self.tail = (self.tail + 1) % capacity;
|
||||
self.count += 1;
|
||||
scheduler.wakeLocked(&self.not_empty); // a receiver can now proceed
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// Receive a message, blocking while the channel is empty.
|
||||
pub fn receive(self: *Self) T {
|
||||
const flags = sync.enter();
|
||||
while (self.count == 0) scheduler.waitLocked(&self.not_empty);
|
||||
const message = self.buffer[self.head];
|
||||
self.head = (self.head + 1) % capacity;
|
||||
self.count -= 1;
|
||||
scheduler.wakeLocked(&self.not_full); // a sender can now proceed
|
||||
sync.leave(flags);
|
||||
return message;
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,190 @@
|
||||
//! IRQ-as-IPC: delivering a hardware interrupt to a user-space driver.
|
||||
//!
|
||||
//! A microkernel can't run driver code in the ISR — the driver is a ring-3 process
|
||||
//! in another address space. So the kernel's ISR does the least it can: quiet the
|
||||
//! line, acknowledge the CPU, and post an asynchronous notification to the endpoint
|
||||
//! the driver is blocked on (`ipc_sync.notifyFromIsr`). The driver wakes out of
|
||||
//! `IPC_ReplyWait`, services the device, and calls `irq_ack` to re-arm.
|
||||
//!
|
||||
//! The full cycle, and why each step is where it is:
|
||||
//!
|
||||
//! ISR irqMask(gsi) -- the line is still asserted; stop it reaching a CPU
|
||||
//! irqEoi() -- now safe to tell the LAPIC we're done
|
||||
//! notifyFromIsr() -- wake the driver (it runs much later)
|
||||
//! driver <services device> -- reads/clears the device's status register
|
||||
//! driver irq_ack(device,resource) -- irqUnmask(gsi): the line is quiet, let it through
|
||||
//!
|
||||
//! Mask-before-EOI is the load-bearing part. A level-triggered line stays asserted
|
||||
//! until the *device* is quieted, which only the ring-3 driver can do. EOI with the
|
||||
//! entry unmasked and the I/O APIC redelivers immediately, forever, before the
|
||||
//! driver is ever scheduled. Masking converts "level" into something a deferred
|
||||
//! handler can cope with; `irq_ack` is what closes the loop.
|
||||
//!
|
||||
//! Binding is capability-gated exactly like `mmio_map`: the caller must have
|
||||
//! `device_claim`ed the device, and the GSI must come from one of that device's `irq`
|
||||
//! resources in the discovered device table (system/kernel/device-service.zig). A driver can
|
||||
//! therefore never bind an interrupt it doesn't own — a raw-GSI system_call would let
|
||||
//! any process steal the keyboard's line.
|
||||
//!
|
||||
//! KNOWN ISSUE (real hardware, not QEMU). Masking a level-triggered redirection entry
|
||||
//! while its remote-IRR bit is set does not clear remote-IRR on some chipsets, and the
|
||||
//! line then never fires again — a driver would take exactly one interrupt and block
|
||||
//! forever. QEMU's I/O APIC clears remote-IRR on EOI regardless of the mask, so the
|
||||
//! `hpet` test cannot see this. Linux's workaround is to flush remote-IRR by briefly
|
||||
//! flipping the entry to edge-trigger and back. Revisit when danos first boots on
|
||||
//! metal; MSI (no mask cycle at all) sidesteps it entirely.
|
||||
|
||||
const architecture = @import("architecture");
|
||||
const sync = @import("sync.zig");
|
||||
const ipc_sync = @import("ipc-synchronous.zig");
|
||||
|
||||
/// GSIs a single I/O APIC covers. 24 is the standard redirection-table size; a
|
||||
/// second I/O APIC (none on QEMU's q35) would extend this.
|
||||
pub const maximum_gsi = 24;
|
||||
|
||||
/// The endpoint to notify for each bound GSI, or null if unbound. Read from the ISR
|
||||
/// and written from syscalls, always under the big kernel lock.
|
||||
var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi;
|
||||
|
||||
/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint
|
||||
/// pointer: an endpoint can be shared between processes (ipc_register/ipc_lookup hand
|
||||
/// out extra references), so "every GSI pointing at this endpoint" is not the same
|
||||
/// set as "every GSI this process bound", and releasing the former on exit would mask
|
||||
/// a live sibling's device line.
|
||||
var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi;
|
||||
|
||||
/// No GSI assigned to this vector.
|
||||
const no_gsi: u32 = 0xFFFF_FFFF;
|
||||
|
||||
/// Reverse map for the ISR: which GSI does this vector carry? Populated at bind.
|
||||
/// `interruptDispatch` hands a handler no arguments, so the vector→GSI edge has to
|
||||
/// be recovered from somewhere — the trampolines below capture the vector at
|
||||
/// comptime, and this turns it back into a GSI. Trampolines are installed on every
|
||||
/// vector in the window at boot, so an unbound one must be distinguishable from
|
||||
/// GSI 0 — hence the sentinel rather than a zero default.
|
||||
var vector_gsi: [256]u32 = .{no_gsi} ** 256;
|
||||
|
||||
/// Vector currently assigned to each GSI (0 = none), so a rebind reuses it.
|
||||
var gsi_vector: [maximum_gsi]u8 = .{0} ** maximum_gsi;
|
||||
|
||||
/// Set once the trampolines are installed.
|
||||
var installed = false;
|
||||
|
||||
/// The ISR body for a bound device line. Runs with interrupts off, on the
|
||||
/// interrupted task's kernel stack, on whichever core the I/O APIC picked.
|
||||
fn dispatch(vector: u8) void {
|
||||
const gsi = vector_gsi[vector];
|
||||
if (gsi == no_gsi) {
|
||||
// Nothing is routed here. Acknowledge so the LAPIC doesn't wedge on an
|
||||
// in-service bit that never clears, but touch no redirection entry.
|
||||
architecture.irqEoi();
|
||||
return;
|
||||
}
|
||||
|
||||
// One lock region for the whole cycle. Two reasons, and the second is subtle:
|
||||
//
|
||||
// - The I/O APIC is an index/data register pair, so two cores interleaving a
|
||||
// read-modify-write of a redirection entry would corrupt it.
|
||||
// - `bound[gsi]` must be *read and used* under the same acquisition that
|
||||
// `unbind` writes it under. Dropping the lock between the load and
|
||||
// `notifyLocked` would let a driver exiting on another core free the endpoint
|
||||
// in the gap, and we would post a notification into freed memory. The GSI is
|
||||
// routed to the core that bound it, but a driver may migrate and exit
|
||||
// elsewhere, so this is reachable on SMP.
|
||||
//
|
||||
// No deadlock: the lock is non-recursive, but a core holding it runs with
|
||||
// interrupts disabled and so cannot interrupt itself into here.
|
||||
_ = sync.enter();
|
||||
defer sync.leaveIsr();
|
||||
|
||||
architecture.irqMask(gsi); // the line is still asserted; stop it reaching a CPU
|
||||
architecture.irqEoi(); // now safe to release the LAPIC's in-service bit
|
||||
|
||||
// Wakes the driver if it's blocked in ReplyWait; otherwise queues the badge on
|
||||
// the endpoint's notify ring, so an interrupt taken while the driver is off
|
||||
// doing something else is not lost.
|
||||
if (bound[gsi]) |endpoint| ipc_sync.notifyLocked(endpoint, gsi);
|
||||
}
|
||||
|
||||
/// Install one no-argument trampoline per usable vector. Each closes over its own
|
||||
/// `vector` as a comptime constant — that's the trick that gets an argument into
|
||||
/// `idt.Handler` (`*const fn () void`) without a per-vector hand-written stub.
|
||||
pub fn init() void {
|
||||
if (installed) return;
|
||||
inline for (0..architecture.irq_vector_count) |i| {
|
||||
const vector: u8 = @intCast(@as(usize, architecture.irq_vector_base) + i);
|
||||
architecture.irqSetHandler(vector, &struct {
|
||||
fn trampoline() void {
|
||||
dispatch(vector);
|
||||
}
|
||||
}.trampoline);
|
||||
}
|
||||
installed = true;
|
||||
}
|
||||
|
||||
/// Lowest unused vector in the device window, or null if they're all spoken for.
|
||||
fn allocVector() ?u8 {
|
||||
var v: u8 = architecture.irq_vector_base;
|
||||
while (v < architecture.irq_vector_base + architecture.irq_vector_count) : (v += 1) {
|
||||
var used = false;
|
||||
for (gsi_vector) |gv| {
|
||||
if (gv == v) used = true;
|
||||
}
|
||||
if (!used) return v;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub const BindError = error{ BadGsi, InUse, NoVector };
|
||||
|
||||
/// Deliver `gsi` to `endpoint` as an IPC notification, on behalf of task `owner`. Routes the
|
||||
/// line to this core, installs the binding, and unmasks. Caller must hold the big
|
||||
/// kernel lock, and must already have checked that `owner` claimed the device this GSI
|
||||
/// belongs to.
|
||||
pub fn bind(gsi: u32, endpoint: *ipc_sync.Endpoint, owner: u32) BindError!void {
|
||||
if (gsi >= maximum_gsi or !architecture.irqOwnsGsi(gsi)) return error.BadGsi;
|
||||
if (bound[gsi] != null) return error.InUse;
|
||||
const vector = allocVector() orelse return error.NoVector;
|
||||
|
||||
vector_gsi[vector] = gsi;
|
||||
gsi_vector[gsi] = vector;
|
||||
bound[gsi] = endpoint;
|
||||
bound_owner[gsi] = owner;
|
||||
|
||||
// Level-triggered, active-high. Level is the general case a driver must survive
|
||||
// (and what hpetd configures its comparator for); an edge source simply never
|
||||
// leaves the line asserted, so the mask/ack cycle is harmless there.
|
||||
//
|
||||
// Hardcoded for now: a device whose MADT interrupt-source override declares the
|
||||
// line active-*low* (most legacy PCI INTx) will need the polarity threaded
|
||||
// through from discovery. Nothing danos binds today is such a device.
|
||||
architecture.irqRoute(gsi, vector, true, false);
|
||||
architecture.irqUnmask(gsi);
|
||||
}
|
||||
|
||||
/// Re-arm `gsi` after the driver has quieted the device. Caller holds the big lock
|
||||
/// and has verified ownership.
|
||||
pub fn ack(gsi: u32) bool {
|
||||
if (gsi >= maximum_gsi or bound[gsi] == null) return false;
|
||||
architecture.irqUnmask(gsi);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Drop every binding made by task `owner` — called as that task exits, *before* its
|
||||
/// endpoints are freed. Each line is left masked, so a dead driver's device goes quiet
|
||||
/// rather than interrupting into a freed endpoint. Caller holds the big kernel lock,
|
||||
/// which is what makes this safe against a concurrent `dispatch` on another core.
|
||||
///
|
||||
/// Keyed on the owner, not the endpoint: endpoints are shared (a registered service's
|
||||
/// endpoint has references in several processes), so releasing "everything pointing at
|
||||
/// this endpoint" would tear down bindings this task never made.
|
||||
pub fn releaseOwner(owner: u32) void {
|
||||
for (&bound, 0..) |*slot, gsi| {
|
||||
if (slot.* != null and bound_owner[gsi] == owner) {
|
||||
architecture.irqMask(@intCast(gsi));
|
||||
slot.* = null;
|
||||
bound_owner[gsi] = 0;
|
||||
gsi_vector[gsi] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
//! The kernel's multi-sink **diagnostic** log — the machine-readable stream of
|
||||
//! what the kernel is doing, separate from any user-facing display.
|
||||
//!
|
||||
//! Output is a *diagnostic convenience, never a correctness dependency* — the
|
||||
//! kernel must boot and run correctly with zero output channels. So logging fans
|
||||
//! out to a set of registered **sinks**, each best-effort and self-guarding: the
|
||||
//! serial UART, the 0xE9 debug console, and — later — a file on a ramdisk/USB/SSD.
|
||||
//! A message reaches whatever channels exist; if none do, the kernel runs on,
|
||||
//! silent but correct.
|
||||
//!
|
||||
//! The **framebuffer is deliberately not a sink here.** It's a separate output
|
||||
//! surface (a bootstrap text console today, a graphics device driver later), so
|
||||
//! the log never assumes the machine is text-based. `main.zig` mirrors a few
|
||||
//! user-facing status lines and panics to it explicitly; the verbose log does not.
|
||||
//!
|
||||
//! No allocation: the sink table is fixed, so the log works before the heap is up
|
||||
//! and inside a panic. Two channels don't go through the sink list because they
|
||||
//! must survive even a total-output failure: `checkpoint` (a one-byte POST code)
|
||||
//! and `recordPanic` (a breadcrumb in a fixed record).
|
||||
|
||||
const std = @import("std");
|
||||
const architecture = @import("architecture");
|
||||
|
||||
pub const SinkFn = *const fn ([]const u8) void;
|
||||
|
||||
const maximum_sinks = 8;
|
||||
var sinks: [maximum_sinks]SinkFn = undefined;
|
||||
var sink_count: usize = 0;
|
||||
|
||||
/// Register an output sink. Every registered sink receives every message; sinks
|
||||
/// must be self-guarding (safe to call when their device is absent).
|
||||
pub fn addSink(sink: SinkFn) void {
|
||||
if (sink_count < maximum_sinks) {
|
||||
sinks[sink_count] = sink;
|
||||
sink_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Fan `bytes` out to every registered sink.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
for (sinks[0..sink_count]) |sink| sink(bytes);
|
||||
}
|
||||
|
||||
/// A formatted log line. Truncates past 256 bytes; the buffer is on the stack, so
|
||||
/// this is safe to call from interrupt context and from a panic.
|
||||
pub fn print(comptime fmt: []const u8, args: anytype) void {
|
||||
var buffer: [256]u8 = undefined;
|
||||
write(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// Emit a one-byte checkpoint/POST code (I/O port 0x80) — the always-available
|
||||
/// progress channel for when there is no text output at all. Independent of the
|
||||
/// sink list, so it works even before any sink is registered.
|
||||
pub fn checkpoint(code: u8) void {
|
||||
architecture.checkpoint(code);
|
||||
}
|
||||
|
||||
// --- persistent panic breadcrumb -------------------------------------------
|
||||
//
|
||||
// A fixed record in the kernel image that a panic fills in, so a post-mortem — an
|
||||
// attached debugger, a RAM dump, or (later) a file/pstore reader — can recover
|
||||
// what killed the kernel even when there was no live console. `magic` is written
|
||||
// *last*, so a reader only trusts a fully-written record.
|
||||
|
||||
pub const panic_magic: u64 = 0xD1ED_B00B_5EED_F00D;
|
||||
|
||||
pub const PanicRecord = extern struct {
|
||||
magic: u64 = 0,
|
||||
len: u32 = 0,
|
||||
_pad: u32 = 0,
|
||||
message: [512]u8 = undefined,
|
||||
};
|
||||
|
||||
/// Findable by symbol (`log.panic_record`) for a debugger or RAM dump.
|
||||
pub var panic_record: PanicRecord = .{};
|
||||
|
||||
/// Stamp the panic message into the breadcrumb record.
|
||||
pub fn recordPanic(message: []const u8) void {
|
||||
const n: u32 = @intCast(@min(message.len, panic_record.message.len));
|
||||
@memcpy(panic_record.message[0..n], message[0..n]);
|
||||
panic_record.len = n;
|
||||
panic_record.magic = panic_magic; // set last: a reader sees a complete record
|
||||
}
|
||||
@@ -0,0 +1,431 @@
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const parameters = @import("parameters");
|
||||
const architecture = @import("architecture");
|
||||
const console = @import("console.zig");
|
||||
const log = @import("log.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const process = @import("process.zig");
|
||||
const device_service = @import("device-service.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const initrd = @import("initrd");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = danos.BootInformation;
|
||||
|
||||
/// The calling convention used to enter the kernel. Pinned to SystemV explicitly:
|
||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
||||
/// x64 (first argument in RCX), while the kernel is SystemV (first argument in
|
||||
/// RDI). Both sides reference this so the `boot_information` pointer lands in the
|
||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
||||
pub const kernel_abi = danos.kernel_abi;
|
||||
|
||||
// POST/checkpoint codes emitted to I/O port 0x80 at boot milestones — the
|
||||
// last-resort progress signal on a machine with no text output at all.
|
||||
const cp_entry = 0x10;
|
||||
const cp_paging = 0x20;
|
||||
const cp_heap = 0x30;
|
||||
const cp_discovery = 0x40;
|
||||
const cp_scheduler = 0x50;
|
||||
const cp_timer = 0x60;
|
||||
const cp_running = 0x70;
|
||||
const cp_exception = 0xE0;
|
||||
const cp_panic = 0xEE;
|
||||
|
||||
/// Physical address of the low page reserved at boot for the AP trampoline (0 = none
|
||||
/// was available). Claimed right after the frame allocator comes up, before paging
|
||||
/// and the heap consume the scarce sub-1 MiB frames.
|
||||
var ap_trampoline_page: u64 = 0;
|
||||
|
||||
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
||||
/// caller to return to, so this never returns.
|
||||
/// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss
|
||||
/// then calls this with the loader's `boot_information` pointer in RDI. We can't keep
|
||||
/// running on the loader's stack: it's a low physical address that the identity
|
||||
/// map covers only transitionally, and vanishes once the kernel drops the low
|
||||
/// half. `boot_information` (also low) is reached through the physmap — its base is the
|
||||
/// same under the loader's bootstrap tables and the kernel's own.
|
||||
export fn kmainEntry(boot_information: *const BootInformation) callconv(kernel_abi) noreturn {
|
||||
kmain(@ptrFromInt(danos.physicalToVirtual(@intFromPtr(boot_information))));
|
||||
}
|
||||
|
||||
fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// The **log** is the machine-readable diagnostic stream: it fans out to every
|
||||
// *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a
|
||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
||||
// with port-0x80 checkpoints as the only progress signal.
|
||||
architecture.serialInit();
|
||||
log.addSink(architecture.serialWrite);
|
||||
if (architecture.debugconPresent()) log.addSink(architecture.debugconWrite);
|
||||
|
||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
||||
const fb = boot_information.framebuffer;
|
||||
console.init(fb);
|
||||
|
||||
log.checkpoint(cp_entry);
|
||||
|
||||
// Catch CPU exceptions before doing anything that might fault: install our
|
||||
// reporter, then bring up the GDT + IDT.
|
||||
architecture.setFaultHandler(onException);
|
||||
architecture.init();
|
||||
|
||||
status("danos: initialising kernel...\n");
|
||||
log.write(if (console.present())
|
||||
"danos: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
else
|
||||
"danos: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
log.write("danos: cpu tables online (GDT, IDT, TSS)\n");
|
||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
||||
log.print(" format : {s}\n", .{@tagName(fb.format)});
|
||||
log.print(" framebuffer: 0x{x:0>16}\n", .{fb.base});
|
||||
log.print (" footprint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
|
||||
|
||||
// Summarise the physical memory the loader handed us. The array is danos's
|
||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
|
||||
var usable_pages: u64 = 0;
|
||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
||||
for (regions) |r| {
|
||||
switch (r.kind) {
|
||||
.usable => usable_pages += r.pages,
|
||||
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
|
||||
.mmio => {},
|
||||
}
|
||||
}
|
||||
const total_pages = usable_pages + reserved_pages;
|
||||
const total_bytes = total_pages * danos.page_size;
|
||||
const gib = 1 << 30;
|
||||
|
||||
log.write("\ndanos: physical memory\n");
|
||||
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
||||
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
||||
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
||||
log.print(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
|
||||
|
||||
// Bring up the physical frame allocator over that map, and prove it works:
|
||||
// allocate three frames, then hand them back.
|
||||
pmm.init(boot_information.memory_map);
|
||||
// Claim the AP trampoline's low (<1 MiB) page *now*, before paging and the heap
|
||||
// draw down sub-1 MiB frames (the allocator scans upward from frame 0). Held
|
||||
// until SMP bring-up; 0 means none was available (we stay uniprocessor).
|
||||
ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0;
|
||||
const s1 = pmm.stats();
|
||||
log.print("\ndanos: frame allocator online\n", .{});
|
||||
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
||||
const f0 = pmm.alloc();
|
||||
const f1 = pmm.alloc();
|
||||
const f2 = pmm.alloc();
|
||||
log.print(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
|
||||
if (f0) |p| pmm.free(p);
|
||||
if (f1) |p| pmm.free(p);
|
||||
if (f2) |p| pmm.free(p);
|
||||
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
||||
|
||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||
architecture.enablePaging(pmm.alloc, pmm.free, boot_information);
|
||||
log.checkpoint(cp_paging);
|
||||
log.print("\ndanos: paging enabled\n", .{});
|
||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||
|
||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||
heap.init();
|
||||
log.checkpoint(cp_heap);
|
||||
log.write("\ndanos: kernel heap online\n");
|
||||
// Measure the amount of resources the kernel is actually using
|
||||
const s2 = pmm.stats();
|
||||
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
||||
|
||||
// Enumerate hardware from the firmware tables (ACPI here) into a generic
|
||||
// device tree, then list it. Discovery walks ACPI memory directly (identity-
|
||||
// mapped) and maps PCIe configuration space on demand via the VMM. A failure here is
|
||||
// not fatal yet — log it and carry on.
|
||||
const hal = platform.Hal{
|
||||
.mapMmio = architecture.mapMmio,
|
||||
.pioRead = architecture.pioRead,
|
||||
.pioWrite = architecture.pioWrite,
|
||||
};
|
||||
if (platform.discover(boot_information, heap.allocator(), hal)) |devtree| {
|
||||
var device_tree = devtree;
|
||||
log.write("\ndanos: device discovery online\n");
|
||||
device_tree.dump(log.write);
|
||||
|
||||
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
|
||||
// mmio_map operate on this flat, id-indexed table + claim map).
|
||||
device_service.init(&device_tree);
|
||||
if (device_service.dropped > 0) {
|
||||
// Otherwise entirely silent: drivers would just never see that hardware.
|
||||
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{device_service.dropped});
|
||||
}
|
||||
|
||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||
irq.init();
|
||||
|
||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
||||
const pw = platform.powerInformation();
|
||||
log.write("danos: power\n");
|
||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||
if (pw.s5) |s| {
|
||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
||||
} else {
|
||||
log.write(" S5 slp_typ : (not found)\n");
|
||||
}
|
||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||
|
||||
// AML namespace parse integrity: consumed should equal total.
|
||||
const am = platform.amlStats();
|
||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
||||
|
||||
// Feed the architecture layer the discovered addresses/facts so it makes no legacy
|
||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||
const pinfo = platform.platformInformation();
|
||||
const hpet_base: u64 = if (device_tree.firstOfClass(.timer)) |t|
|
||||
(if (t.firstResource(.memory)) |r| r.start else 0)
|
||||
else
|
||||
0;
|
||||
var ioapic_base: u64 = 0;
|
||||
var ioapic_gsi: u32 = 0;
|
||||
if (device_tree.firstOfClass(.interrupt_controller)) |ic| {
|
||||
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
|
||||
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
|
||||
}
|
||||
var isos: [16]architecture.IsoEntry = undefined;
|
||||
const iso_n = @min(pinfo.override_count, isos.len);
|
||||
for (0..iso_n) |i| isos[i] = .{
|
||||
.source = pinfo.overrides[i].source,
|
||||
.gsi = pinfo.overrides[i].gsi,
|
||||
.flags = pinfo.overrides[i].flags,
|
||||
};
|
||||
const pm_timer: ?architecture.PmTimer = if (pinfo.pm_timer.present())
|
||||
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
|
||||
else
|
||||
null;
|
||||
architecture.configurePlatform(.{
|
||||
.pic_present = pinfo.pic_present,
|
||||
.hpet_base = hpet_base,
|
||||
.pm_timer = pm_timer,
|
||||
.ioapic_base = ioapic_base,
|
||||
.ioapic_gsi_base = ioapic_gsi,
|
||||
.overrides = isos[0..iso_n],
|
||||
});
|
||||
if (pinfo.spcr_uart) |u| architecture.serialReconfigure(u.mmio, u.address);
|
||||
|
||||
log.write("danos: platform\n");
|
||||
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
||||
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
||||
log.print(" hpet base : 0x{x}\n", .{hpet_base});
|
||||
log.print(" pm timer : {s} 0x{x} ({s})\n", .{ if (pinfo.pm_timer.mmio) "mmio" else "io", pinfo.pm_timer.address, if (pinfo.pm_timer_32bit) "32-bit" else "24-bit" });
|
||||
if (pinfo.spcr_uart) |u| {
|
||||
log.print(" console UART: {s} 0x{x} (SPCR type {d})\n", .{ if (u.mmio) "mmio" else "io", u.address, pinfo.spcr_kind });
|
||||
} else {
|
||||
log.write(" console UART: none in SPCR -> legacy COM1\n");
|
||||
}
|
||||
log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, architecture.irqRouteCount(), architecture.irqRouteRaw(0) });
|
||||
const cores = platform.cpus();
|
||||
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
||||
if (platform.cpusDropped() > 0)
|
||||
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
||||
} else |err| {
|
||||
log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
log.checkpoint(cp_discovery);
|
||||
|
||||
// Install the system_call handler (int 0x80 gate + system_call stub) once, before any
|
||||
// user code runs.
|
||||
process.init();
|
||||
|
||||
// Register the current context as the first task before enabling preemption.
|
||||
scheduler.init(4);
|
||||
log.checkpoint(cp_scheduler);
|
||||
log.write("\ndanos: scheduler online\n");
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||
// the timer preempts among tasks.
|
||||
architecture.startTimer();
|
||||
architecture.enableInterrupts();
|
||||
log.checkpoint(cp_timer);
|
||||
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ architecture.timer_hz, architecture.timerClockHz() / 1_000_000, architecture.clockHz() / 1_000_000, architecture.timerCalibrationSource() });
|
||||
|
||||
// Wake the other cores (application processors). A no-op on a single-core
|
||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
||||
bringUpSecondaries();
|
||||
|
||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||
// Normal builds fall through to the idle halt.
|
||||
if (build_options.test_case) |case| {
|
||||
tests.run(case, boot_information);
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
log.checkpoint(cp_running);
|
||||
status("kernel initialised.\n");
|
||||
|
||||
// Hand over to user space: load /sbin/init (read off the boot volume by the
|
||||
// loader) and spawn it as a real ring-3 process, PID 1. It runs on its own
|
||||
// address space, preemptively, alongside the kernel — no cooperative
|
||||
// borrowing. This boot context then becomes the BSP's idle loop.
|
||||
if (boot_information.init_len != 0) {
|
||||
status("starting /sbin/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
process.spawnProcess(image, 4) catch |err| {
|
||||
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
} else {
|
||||
status("no /sbin/init on the boot volume.\n");
|
||||
}
|
||||
|
||||
// Spawn the extra user binaries the loader ferried in the initrd (the VFS
|
||||
// server, and later device drivers). For now the kernel launches them all;
|
||||
// once init is a real service supervisor it will spawn them itself (system_spawn).
|
||||
startInitrdBinaries(boot_information);
|
||||
|
||||
// Become the idle task: drop below every real task and halt until an
|
||||
// interrupt. The timer keeps preempting into init and any other work.
|
||||
scheduler.setPriority(0);
|
||||
status("\nkernel idle; /sbin/init is running.\n");
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
/// Spawn every program bundled in the initrd as its own ring-3 process. A bad
|
||||
/// image or a program that fails to load is logged and skipped — the rest of the
|
||||
/// system still runs.
|
||||
fn startInitrdBinaries(boot_information: *const danos.BootInformation) void {
|
||||
if (boot_information.initrd_len == 0) return;
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
|
||||
const rd = initrd.Reader.init(image) orelse {
|
||||
status("initrd: bad image, skipping\n");
|
||||
return;
|
||||
};
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
statusPrint("starting /sbin/{s} (from initrd)...\n", .{item.name});
|
||||
process.spawnProcess(item.blob, 4) catch |err| {
|
||||
statusPrint("initrd: {s} failed to load: {s}\n", .{ item.name, @errorName(err) });
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/// Wake the application processors the firmware left parked. Allocates the low
|
||||
/// trampoline page (and makes it executable), then wakes each non-boot core in turn,
|
||||
/// handing it a fresh kernel stack and its per-CPU slot. Cores that don't report in
|
||||
/// are left parked — the running system is unaffected. See docs/smp.md.
|
||||
fn bringUpSecondaries() void {
|
||||
const cores = platform.cpus();
|
||||
if (cores.len <= 1) return;
|
||||
|
||||
// A low (<1 MiB) frame was reserved at boot for the real-mode trampoline (a SIPI
|
||||
// vector addresses it). It's kept for the system's life — armed only during a
|
||||
// wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later.
|
||||
if (ap_trampoline_page == 0) {
|
||||
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
|
||||
return;
|
||||
}
|
||||
architecture.setTrampolinePage(ap_trampoline_page);
|
||||
architecture.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop
|
||||
|
||||
// Test hook: the smp-retry case forces the first wake to fail, so the retry below
|
||||
// must still bring every core online. Inert in a normal build (test_case is null).
|
||||
if (build_options.test_case) |tc| {
|
||||
if (std.mem.eql(u8, tc, "smp-retry")) architecture.testFailNextWakes(1);
|
||||
}
|
||||
|
||||
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||
const maximum_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
|
||||
for (cores[1..], 1..) |core, index| {
|
||||
const stack = heap.allocator().alloc(u8, parameters.kernel_stack_size) catch {
|
||||
log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id});
|
||||
continue;
|
||||
};
|
||||
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
|
||||
// This core's dedicated fault stack — allocated only now that the core is
|
||||
// real, rather than reserved statically for every possible core.
|
||||
const fault_stack = heap.allocator().alloc(u8, architecture.fault_stack_size) catch {
|
||||
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||
continue;
|
||||
};
|
||||
architecture.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
||||
const pc = scheduler.prepareSecondary(index, core.apic_id);
|
||||
var attempt: u32 = 1;
|
||||
while (attempt <= maximum_wake_attempts) : (attempt += 1) {
|
||||
if (architecture.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
||||
pc.online = true;
|
||||
log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
|
||||
break;
|
||||
}
|
||||
if (attempt == maximum_wake_attempts)
|
||||
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, maximum_wake_attempts });
|
||||
}
|
||||
}
|
||||
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||
}
|
||||
|
||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
||||
/// touches the framebuffer.
|
||||
fn status(message: []const u8) void {
|
||||
log.write(message);
|
||||
console.write(message);
|
||||
}
|
||||
|
||||
fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
||||
var buffer: [256]u8 = undefined;
|
||||
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// Frames (4 KiB pages) to whole MiB.
|
||||
fn mib(pages: u64) u64 {
|
||||
return pages * danos.page_size / (1024 * 1024);
|
||||
}
|
||||
|
||||
fn kib(frames: u64) u64 {
|
||||
return frames * danos.page_size / (1024);
|
||||
}
|
||||
|
||||
/// Report a CPU exception and halt **this core**. There's no fault recovery yet, so
|
||||
/// the faulting core is terminal — but the fault is *contained* to it: on an
|
||||
/// application processor only that core stops, and the rest of the system keeps
|
||||
/// running (full recovery — kill the task, keep the core — is the resilience track,
|
||||
/// see docs/resilience.md). The report names the core so an AP fault is attributed,
|
||||
/// and goes to every output sink plus a POST code and a persistent breadcrumb.
|
||||
fn onException(state: *const architecture.CpuState) noreturn {
|
||||
log.checkpoint(cp_exception);
|
||||
const core = scheduler.currentCpuIndex();
|
||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
||||
// top of the diagnostic log.
|
||||
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
|
||||
var buffer: [128]u8 = undefined;
|
||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
|
||||
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
|
||||
/// no live console), then halt. Assumes no console — the sinks self-guard.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(message: []const u8, first_trace_address: ?usize) noreturn {
|
||||
_ = first_trace_address;
|
||||
log.checkpoint(cp_panic);
|
||||
log.recordPanic(message);
|
||||
status("\nKERNEL PANIC: ");
|
||||
status(message);
|
||||
status("\n");
|
||||
architecture.halt();
|
||||
}
|
||||
}.panic);
|
||||
@@ -0,0 +1,174 @@
|
||||
//! Physical memory manager: a bitmap frame allocator. It hands out and reclaims
|
||||
//! 4 KiB physical frames — the primitive every later memory feature (page
|
||||
//! tables, the heap) is built on top of.
|
||||
//!
|
||||
//! This is generic kernel code: it works on the neutral `danos.MemoryRegion`
|
||||
//! array the loader hands over (see docs/memory-map.md), so it carries no UEFI
|
||||
//! and nothing architecture-specific beyond the 4 KiB page.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
/// One bit per frame, covering physical RAM from 0 up to the highest usable
|
||||
/// address: 1 = used/unavailable, 0 = free. The bitmap itself lives in a frame
|
||||
/// we carve out of usable memory during init.
|
||||
var bitmap: []u8 = &.{};
|
||||
var total_frames: usize = 0;
|
||||
var used_frames: usize = 0;
|
||||
/// Where the next allocation scan begins, so we don't rescan from frame 0 every
|
||||
/// time. Pulled back on free() so reclaimed low frames get reused.
|
||||
var next_hint: usize = 0;
|
||||
|
||||
pub const Stats = struct {
|
||||
total_frames: usize,
|
||||
used_frames: usize,
|
||||
free_frames: usize,
|
||||
};
|
||||
|
||||
pub fn stats() Stats {
|
||||
return .{
|
||||
.total_frames = total_frames,
|
||||
.used_frames = used_frames,
|
||||
.free_frames = total_frames - used_frames,
|
||||
};
|
||||
}
|
||||
|
||||
inline fn bit(frame: usize) u3 {
|
||||
return @intCast(frame & 7);
|
||||
}
|
||||
inline fn isUsed(frame: usize) bool {
|
||||
return (bitmap[frame >> 3] >> bit(frame)) & 1 != 0;
|
||||
}
|
||||
inline fn setUsed(frame: usize) void {
|
||||
bitmap[frame >> 3] |= @as(u8, 1) << bit(frame);
|
||||
}
|
||||
inline fn setFree(frame: usize) void {
|
||||
bitmap[frame >> 3] &= ~(@as(u8, 1) << bit(frame));
|
||||
}
|
||||
|
||||
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(map.regions)))[0..map.len];
|
||||
}
|
||||
|
||||
/// Build the allocator from the loader's memory map. Reaches physical memory
|
||||
/// (the region array, the bitmap's own storage) through the physmap, which the
|
||||
/// loader's bootstrap tables already provide — so this works before the kernel
|
||||
/// installs its own tables. Invariant: the bitmap lands in the first usable
|
||||
/// region (lowest address), which must sit under the bootstrap physmap's reach
|
||||
/// (4 GiB); it always does, as both this and the page-table allocator scan from
|
||||
/// low addresses up.
|
||||
pub fn init(map: danos.MemoryMap) void {
|
||||
const regs = regions(map);
|
||||
|
||||
// 1. Size the bitmap to cover every frame up to the highest RAM address —
|
||||
// including reserved RAM, so those frames are trackable (e.g. to free the
|
||||
// boot buffers later). Only MMIO (device address space) is excluded.
|
||||
// Everything starts unallocatable; usable regions are freed below.
|
||||
var highest: u64 = 0;
|
||||
for (regs) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
const end = r.base + r.pages * page_size;
|
||||
if (end > highest) highest = end;
|
||||
}
|
||||
total_frames = @intCast(highest / page_size);
|
||||
if (total_frames == 0) @panic("pmm: no usable memory");
|
||||
const bitmap_bytes = (total_frames + 7) / 8;
|
||||
const bitmap_pages = (bitmap_bytes + page_size - 1) / page_size;
|
||||
|
||||
// 2. Park the bitmap in the first usable region large enough to hold it.
|
||||
// Start at least one page in, so we never place it on frame 0 (which is
|
||||
// kept reserved as the "none" address, and is an awkward pointer besides).
|
||||
var storage: ?u64 = null;
|
||||
for (regs) |r| {
|
||||
if (r.kind != .usable) continue;
|
||||
const base = if (r.base == 0) page_size else r.base;
|
||||
const skipped = (base - r.base) / page_size;
|
||||
if (r.pages - skipped >= bitmap_pages) {
|
||||
storage = base;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
|
||||
bitmap = @as([*]u8, @ptrFromInt(danos.physicalToVirtual(bitmap_base)))[0..bitmap_bytes];
|
||||
|
||||
// 3. Start with everything marked used, then free the usable regions. Doing
|
||||
// it this way means every gap, reserved span and MMIO hole is unallocatable
|
||||
// by default — we only ever hand back memory the firmware called usable.
|
||||
@memset(bitmap, 0xff);
|
||||
used_frames = total_frames;
|
||||
for (regs) |r| {
|
||||
if (r.kind != .usable) continue;
|
||||
var f: usize = @intCast(r.base / page_size);
|
||||
const end = f + @as(usize, @intCast(r.pages));
|
||||
while (f < end and f < total_frames) : (f += 1) {
|
||||
setFree(f);
|
||||
used_frames -= 1;
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Take back the frames the bitmap occupies, and frame 0 — so a 0 result
|
||||
// stays reserved to mean "no frame".
|
||||
reserve(bitmap_base, bitmap_pages);
|
||||
reserve(0, 1);
|
||||
}
|
||||
|
||||
/// Mark `count` frames from physical `base` as used, counting only those that
|
||||
/// were actually free.
|
||||
fn reserve(base: u64, count: usize) void {
|
||||
var f: usize = @intCast(base / page_size);
|
||||
const end = f + count;
|
||||
while (f < end and f < total_frames) : (f += 1) {
|
||||
if (!isUsed(f)) {
|
||||
setUsed(f);
|
||||
used_frames += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate one physical frame, or null if none are free. The address is
|
||||
/// page-aligned; the frame's contents are undefined.
|
||||
pub fn alloc() ?u64 {
|
||||
var scanned: usize = 0;
|
||||
var f = next_hint;
|
||||
while (scanned < total_frames) : (scanned += 1) {
|
||||
if (f >= total_frames) f = 0;
|
||||
if (!isUsed(f)) {
|
||||
setUsed(f);
|
||||
used_frames += 1;
|
||||
next_hint = f + 1;
|
||||
return @as(u64, f) * page_size;
|
||||
}
|
||||
f += 1;
|
||||
}
|
||||
return null; // out of physical memory
|
||||
}
|
||||
|
||||
/// Allocate one free frame whose physical address is below `limit`, or null if
|
||||
/// none is free down there. The AP trampoline needs this: an x86 STARTUP IPI vectors
|
||||
/// a waking core to physical `vector << 12`, and `vector` is a byte — so the
|
||||
/// trampoline must live under 1 MiB. A short linear scan of the low frames; only run
|
||||
/// a handful of times at boot, so it needn't be fast.
|
||||
pub fn allocBelow(limit: u64) ?u64 {
|
||||
const cap = @min(total_frames, @as(usize, @intCast(limit / page_size)));
|
||||
var f: usize = 1; // frame 0 stays reserved as the "none" address
|
||||
while (f < cap) : (f += 1) {
|
||||
if (!isUsed(f)) {
|
||||
setUsed(f);
|
||||
used_frames += 1;
|
||||
return @as(u64, f) * page_size;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Return a frame obtained from alloc() to the pool. Bogus or double frees are
|
||||
/// ignored rather than corrupting the count.
|
||||
pub fn free(address: u64) void {
|
||||
const f: usize = @intCast(address / page_size);
|
||||
if (f >= total_frames or !isUsed(f)) return;
|
||||
setFree(f);
|
||||
used_frames -= 1;
|
||||
if (f < next_hint) next_hint = f;
|
||||
}
|
||||
@@ -0,0 +1,579 @@
|
||||
//! User-space processes: loading a user ELF and running it in ring 3. danos is a
|
||||
//! microkernel, so this only ever loads *user* binaries — there is no kernel-space
|
||||
//! loader; in-kernel code is linked into the kernel image, not loaded here.
|
||||
//!
|
||||
//! Two entry points:
|
||||
//! - `spawnProcess` loads a user ELF (`/sbin/init`, and later servers/drivers)
|
||||
//! into a fresh address space and schedules it as a real preemptive ring-3
|
||||
//! process on its own page tables. This is the production path.
|
||||
//! - `run` executes a raw code blob (the user-pf isolation test program) on the
|
||||
//! *current* kernel context via the borrowed-thread path — a minimal probe of
|
||||
//! the ring-transition mechanisms, kept for that test.
|
||||
//! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program
|
||||
//! talks to the kernel only through the system_call instruction (or the int 0x80
|
||||
//! gate). The shared handler is installed once by `init`.
|
||||
//!
|
||||
//! Borrowed-path caveat (`run` only): it publishes TSS.rsp0 on the *current*
|
||||
//! core and uses a single global unwind slot (`user_saved_rsp` in isr.s), so the
|
||||
//! caller must disable preemption and only one core may be inside it at a time.
|
||||
//! Real processes (`spawnProcess`) have none of these limits — the scheduler
|
||||
//! maintains rsp0/CR3 per switch.
|
||||
|
||||
const std = @import("std");
|
||||
const elf = std.elf;
|
||||
const danos = @import("danos");
|
||||
const architecture = @import("architecture");
|
||||
const pmm = @import("pmm.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const ipc = @import("ipc-synchronous.zig");
|
||||
const device_service = @import("device-service.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const log = @import("log.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const SystemCall = danos.SystemCall;
|
||||
|
||||
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
||||
/// the identity map (low indices) and the vmm test address (index 128), so
|
||||
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
|
||||
/// An ELF image may occupy [code_virtual, stack_virtual); the stack page sits above.
|
||||
pub const code_virtual: u64 = 0x0000_7000_0000_0000;
|
||||
pub const stack_virtual: u64 = 0x0000_7000_0020_0000;
|
||||
|
||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
|
||||
/// 1 GiB window is far more than any user heap needs today.
|
||||
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
||||
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
||||
|
||||
/// End of the user (low) canonical half. Any legitimate user pointer is below it;
|
||||
/// used to bound the addresses a system_call will dereference on the caller's behalf.
|
||||
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
|
||||
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
||||
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
||||
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
|
||||
/// `Task.device_map_next`.
|
||||
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
|
||||
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||
|
||||
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
|
||||
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
||||
const maximum_mmap_pages = 256;
|
||||
|
||||
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
|
||||
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
|
||||
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
|
||||
|
||||
/// The isolation-proof program: reads a kernel-only page, must #PF.
|
||||
pub fn pfBlob() []const u8 {
|
||||
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
|
||||
}
|
||||
|
||||
/// What debug_write syscalls produced (accumulated), and the exit system_call's code.
|
||||
pub var write_buffer: [256]u8 = undefined;
|
||||
pub var write_len: usize = 0;
|
||||
pub var write_from_user: bool = false;
|
||||
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
|
||||
pub var exit_code: u64 = 0;
|
||||
|
||||
/// The system_call surface, dispatched on the saved system_call number (`danos.SystemCall`).
|
||||
/// This is the microkernel-minimal set — memory + scheduling only; file/device
|
||||
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
|
||||
/// written back into the trap frame, since the entry paths restore user registers
|
||||
/// from it. One handler serves both the system_call/sysret and int-0x80 entry paths.
|
||||
///
|
||||
/// Install it once at boot (before any user code runs) via `init`.
|
||||
pub fn init() void {
|
||||
architecture.setSystemCallHandler(system_call);
|
||||
}
|
||||
|
||||
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
||||
fn fail(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, @bitCast(@as(i64, -1)));
|
||||
}
|
||||
|
||||
fn system_call(state: *architecture.CpuState) void {
|
||||
switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) {
|
||||
.exit => {
|
||||
exit_code = architecture.systemCallArg(state, 0);
|
||||
// A scheduled process drops its endpoint references, frees its address
|
||||
// space, and reschedules; a borrowed test thread unwinds back to the
|
||||
// kernel that entered it.
|
||||
if (scheduler.currentIsUserProcess()) {
|
||||
// Unbind before closeHandles: dropping the last reference destroys the
|
||||
// Endpoint, and a still-bound GSI would have an ISR call
|
||||
// notifyFromIsr on freed memory the next time the device fired.
|
||||
// unbindAll also leaves the line masked, so a dead driver's device
|
||||
// goes quiet rather than storming.
|
||||
releaseIrqs(scheduler.current());
|
||||
ipc.closeHandles(scheduler.current());
|
||||
scheduler.exitUser();
|
||||
} else architecture.userExit();
|
||||
},
|
||||
.yield => {
|
||||
scheduler.yield();
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
},
|
||||
.sleep => {
|
||||
scheduler.sleep(architecture.systemCallArg(state, 0));
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
},
|
||||
.debug_write => systemDebugWrite(state),
|
||||
.mmap => systemMmap(state),
|
||||
.munmap => systemMunmap(state),
|
||||
.create_endpoint => systemCreateEndpoint(state),
|
||||
.ipc_register => systemIpcRegister(state),
|
||||
.ipc_lookup => systemIpcLookup(state),
|
||||
.ipc_call => systemIpcCall(state),
|
||||
.ipc_reply_wait => systemIpcReplyWait(state),
|
||||
.device_enumerate => systemDeviceEnumerate(state),
|
||||
.device_claim => systemDeviceClaim(state),
|
||||
.mmio_map => systemMmioMap(state),
|
||||
.irq_bind => systemIrqBind(state),
|
||||
.irq_ack => systemIrqAck(state),
|
||||
.device_register => systemDeviceRegister(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
|
||||
/// Return `-errno` in the system_call result register.
|
||||
fn failErr(state: *architecture.CpuState, errno: i64) void {
|
||||
architecture.setSystemCallResult(state, @bitCast(-errno));
|
||||
}
|
||||
|
||||
/// create_endpoint() -> handle: allocate an endpoint and install it in the
|
||||
/// caller's handle table.
|
||||
fn systemCreateEndpoint(state: *architecture.CpuState) void {
|
||||
const endpoint = ipc.createEndpoint() orelse return failErr(state, ipc.ENOMEM);
|
||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||
if (h < 0) {
|
||||
ipc.dropRef(endpoint);
|
||||
return failErr(state, ipc.ENOSPC);
|
||||
}
|
||||
architecture.setSystemCallResult(state, @intCast(h));
|
||||
}
|
||||
|
||||
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
||||
/// well-known id so other processes can find it.
|
||||
fn systemIpcRegister(state: *architecture.CpuState) void {
|
||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
|
||||
}
|
||||
|
||||
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
||||
/// handle to it in the caller.
|
||||
fn systemIpcLookup(state: *architecture.CpuState) void {
|
||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||
if (h < 0) {
|
||||
ipc.dropRef(endpoint);
|
||||
return failErr(state, ipc.ENOSPC);
|
||||
}
|
||||
architecture.setSystemCallResult(state, @intCast(h));
|
||||
}
|
||||
|
||||
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
|
||||
/// Blocks until the server replies; the trap frame lives on this task's kernel
|
||||
/// stack, so it survives the block and receives the result on resume.
|
||||
fn systemIpcCall(state: *architecture.CpuState) void {
|
||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const r = ipc.call(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4));
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
}
|
||||
|
||||
/// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len,
|
||||
/// with the sender's badge in the secondary result register (rdx).
|
||||
fn systemIpcReplyWait(state: *architecture.CpuState) void {
|
||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
var badge: u64 = 0;
|
||||
const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), &badge);
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
architecture.setSystemCallResult2(state, badge);
|
||||
}
|
||||
|
||||
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
|
||||
/// buffer (up to `maximum` entries), returning the total device count.
|
||||
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||
const maximum = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
const sz = @sizeOf(danos.DeviceDescriptor);
|
||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||
const out: [*]danos.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
||||
architecture.setSystemCallResult(state, device_service.enumerate(out[0..@intCast(cap)]));
|
||||
}
|
||||
|
||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
if (device_service.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
||||
architecture.setSystemCallResult(state, 0)
|
||||
else
|
||||
fail(state);
|
||||
}
|
||||
|
||||
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
||||
/// this address space (strong-uncacheable) and return the register base address.
|
||||
/// The claim is the capability — a process can only map hardware it owns.
|
||||
fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const resource_index = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const owner = device_service.ownerOf(device_id) orelse return fail(state);
|
||||
if (owner != t.id) return fail(state); // not claimed by this process
|
||||
const r = device_service.resourceOf(device_id, resource_index) orelse return fail(state);
|
||||
if (r.kind != @intFromEnum(danos.ResourceKind.memory)) return fail(state);
|
||||
|
||||
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
||||
const first = r.start & ~@as(u64, page_size - 1);
|
||||
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
|
||||
const pages = (last - first) / page_size + 1;
|
||||
const base_v = t.device_map_next;
|
||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||
|
||||
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len);
|
||||
t.device_map_next = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||
}
|
||||
|
||||
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
||||
/// this process has claimed. The bus-driver primitive: a process that owns a bus
|
||||
/// enumerates it and hands each device it finds to the table, where a class driver
|
||||
/// can claim it.
|
||||
///
|
||||
/// The kernel copies the descriptor into a kernel local *once* (via the same
|
||||
/// physmap-walking path as IPC, so an unmapped user page fails the call rather than
|
||||
/// faulting the kernel), then validates and uses that copy — no second read of user
|
||||
/// memory, so nothing it checked can change under it. It refuses any child resource
|
||||
/// that escapes the parent's windows: a descriptor is a licence to map physical
|
||||
/// memory, so a bus may only subdivide what it already holds. `id`/`parent` in the
|
||||
/// supplied descriptor are ignored.
|
||||
fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||
const parent_id = architecture.systemCallArg(state, 0);
|
||||
const descriptor_ptr = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
|
||||
var descriptor: danos.DeviceDescriptor = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
|
||||
const id = device_service.register(parent_id, t.id, &descriptor) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, id);
|
||||
}
|
||||
|
||||
/// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed
|
||||
/// (which is what frees the endpoints an ISR would otherwise notify into).
|
||||
fn releaseIrqs(t: *scheduler.Task) void {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
irq.releaseOwner(t.id);
|
||||
}
|
||||
|
||||
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
||||
/// The two checks are the whole security story: the device must be *claimed* by the
|
||||
/// caller, and the resource must be one of that device's `irq` resources as recorded
|
||||
/// by discovery. Neither a raw GSI nor an unclaimed device can get through — which
|
||||
/// is why irq_bind takes a resource index and not an interrupt number.
|
||||
fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
||||
const owner = device_service.ownerOf(device_id) orelse return null;
|
||||
if (owner != t.id) return null;
|
||||
const r = device_service.resourceOf(device_id, resource_index) orelse return null;
|
||||
if (r.kind != @intFromEnum(danos.ResourceKind.irq)) return null;
|
||||
if (r.start >= irq.maximum_gsi) return null;
|
||||
return @intCast(r.start);
|
||||
}
|
||||
|
||||
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
|
||||
/// endpoint as an asynchronous IPC notification. The driver then blocks in
|
||||
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
||||
fn systemIrqBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
irq.bind(gsi, endpoint, t.id) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// irq_ack(device_id, resource_index) -> 0/-1: re-arm a bound IRQ. The ISR left the line
|
||||
/// masked (it could not quiet the device — that's this driver's job), so nothing
|
||||
/// more arrives until the driver says it has serviced the hardware.
|
||||
fn systemIrqAck(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (irq.ack(gsi)) architecture.setSystemCallResult(state, 0) else fail(state);
|
||||
}
|
||||
|
||||
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
|
||||
/// A bring-up diagnostic — real output goes through the VFS/console later.
|
||||
///
|
||||
/// The pointer must lie in the user (low) half, so kernel addresses and
|
||||
/// non-canonical values fall outside it and the read below can't be steered at
|
||||
/// kernel data. Length is checked first so the upper-bound add can't overflow.
|
||||
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
|
||||
/// the user half passes the check and the read #PFs -> on_fault halts — a
|
||||
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
|
||||
fn systemDebugWrite(state: *architecture.CpuState) void {
|
||||
const ptr = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
||||
const source: [*]const u8 = @ptrFromInt(ptr);
|
||||
@memcpy(write_buffer[0..len], source[0..len]); // keep the latest message
|
||||
write_len = len;
|
||||
write_from_user = architecture.fromUser(state);
|
||||
write_count += 1;
|
||||
log.write("DANOS-INIT: ");
|
||||
log.write(source[0..len]);
|
||||
architecture.setSystemCallResult(state, len);
|
||||
} else {
|
||||
fail(state);
|
||||
}
|
||||
}
|
||||
|
||||
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
|
||||
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
||||
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
||||
/// always RW+NX; W^X for user code stays with the ELF loader). Failure returns
|
||||
/// -1. The user-space allocator (lib `runtime`) carves these pages into malloc blocks.
|
||||
fn systemMmap(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
|
||||
|
||||
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
|
||||
const base = t.heap_next;
|
||||
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||
|
||||
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
|
||||
// (no partially-mapped grant leaks into the address space).
|
||||
var frames: [maximum_mmap_pages]u64 = undefined;
|
||||
var got: usize = 0;
|
||||
while (got < pages) : (got += 1) {
|
||||
frames[got] = pmm.alloc() orelse {
|
||||
for (frames[0..got]) |f| pmm.free(f);
|
||||
return fail(state);
|
||||
};
|
||||
}
|
||||
|
||||
for (frames[0..pages], 0..) |frame, i| {
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||
architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
|
||||
}
|
||||
t.heap_next = base + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base);
|
||||
}
|
||||
|
||||
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
||||
/// each page and frees its frame. The arena is a bump allocator, so the virtual
|
||||
/// range is not recycled (the user-space allocator reuses freed *blocks* itself);
|
||||
/// this just returns the physical frames to the kernel. Returns 0, or -1 if the
|
||||
/// range is not page-aligned or lies outside the arena.
|
||||
fn systemMunmap(state: *architecture.CpuState) void {
|
||||
const base = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or base % page_size != 0) return fail(state);
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base + i * page_size;
|
||||
if (architecture.translate(t.aspace, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.aspace, va);
|
||||
pmm.free(physical);
|
||||
}
|
||||
}
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// Reset the recorded system_call evidence before a user-mode run.
|
||||
fn resetRecords() void {
|
||||
write_len = 0;
|
||||
write_from_user = false;
|
||||
write_count = 0;
|
||||
exit_code = 0;
|
||||
}
|
||||
|
||||
pub const RunError = error{ ProgramTooBig, OutOfMemory };
|
||||
|
||||
/// Map `blob` at code_virtual with a fresh user stack, drop to ring 3, and return
|
||||
/// once the program exits via system_call 0. See the migration caveat in the module
|
||||
/// doc. A program that faults instead never returns (on_fault halts the core).
|
||||
pub fn run(blob: []const u8) RunError!void {
|
||||
if (blob.len > page_size) return error.ProgramTooBig;
|
||||
const code_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const stack_frame = pmm.alloc() orelse {
|
||||
pmm.free(code_frame);
|
||||
return error.OutOfMemory;
|
||||
};
|
||||
|
||||
// Fill the code frame through the physmap (supervisor RW): the user-facing
|
||||
// mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is
|
||||
// padded with int3 so a stray jump traps instead of sliding.
|
||||
const code: [*]u8 = @ptrFromInt(danos.physicalToVirtual(code_frame));
|
||||
@memcpy(code[0..blob.len], blob);
|
||||
@memset(code[blob.len..page_size], 0xCC);
|
||||
|
||||
architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X
|
||||
architecture.mapUserPage(stack_virtual, stack_frame, true, false); // RW + NX
|
||||
resetRecords();
|
||||
|
||||
architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_virtual + page_size);
|
||||
|
||||
// Back via the exit system_call; the interrupt gate left IF clear.
|
||||
architecture.enableInterrupts();
|
||||
architecture.unmapPage(code_virtual);
|
||||
architecture.unmapPage(stack_virtual);
|
||||
pmm.free(code_frame);
|
||||
pmm.free(stack_frame);
|
||||
}
|
||||
|
||||
// --- user ELF loading (/sbin/init) ------------------------------------------
|
||||
|
||||
pub const InitError = error{
|
||||
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
|
||||
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
|
||||
BadEntry, // e_entry not inside an executable segment
|
||||
ProgramTooBig, // more pages than the loader's budget
|
||||
OutOfMemory,
|
||||
};
|
||||
|
||||
const maximum_segments = 16;
|
||||
const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
||||
|
||||
const Segment = struct {
|
||||
vaddr: u64,
|
||||
memsz: u64,
|
||||
filesz: u64,
|
||||
off: u64,
|
||||
writable: bool,
|
||||
executable: bool,
|
||||
|
||||
fn pages(self: Segment) u64 {
|
||||
return (self.memsz + page_size - 1) / page_size;
|
||||
}
|
||||
};
|
||||
|
||||
/// Parse and validate every PT_LOAD before touching memory. Bounds are checked
|
||||
/// against the image and the user region; segments must be page-aligned,
|
||||
/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only
|
||||
/// segment, mapped RO+NX).
|
||||
fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!struct { count: usize, entry: u64 } {
|
||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf;
|
||||
const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]);
|
||||
if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf;
|
||||
if (image[elf.EI_CLASS] != elf.ELFCLASS64) return error.BadElf;
|
||||
if (ehdr.e_machine != .X86_64) return error.BadElf;
|
||||
if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation
|
||||
if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf;
|
||||
if (ehdr.e_phnum > maximum_segments) return error.BadElf;
|
||||
const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize;
|
||||
if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf;
|
||||
|
||||
var count: usize = 0;
|
||||
var total_pages: u64 = 0;
|
||||
for (0..ehdr.e_phnum) |i| {
|
||||
const off = ehdr.e_phoff + i * ehdr.e_phentsize;
|
||||
const phdr = std.mem.bytesToValue(elf.Elf64_Phdr, image[off..][0..@sizeOf(elf.Elf64_Phdr)]);
|
||||
if (phdr.p_type != elf.PT_LOAD) continue;
|
||||
if (phdr.p_memsz == 0) continue;
|
||||
|
||||
if (phdr.p_vaddr % page_size != 0) return error.BadSegment;
|
||||
if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment;
|
||||
if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment;
|
||||
// Inside the user image region, strictly below the stack page.
|
||||
if (phdr.p_vaddr < code_virtual) return error.BadSegment;
|
||||
if (phdr.p_memsz > stack_virtual - phdr.p_vaddr) return error.BadSegment;
|
||||
|
||||
const w = phdr.p_flags & elf.PF_W != 0;
|
||||
const x = phdr.p_flags & elf.PF_X != 0;
|
||||
if (w and x) return error.BadSegment; // W^X, even for init
|
||||
|
||||
const seg = Segment{
|
||||
.vaddr = phdr.p_vaddr,
|
||||
.memsz = phdr.p_memsz,
|
||||
.filesz = phdr.p_filesz,
|
||||
.off = phdr.p_offset,
|
||||
.writable = w,
|
||||
.executable = x,
|
||||
};
|
||||
// No overlap with any earlier segment (page-granular, since mapping is).
|
||||
for (segs[0..count]) |other| {
|
||||
const a_end = seg.vaddr + seg.pages() * page_size;
|
||||
const b_end = other.vaddr + other.pages() * page_size;
|
||||
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
|
||||
}
|
||||
total_pages += seg.pages();
|
||||
if (total_pages > maximum_pages) return error.ProgramTooBig;
|
||||
segs[count] = seg;
|
||||
count += 1;
|
||||
}
|
||||
if (count == 0) return error.BadElf;
|
||||
|
||||
// The entry point must land inside an executable segment.
|
||||
for (segs[0..count]) |seg| {
|
||||
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
|
||||
return .{ .count = count, .entry = ehdr.e_entry };
|
||||
}
|
||||
return error.BadEntry;
|
||||
}
|
||||
|
||||
/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed
|
||||
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
||||
/// On a later failure the whole address space is torn down, which frees every
|
||||
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0);
|
||||
const page_off = page_index * page_size;
|
||||
if (page_off < seg.filesz) {
|
||||
const n = @min(page_size, seg.filesz - page_off);
|
||||
@memcpy(destination[0..n], image[seg.off + page_off ..][0..n]);
|
||||
}
|
||||
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
||||
}
|
||||
|
||||
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
||||
/// ring-3 process at `priority`. Returns immediately — the process runs
|
||||
/// preemptively on its own page tables alongside everything else, and its exit
|
||||
/// is handled by the system_call layer. The whole build (address space + ELF load +
|
||||
/// task) runs under the kernel lock so it appears atomically and can't race
|
||||
/// pmm/heap on another core.
|
||||
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
|
||||
var segs: [maximum_segments]Segment = undefined;
|
||||
const parsed = try parseSegments(image, &segs);
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
||||
errdefer architecture.destroyAddressSpace(aspace);
|
||||
|
||||
for (segs[0..parsed.count]) |seg| {
|
||||
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
||||
}
|
||||
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
architecture.mapUserPageInto(aspace, stack_virtual, stack_frame, true, false); // RW + NX
|
||||
|
||||
if (!scheduler.spawnUserLocked(aspace, parsed.entry, stack_virtual + page_size, priority))
|
||||
return error.OutOfMemory;
|
||||
}
|
||||
@@ -0,0 +1,561 @@
|
||||
//! The scheduler: fixed-priority preemptive multitasking.
|
||||
//!
|
||||
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
||||
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
||||
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
||||
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
||||
//! docs/vision.md).
|
||||
//!
|
||||
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
||||
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
||||
//! makes those two paths coexist.
|
||||
//!
|
||||
//! Cross-core safety is the **big kernel lock** (`sync.zig`): every critical
|
||||
//! section here runs under it, and it is held across a context switch and released
|
||||
//! by the task that resumes (see sync.zig's hand-off rule). On a single core the
|
||||
//! lock is never contended, so the behaviour is exactly the old interrupt-flag
|
||||
//! model; it's what lets a second core enter `schedule()` without corrupting the
|
||||
//! shared queues.
|
||||
|
||||
const std = @import("std");
|
||||
const parameters = @import("parameters");
|
||||
const architecture = @import("architecture");
|
||||
const heap = @import("heap.zig");
|
||||
const sync = @import("sync.zig");
|
||||
|
||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
||||
pub const Priority = u3;
|
||||
const number_priorities = 8;
|
||||
|
||||
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
|
||||
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
|
||||
|
||||
const State = enum { free, ready, running, blocked };
|
||||
|
||||
pub const Task = struct {
|
||||
id: u32 = 0,
|
||||
state: State = .free,
|
||||
priority: Priority = 0,
|
||||
sp: usize = 0, // saved stack pointer, valid while not running
|
||||
stack: []u8 = &.{},
|
||||
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
|
||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
||||
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
|
||||
// Physical root of this task's address space, or 0 for a kernel task (which
|
||||
// runs on the shared kernel page tables). A user task carries its own.
|
||||
aspace: u64 = 0,
|
||||
user_ip: u64 = 0, // user-mode entry point (user task only)
|
||||
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
||||
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
|
||||
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
||||
// as the user heap grows; user task only.
|
||||
heap_next: u64 = 0,
|
||||
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
|
||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
||||
device_map_next: u64 = 0,
|
||||
// --- synchronous IPC (ipc_sync.zig) ---
|
||||
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
||||
// opaque here so the scheduler and IPC modules don't import each other.
|
||||
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
|
||||
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||
ipc_client: ?*Task = null,
|
||||
ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS)
|
||||
ipc_send_len: u64 = 0,
|
||||
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
|
||||
ipc_reply_cap: u64 = 0,
|
||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||
};
|
||||
|
||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
pub const ipc_maximum_handles = 16;
|
||||
|
||||
var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||
var next_id: u32 = 1;
|
||||
|
||||
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
||||
/// queue of tasks **pinned** to it. One entry per core; the architecture layer stashes a
|
||||
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
|
||||
/// with a single read and no lock.
|
||||
///
|
||||
/// Most work stays in the **global** ready queue (below), which any idle core pulls
|
||||
/// from — work-conserving. A task given an *affinity* instead goes to that core's
|
||||
/// `pinned_*` queue and is only ever run there (no surprise migration — the more
|
||||
/// real-time-predictable model, docs/smp.md). The two queues are merged at selection
|
||||
/// time. Both are still mutated only under the big kernel lock, so one core enqueuing
|
||||
/// into another core's pinned queue is safe.
|
||||
pub const PerCpu = struct {
|
||||
current: *Task = undefined, // the task running on this core
|
||||
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
||||
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
|
||||
index: u32 = 0, // dense 0-based core index
|
||||
online: bool = false, // has this core finished bring-up?
|
||||
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
||||
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||
pinned_bitmap: u8 = 0,
|
||||
};
|
||||
|
||||
const maximum_cpus = parameters.maximum_cpus;
|
||||
var cpus = [_]PerCpu{.{}} ** maximum_cpus;
|
||||
|
||||
/// This core's per-CPU state, via the architecture layer's GS-base pointer. Valid only once
|
||||
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
|
||||
inline fn thisCpu() *PerCpu {
|
||||
return @ptrFromInt(architecture.cpuLocal());
|
||||
}
|
||||
|
||||
/// The task running on this core — the per-CPU replacement for the old global
|
||||
/// `current`. A convenience reader; writes go through `thisCpu().current`.
|
||||
pub inline fn current() *Task {
|
||||
return thisCpu().current;
|
||||
}
|
||||
|
||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
|
||||
// are shared across all cores and mutated only under the big kernel lock.
|
||||
var ready_head: [number_priorities]?*Task = .{null} ** number_priorities;
|
||||
var ready_tail: [number_priorities]?*Task = .{null} ** number_priorities;
|
||||
var ready_bitmap: u8 = 0;
|
||||
|
||||
var preemption_enabled = true;
|
||||
|
||||
/// Bring up scheduling on the bootstrap processor: register the currently-running
|
||||
/// kernel context as task 0, publish this core's per-CPU state (via the GS base),
|
||||
/// give the core an idle task, and hook the timer for preemption. Runs once, at
|
||||
/// boot, before interrupts are enabled — so no lock is needed here.
|
||||
pub fn init(boot_priority: Priority) void {
|
||||
const pc = &cpus[0];
|
||||
pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() };
|
||||
architecture.setCpuLocal(0, @intFromPtr(pc));
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
pc.current = &tasks[0];
|
||||
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
|
||||
architecture.setTickHook(tick);
|
||||
}
|
||||
|
||||
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
||||
/// for the next interrupt at near-zero power (see docs/halting.md).
|
||||
fn idle() void {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
/// Reserve and initialise the per-CPU slot for an application processor at dense
|
||||
/// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a
|
||||
/// pointer the architecture bring-up hands to the core (it publishes it in its GS base).
|
||||
/// Called on the BSP before waking each AP; the AP marks itself `online`.
|
||||
pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu {
|
||||
const pc = &cpus[index];
|
||||
pc.* = .{ .index = @intCast(index), .hw_id = hw_id, .online = false };
|
||||
return pc;
|
||||
}
|
||||
|
||||
/// Entry for an application processor once the architecture layer has set up its per-CPU
|
||||
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
|
||||
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
|
||||
/// interrupts enabled the timer preempts this idle context into whatever the global
|
||||
/// ready queue offers, so the core runs real work in parallel with the others. The
|
||||
/// `.c` calling convention lets the architecture trampoline path jump here. Never returns.
|
||||
pub fn secondaryMain() callconv(.c) noreturn {
|
||||
const flags = sync.enter();
|
||||
const pc = thisCpu();
|
||||
const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");
|
||||
t.* = .{ .id = next_id, .state = .running, .priority = 0 };
|
||||
next_id += 1;
|
||||
pc.current = t;
|
||||
pc.idle = t;
|
||||
pc.online = true;
|
||||
pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||
sync.leave(flags);
|
||||
|
||||
architecture.enableInterrupts(); // the timer now preempts this idle context into work
|
||||
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
|
||||
}
|
||||
|
||||
/// Number of cores that have finished bring-up (the BSP plus every online AP).
|
||||
pub fn onlineCount() usize {
|
||||
var n: usize = 0;
|
||||
for (&cpus) |*pc| {
|
||||
if (pc.online) n += 1;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue;
|
||||
/// everything else goes to the shared global queue.
|
||||
fn enqueue(t: *Task) void {
|
||||
if (t.affinity) |cpu| {
|
||||
const pc = &cpus[cpu];
|
||||
enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
|
||||
} else {
|
||||
enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t);
|
||||
}
|
||||
}
|
||||
|
||||
fn enqueueTo(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, t: *Task) void {
|
||||
t.next = null;
|
||||
const p: usize = t.priority;
|
||||
if (tail[p]) |tl| tl.next = t else head[p] = t;
|
||||
tail[p] = t;
|
||||
bitmap.* |= @as(u8, 1) << t.priority;
|
||||
}
|
||||
|
||||
/// The highest non-empty priority level in a bitmap, or -1 if empty.
|
||||
fn topLevel(bitmap: u8) i32 {
|
||||
if (bitmap == 0) return -1;
|
||||
return @as(i32, number_priorities - 1) - @as(i32, @clz(bitmap));
|
||||
}
|
||||
|
||||
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
|
||||
/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task
|
||||
/// wins an equal-priority tie, so it can't be starved by global work at its level.
|
||||
fn dequeueHighest(pc: *PerCpu) ?*Task {
|
||||
const g = topLevel(ready_bitmap);
|
||||
const p = topLevel(pc.pinned_bitmap);
|
||||
if (g < 0 and p < 0) return null;
|
||||
if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p));
|
||||
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
|
||||
}
|
||||
|
||||
fn dequeueFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
|
||||
const t = head[level].?;
|
||||
head[level] = t.next;
|
||||
if (head[level] == null) {
|
||||
tail[level] = null;
|
||||
bitmap.* &= ~(@as(u8, 1) << @intCast(level));
|
||||
}
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
/// Create a task that runs `entry` at `priority`, runnable on any core. It becomes
|
||||
/// ready immediately. Takes the kernel lock: it mutates the shared task table and
|
||||
/// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be
|
||||
/// serialised.
|
||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
||||
const flags = sync.enter();
|
||||
_ = create(entry, priority, null);
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there.
|
||||
/// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case
|
||||
/// the task is still created but left unpinned (so it runs *somewhere* rather than
|
||||
/// stranding in a queue no core services). Callers that require the pin (e.g. tests)
|
||||
/// should check the result.
|
||||
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const ok = cpu < maximum_cpus and cpus[cpu].online;
|
||||
_ = create(entry, priority, if (ok) cpu else null);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
||||
/// in user mode at `entry` on `user_sp`. It gets a fresh kernel stack for
|
||||
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
|
||||
/// Returns false (creating nothing) if the table is full or out of memory.
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
||||
/// across the whole spawn, so the address space and the task appear atomically).
|
||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority) bool {
|
||||
const t = freeSlot() orelse return false;
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
||||
t.* = .{
|
||||
.id = next_id,
|
||||
.state = .ready,
|
||||
.priority = priority,
|
||||
.stack = stack,
|
||||
.aspace = aspace,
|
||||
.user_ip = entry,
|
||||
.user_sp = user_sp,
|
||||
};
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.kstack_top = top;
|
||||
// First switch-in lands in startUserTask (no register smuggling — it reads
|
||||
// the user entry/stack from the Task itself).
|
||||
t.sp = architecture.initTaskStack(top, @intFromPtr(&startUserTask));
|
||||
enqueue(t);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It
|
||||
/// drops to ring 3 at the task's recorded entry/stack. Reading them from the
|
||||
/// Task avoids smuggling values through callee-saved registers across the
|
||||
/// context switch and lock release.
|
||||
fn startUserTask() void {
|
||||
const t = current();
|
||||
var buffer: [96]u8 = undefined;
|
||||
architecture.serialWrite(std.fmt.bufPrint(&buffer, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
|
||||
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
||||
}
|
||||
|
||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
|
||||
/// Returns the new task so a core can keep a handle to its idle task.
|
||||
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
||||
const t = freeSlot() orelse @panic("sched: task table full");
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.kstack_top = top;
|
||||
t.sp = architecture.initTaskStack(top, @intFromPtr(entry));
|
||||
enqueue(t);
|
||||
return t;
|
||||
}
|
||||
|
||||
fn freeSlot() ?*Task {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .free) return t;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Pick the highest-priority ready task and switch this core to it. The big kernel
|
||||
/// lock must be held by the caller (which also keeps local interrupts disabled);
|
||||
/// it serialises every core's scheduling, so no other core can touch the shared
|
||||
/// queues while we requeue `previous` and dequeue `next`. A dequeued task is `.ready`,
|
||||
/// never running elsewhere, so two cores never run the same task.
|
||||
fn schedule() void {
|
||||
const pc = thisCpu();
|
||||
const previous = pc.current;
|
||||
if (previous.state == .running) {
|
||||
previous.state = .ready;
|
||||
enqueue(previous); // back of its level's queue (round-robin)
|
||||
}
|
||||
const next = dequeueHighest(pc) orelse {
|
||||
previous.state = .running; // nothing else ready — keep running
|
||||
return;
|
||||
};
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
if (next != previous) switchTo(pc, &previous.sp, next);
|
||||
}
|
||||
|
||||
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
||||
/// user-mode interrupt lands on a good stack) and its address space (only when
|
||||
/// it differs from what's loaded — every page-table switch is a full TLB flush),
|
||||
/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used
|
||||
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
|
||||
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
|
||||
/// workload. The big kernel lock is held and interrupts are off throughout, so no
|
||||
/// interrupt can observe a half-updated (kernel stack, address space) pair.
|
||||
/// `save_sp` receives the outgoing task's stack pointer.
|
||||
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
||||
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
|
||||
const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable();
|
||||
if (want != pc.loaded_aspace) {
|
||||
architecture.loadPageTable(want);
|
||||
pc.loaded_aspace = want;
|
||||
}
|
||||
architecture.switchContext(save_sp, next.sp);
|
||||
}
|
||||
|
||||
/// Voluntarily give up the CPU to the next ready task.
|
||||
pub fn yield() void {
|
||||
const flags = sync.enter();
|
||||
schedule();
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// Block the current task for `ms` milliseconds, then let it become runnable
|
||||
/// again. The idle task (or other work) runs in the meantime.
|
||||
pub fn sleep(ms: u64) void {
|
||||
const flags = sync.enter();
|
||||
const t = current();
|
||||
t.wake_at = architecture.millis() + ms;
|
||||
t.state = .blocked;
|
||||
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
// --- event-based blocking -------------------------------------------------
|
||||
//
|
||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
||||
// message). Tasks link into it through the same `next` field the ready queues
|
||||
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
||||
// semaphores and IPC channels are built on.
|
||||
|
||||
pub const WaitQueue = struct {
|
||||
head: ?*Task = null,
|
||||
};
|
||||
|
||||
/// Block the current task on `wait_queue` and switch away. Precondition: the big kernel
|
||||
/// lock is held (so a condition can be checked and the block committed atomically;
|
||||
/// it also keeps local interrupts disabled). On return — when woken — the lock is
|
||||
/// still held.
|
||||
pub fn waitLocked(wait_queue: *WaitQueue) void {
|
||||
const t = current();
|
||||
t.state = .blocked;
|
||||
t.next = wait_queue.head;
|
||||
wait_queue.head = t;
|
||||
schedule();
|
||||
}
|
||||
|
||||
/// Move the highest-priority waiter on `wait_queue` (if any) to the ready queue.
|
||||
/// Precondition: the big kernel lock is held. Does not preempt — the caller decides.
|
||||
pub fn wakeLocked(wait_queue: *WaitQueue) void {
|
||||
// Find the highest-priority waiter (bounded scan) and unlink it.
|
||||
var best_previous: ?*Task = null;
|
||||
var best: ?*Task = null;
|
||||
var previous: ?*Task = null;
|
||||
var node = wait_queue.head;
|
||||
while (node) |t| : ({
|
||||
previous = t;
|
||||
node = t.next;
|
||||
}) {
|
||||
if (best == null or t.priority > best.?.priority) {
|
||||
best = t;
|
||||
best_previous = previous;
|
||||
}
|
||||
}
|
||||
const t = best orelse return;
|
||||
if (best_previous) |p| p.next = t.next else wait_queue.head = t.next;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Block the current task and switch away, without putting it on any wait queue —
|
||||
/// the caller has already linked it wherever it belongs (e.g. an endpoint's sender
|
||||
/// FIFO). Precondition: the big kernel lock is held; still held on return (when the
|
||||
/// task is made ready again). The IPC layer's counterpart to `waitLocked`.
|
||||
pub fn blockCurrentLocked() void {
|
||||
current().state = .blocked;
|
||||
schedule();
|
||||
}
|
||||
|
||||
/// Make a specific (currently blocked) task ready to run again. Precondition: the
|
||||
/// big kernel lock is held. Used by the IPC layer to wake a specific caller/server
|
||||
/// rather than "some waiter on a queue".
|
||||
pub fn readyLocked(t: *Task) void {
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Block on `wait_queue` (a self-contained critical section).
|
||||
pub fn wait(wait_queue: *WaitQueue) void {
|
||||
const flags = sync.enter();
|
||||
waitLocked(wait_queue);
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// Wake the highest-priority waiter on `wait_queue`, preempting if it outranks us.
|
||||
pub fn wake(wait_queue: *WaitQueue) void {
|
||||
const flags = sync.enter();
|
||||
const pc = thisCpu();
|
||||
wakeLocked(wait_queue);
|
||||
// If a task this core would now pick outranks the running one, run it at once.
|
||||
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
|
||||
// next tick; this core doesn't preempt for work it can't run.)
|
||||
if (highestReadyPriority(pc)) |p| {
|
||||
if (p > pc.current.priority) schedule();
|
||||
}
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// The highest-priority task core `pc` could run right now — the better of the global
|
||||
/// queue and this core's pinned queue — or null if it would fall back to idle.
|
||||
fn highestReadyPriority(pc: *PerCpu) ?Priority {
|
||||
const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap));
|
||||
if (top < 0) return null;
|
||||
return @intCast(top);
|
||||
}
|
||||
|
||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
||||
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
||||
fn wakeExpired() void {
|
||||
const now = architecture.millis();
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
||||
t.wake_at = 0;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Called from the timer interrupt (interrupts already disabled): wake due
|
||||
/// sleepers, then preempt. Takes the kernel lock like any other critical section,
|
||||
/// but releases it *without* touching the interrupt flag — the handler's `iretq`
|
||||
/// restores the interrupted context's flags, so re-enabling here would open a
|
||||
/// nested-interrupt window before the return.
|
||||
pub fn tick() void {
|
||||
_ = sync.enter();
|
||||
wakeExpired();
|
||||
if (preemption_enabled) schedule();
|
||||
sync.leaveIsr();
|
||||
}
|
||||
|
||||
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
||||
pub fn setPreemption(enabled: bool) void {
|
||||
preemption_enabled = enabled;
|
||||
}
|
||||
|
||||
/// End the current task and switch away for good; never returns. The task's stack
|
||||
/// is leaked for now (no reaper yet). Acquires the kernel lock and hands it off to
|
||||
/// the task we switch into (which releases it) — this frame never returns to leave.
|
||||
pub fn exit() noreturn {
|
||||
_ = sync.enter();
|
||||
const pc = thisCpu();
|
||||
pc.current.state = .free;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
var discard: usize = 0;
|
||||
switchTo(pc, &discard, next);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// End the current **user** task: free its address space, then exit. Runs on the
|
||||
/// dying task's kernel stack (in the shared kernel half, so it survives the CR3
|
||||
/// switch to the kernel tables that must happen before we free the process's own
|
||||
/// tables — we can't free the page tables we're standing on). The kernel stack
|
||||
/// itself is leaked, as in `exit` (no reaper yet). Never returns.
|
||||
pub fn exitUser() noreturn {
|
||||
_ = sync.enter();
|
||||
const pc = thisCpu();
|
||||
const dying = pc.current;
|
||||
const as = dying.aspace;
|
||||
if (as != 0) {
|
||||
const kroot = architecture.kernelPageTable();
|
||||
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
||||
pc.loaded_aspace = kroot;
|
||||
architecture.destroyAddressSpace(as);
|
||||
}
|
||||
dying.state = .free;
|
||||
dying.aspace = 0;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
var discard: usize = 0;
|
||||
switchTo(pc, &discard, next);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Whether the running task is a user process (has its own address space).
|
||||
pub fn currentIsUserProcess() bool {
|
||||
return current().aspace != 0;
|
||||
}
|
||||
|
||||
pub fn currentId() u32 {
|
||||
return current().id;
|
||||
}
|
||||
|
||||
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
||||
/// per-CPU state, so a task calling it on different cores sees different values —
|
||||
/// which is how a test can prove work is running in parallel. Returns 0 if the GS
|
||||
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
|
||||
/// reporter can call it unconditionally without a second fault.
|
||||
pub fn currentCpuIndex() u32 {
|
||||
if (architecture.cpuLocal() == 0) return 0;
|
||||
return thisCpu().index;
|
||||
}
|
||||
|
||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||
pub fn setPriority(p: Priority) void {
|
||||
current().priority = p;
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
//! The big kernel lock (BKL) — the coarse mutual exclusion that lets more than one
|
||||
//! CPU run kernel code safely.
|
||||
//!
|
||||
//! Until SMP, the kernel's mutual exclusion *was* the interrupt flag: a critical
|
||||
//! section did `cli`, and since only one core existed, nothing else could touch
|
||||
//! kernel state (the discipline in docs/scheduling.md). That invariant dies the
|
||||
//! instant a second core runs kernel code — `cli` on one core does nothing to
|
||||
//! another. So the kernel's shared state (the scheduler queues, IPC channels) is
|
||||
//! guarded by a spinlock, and the lock is **always held with local interrupts
|
||||
//! disabled**, so a core's own timer interrupt can't re-enter the kernel and
|
||||
//! deadlock against the lock it already holds.
|
||||
//!
|
||||
//! This is deliberately *one coarse lock*, not many fine ones: it's philosophically
|
||||
//! aligned with a tiny kernel and it keeps the single-core correctness model
|
||||
//! (docs/scheduling.md) largely intact — one lock around kernel entry instead of
|
||||
//! rethinking every critical section. It's the first-design choice seL4 makes and
|
||||
//! docs/smp.md endorses; per-core run queues + fine-grained locking come later, if
|
||||
//! contention ever bites. Because the kernel does little, the lock is held briefly.
|
||||
//!
|
||||
//! **The hand-off rule.** The lock is held *across* a context switch and released
|
||||
//! by whichever task resumes, not by the one that switched away. A task that blocks
|
||||
//! or yields calls `enter`, mutates the queues, `schedule()`s — switching to another
|
||||
//! task *with the lock still held* — and only calls `leave` once it is eventually
|
||||
//! resumed and its critical section runs to the end. So every call into `schedule()`
|
||||
//! (and thus `switch_context`) happens with the lock held, and every task resumes
|
||||
//! from a switch holding it. A freshly-spawned task has no `enter`/`leave` frame to
|
||||
//! resume into, so `task_trampoline` releases the lock explicitly on its behalf via
|
||||
//! `releaseForFreshTask` before running the task body.
|
||||
|
||||
const std = @import("std");
|
||||
const architecture = @import("architecture");
|
||||
|
||||
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
||||
var held = std.atomic.Value(u32).init(0);
|
||||
|
||||
/// Enter the kernel: disable interrupts on this core, then spin until we own the
|
||||
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
|
||||
/// Interrupts stay off for the whole critical section so this core's timer tick
|
||||
/// can't try to re-acquire the lock we're holding.
|
||||
pub fn enter() u64 {
|
||||
const flags = architecture.saveInterrupts();
|
||||
acquire();
|
||||
return flags;
|
||||
}
|
||||
|
||||
/// Release the lock and restore the interrupt flags `enter` returned (re-enabling
|
||||
/// interrupts only if they were on beforehand). The normal exit for a critical
|
||||
/// section reached from task context (`yield`, `sleep`, `wait`, `wake`, IPC).
|
||||
pub fn leave(flags: u64) void {
|
||||
release();
|
||||
architecture.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// Release the lock but leave interrupts as they are. The exit for a critical
|
||||
/// section running inside an interrupt handler (the timer `tick`): the handler's
|
||||
/// `iretq` is what restores the interrupted context's flags, so restoring them
|
||||
/// here too would open a nested-interrupt window before the return. Release only.
|
||||
pub fn leaveIsr() void {
|
||||
release();
|
||||
}
|
||||
|
||||
/// Release the lock on behalf of a freshly-spawned task. Such a task is switched to
|
||||
/// (with the lock held) but has no `enter`/`leave` frame of its own to release
|
||||
/// through — `task_trampoline` calls this before running the task body. Interrupts
|
||||
/// are enabled separately by the trampoline. Exported for the assembly trampoline.
|
||||
export fn releaseForFreshTask() callconv(.c) void {
|
||||
release();
|
||||
}
|
||||
|
||||
fn acquire() void {
|
||||
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
||||
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
||||
while (held.swap(1, .acquire) != 0) {
|
||||
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
||||
}
|
||||
}
|
||||
|
||||
fn release() void {
|
||||
held.store(0, .release);
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user