From e80043b611a63ec341daee4ad80896c3b0945696 Mon Sep 17 00:00:00 2001 From: Daniel Samson Date: Fri, 3 Jul 2026 14:17:00 +0100 Subject: [PATCH] Calibrated timer / clock --- docs/README.md | 6 +++ docs/device-interrupts.md | 16 +++++--- docs/testing.md | 1 + docs/vision.md | 84 +++++++++++++++++++++++++++++++++++++++ src/arch/x86_64/apic.zig | 70 +++++++++++++++++++++++++++++--- src/arch/x86_64/cpu.zig | 21 ++++++++-- src/main.zig | 2 +- src/tests.zig | 23 +++++++++++ test/qemu_test.py | 3 ++ 9 files changed, 212 insertions(+), 14 deletions(-) create mode 100644 docs/vision.md diff --git a/docs/README.md b/docs/README.md index 2b3861c..8ecc55a 100644 --- a/docs/README.md +++ b/docs/README.md @@ -34,6 +34,12 @@ rather than restate it. Roughly in the order things happen at runtime: 10. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and how `while (true) hlt` parks the CPU safely once there's nothing left to do. +Start with the north star: + +- **[vision.md](vision.md) — the vision.** danos is aiming to be a real-time + microkernel: minimal kernel, drivers/services isolated in user space, preemptive + scheduling with timing guarantees. The *why* that shapes everything below. + Cutting across all of these: - **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept diff --git a/docs/device-interrupts.md b/docs/device-interrupts.md index 4fb4f08..59ac8d3 100644 --- a/docs/device-interrupts.md +++ b/docs/device-interrupts.md @@ -39,10 +39,15 @@ LVT-timer entry giving it a **vector** (32) and **periodic** mode, then an initi count that becomes the reload value. From then on it fires vector 32 repeatedly, on its own, forever. -> The count isn't calibrated to real time yet — the tick *rate* is arbitrary -> (bus-clock dependent). Turning it into a known frequency (say 100 Hz) needs a -> reference clock to measure against (the PIT, HPET, or the TSC). That's a later -> step; for now it just needs to tick. +The reload count isn't picked arbitrarily — it's **calibrated to real time**, +which the [real-time](vision.md) scheduling guarantees depend on. Since the LAPIC +timer's raw rate is bus-clock dependent and unknown up front, `calibrate` measures +it against the **PIT** (the legacy 8254, whose 1.193182 MHz is fixed): run the +LAPIC timer one-shot from its maximum count while the PIT counts out a known 10 ms +(polling channel 2, no interrupt needed), then see how far the LAPIC got. That +yields its counts-per-millisecond, from which `initTimer(hz)` computes the reload +count for any target frequency. danos runs it at **1000 Hz** (a 1 ms tick), and the +tick count times the known period gives a monotonic `uptimeMs()`. ## Two kinds of vector, one dispatch @@ -101,7 +106,8 @@ spinning in unrelated code — is the whole mechanism working end to end. - **The keyboard**: bring up the IO-APIC, route its IRQ to a vector, and read scancodes from the PS/2 controller — the first *input* device. -- **A calibrated timer** at a known frequency, and a monotonic clock. +- **`sleep()` / timeouts** built on the calibrated clock (the monotonic + `uptimeMs()` is in place). - **Uncacheable MMIO**: the LAPIC page is currently mapped writeback-cacheable like the rest of the identity map. QEMU tolerates it, but real hardware wants MMIO marked uncacheable (via the page's cache bits or an MTRR). diff --git a/docs/testing.md b/docs/testing.md index d957df0..a936ff8 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -47,6 +47,7 @@ Current cases: |------|----------------|-----------------------------| | `smoke` | memory map has usable RAM; frame alloc/free; paging active | `DANOS-TEST-RESULT: PASS` | | `timer` | device interrupts fire and return (tick count advances) | `DANOS-TEST-RESULT: PASS` | +| `clock` | calibrated LAPIC frequency is sane; monotonic uptime advances | `DANOS-TEST-RESULT: PASS` | | `vmm` | on-demand `map` works: a mapped page is writable and reads back | `DANOS-TEST-RESULT: PASS` | | `heap` | kernel heap: alloc/free, block reuse, growth, and a std container on it | `DANOS-TEST-RESULT: PASS` | | `fault-ud` | invalid-opcode exception is caught | serial shows `invalid opcode (vector 6)` | diff --git a/docs/vision.md b/docs/vision.md new file mode 100644 index 0000000..05b9798 --- /dev/null +++ b/docs/vision.md @@ -0,0 +1,84 @@ +# Vision: a real-time microkernel + +danos is aiming to be a **real-time operating system built on a microkernel** — +where drivers and services run isolated in user space for maximum stability, and +scheduling gives real guarantees about timing. This page is the north star: the +*why* that shapes every design decision below it. Read it before adding anything +structural. + +## Microkernel + +The kernel stays **minimal** — only what genuinely must run in privileged mode: + +- scheduling, +- inter-process communication (IPC), +- memory management (address spaces, page tables), +- low-level interrupt dispatch. + +Everything else — device drivers, filesystems, the network stack — runs as an +**isolated user-space server**, each in its own address space with only the +privileges it needs. + +The payoff is **stability through isolation**. A driver bug can't corrupt the +kernel or another driver; a crashing service is contained and can be restarted, +while the rest of the system keeps running. That's the opposite of a monolithic +kernel, where a single driver fault can take everything down. + +The cost is that **IPC becomes the backbone**: whatever used to be a function call +across a monolithic kernel is now a message between address spaces. In a +microkernel, IPC performance essentially *is* system performance (the lesson of +L4). So IPC must be fast, and it's a first-class concern, not an afterthought. +Hardware interrupts, too, become IPC: the kernel turns an IRQ into a message to the +driver task that owns that device. + +## Real-time + +danos schedules **preemptively, with guarantees about quanta** — the system must +be able to promise that a task runs when it's supposed to, within bounded time. +That imposes concrete requirements: + +- **Fixed-priority preemptive scheduling.** The highest-priority ready task always + runs; a higher-priority task that becomes ready preempts a lower one immediately. + Not round-robin (which is fair but not predictable). +- **A calibrated, deterministic clock.** Guarantees measured in "quanta" are + meaningless on an arbitrary tick rate — real time requires a timer calibrated to + a known frequency. +- **Bounded interrupt latency.** Interrupt-disabled sections must be short and + bounded, so a ready high-priority task is never delayed by an unbounded kernel + operation. +- **Deterministic kernel operations.** Scheduling decisions should be O(1) (e.g. a + priority bitmap), not "walk a list of unknown length." +- **Priority inheritance** (once there are locks/IPC), so a high-priority task + blocked on a resource held by a low-priority one can't be delayed indefinitely by + a middle-priority task — bounding priority inversion. + +A consequence worth stating early: the current [kernel heap](heap.md) is a +first-fit free list, which has **unbounded allocation time** and can fragment — it +is *not* real-time safe. It's fine for one-time kernel setup, but real-time paths +must pre-allocate or use a bounded (fixed-size pool) allocator. Don't allocate on a +hot real-time path. + +## What this means for the roadmap + +The vision reorders the obvious hobby-kernel path. Notably, **drivers are not +built into the kernel** — so an in-kernel keyboard driver would be throwaway work. +Input devices arrive later, as the *first user-space drivers*, once the machinery +to isolate them exists. The trajectory: + +1. **Calibrated timer / clock** — a known-frequency, deterministic tick. The + foundation real-time quanta rest on. *(next)* +2. **Real-time scheduler** — fixed-priority preemptive, kernel threads first: + context switch, task struct, priority run-queue, timer-driven preemption. +3. **User mode + address-space isolation** — higher-half kernel, ring 3, per-process + page tables. The substrate for isolated servers. +4. **IPC** — fast message passing between address spaces. The microkernel's heart. +5. **User-space drivers** — interrupts delivered as IPC, plus MMIO/port-access + grants. The keyboard becomes the first one, validating the whole model. + +## Where we are + +The foundation is in place: UEFI boot, framebuffer + [serial](testing.md), +[physical frames](frame-allocator.md), [paging](paging.md) with W^X, [exceptions +and interrupts](interrupts.md), a [timer](device-interrupts.md), and a +[heap](heap.md) — plus a [test harness](testing.md). The kernel boots and has its +core services; the next milestones make it *schedule*, then *isolate*. diff --git a/src/arch/x86_64/apic.zig b/src/arch/x86_64/apic.zig index 30701cb..027d721 100644 --- a/src/arch/x86_64/apic.zig +++ b/src/arch/x86_64/apic.zig @@ -22,8 +22,13 @@ const reg_spurious = 0x0F0; const reg_eoi = 0x0B0; const reg_lvt_timer = 0x320; const reg_timer_initial = 0x380; +const reg_timer_current = 0x390; const reg_timer_divide = 0x3E0; +const lvt_masked = 1 << 16; +const lvt_periodic = 1 << 17; +const timer_divide_16 = 0x3; + const ia32_apic_base_msr = 0x1B; /// LAPIC MMIO base. A runtime var (not a constant) both because we read it from @@ -33,6 +38,12 @@ var base: usize = 0xFEE00000; var tick_count: u64 = 0; +/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate). +/// At divide-by-16, this is the effective counting rate. +var ticks_per_ms: u32 = 0; +/// The periodic-interrupt frequency the timer is armed at, once initTimer runs. +var timer_hz: u32 = 0; + fn read(reg: u32) u32 { return @as(*volatile u32, @ptrFromInt(base + reg)).*; } @@ -67,11 +78,60 @@ pub fn init() void { write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable } -/// Arm the LAPIC timer in periodic mode on `timer_vector`. -pub fn initTimer() void { - write(reg_timer_divide, 0x3); // divide bus clock by 16 - write(reg_lvt_timer, timer_vector | (1 << 17)); // periodic mode - write(reg_timer_initial, 1_000_000); // reload count -> periodic ticks +/// Measure the LAPIC timer's counting rate against the PIT (channel 2, which can +/// be polled without interrupts). We run the LAPIC timer one-shot from its max +/// count while the PIT counts out a known 10 ms, then see how far the LAPIC got. +/// This gives real time, which the RTOS quanta guarantees depend on. +pub fn calibrate() void { + const pit_hz = 1_193_182; + const calib_ms = 10; + const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms); + + // LAPIC timer: divide 16, masked (no interrupt — we just want the count), + // counting down from the maximum. + write(reg_timer_divide, timer_divide_16); + write(reg_lvt_timer, lvt_masked); + write(reg_timer_initial, 0xFFFFFFFF); + + // PIT channel 2, mode 0 (interrupt on terminal count): load the count with the + // gate low, then raise the gate to start it counting. + io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low + io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0 + io.outb(0x42, @truncate(pit_count)); + io.outb(0x42, @truncate(pit_count >> 8)); + io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start + + while (io.inb(0x61) & 0x20 == 0) {} // poll channel-2 output until terminal count + + const elapsed = 0xFFFFFFFF - read(reg_timer_current); + write(reg_timer_initial, 0); // stop the timer + ticks_per_ms = elapsed / calib_ms; +} + +/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires +/// calibrate() to have run. +pub fn initTimer(hz: u32) void { + timer_hz = hz; + const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second + write(reg_timer_divide, timer_divide_16); + write(reg_lvt_timer, timer_vector | lvt_periodic); + write(reg_timer_initial, @intCast(count)); +} + +/// Configured periodic-interrupt frequency (Hz). +pub fn frequencyHz() u32 { + return timer_hz; +} + +/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks. +pub fn lapicHz() u64 { + return @as(u64, ticks_per_ms) * 1000; +} + +/// Milliseconds since the timer started (monotonic). Ticks accrue at timer_hz. +pub fn uptimeMs() u64 { + if (timer_hz == 0) return 0; + return ticks() * 1000 / timer_hz; } /// Acknowledge the current interrupt so the LAPIC will deliver the next one. diff --git a/src/arch/x86_64/cpu.zig b/src/arch/x86_64/cpu.zig index 9a4b7df..a527c66 100644 --- a/src/arch/x86_64/cpu.zig +++ b/src/arch/x86_64/cpu.zig @@ -60,12 +60,17 @@ pub fn readCr3() u64 { ); } -/// Enable the Local APIC and start its periodic timer, the kernel's heartbeat. -/// Interrupts still have to be unmasked with enableInterrupts() to be delivered. +/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum. +pub const timer_hz = 1000; + +/// Enable the Local APIC, calibrate its timer against the PIT, and start it firing +/// at `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be +/// unmasked with enableInterrupts() to be delivered. pub fn startTimer() void { apic.init(); + apic.calibrate(); idt.setHandler(apic.timer_vector, apic.timerTick); - apic.initTimer(); + apic.initTimer(timer_hz); } /// Number of timer ticks since startTimer(). @@ -73,6 +78,16 @@ pub fn ticks() u64 { return apic.ticks(); } +/// Milliseconds since the timer started (monotonic). +pub fn uptimeMs() u64 { + return apic.uptimeMs(); +} + +/// Measured LAPIC timer frequency in Hz (from calibration). +pub fn lapicHz() u64 { + return apic.lapicHz(); +} + /// Unmask maskable interrupts (`sti`) so device interrupts get delivered. pub fn enableInterrupts() void { asm volatile ("sti"); diff --git a/src/main.zig b/src/main.zig index de5eb72..7b53461 100644 --- a/src/main.zig +++ b/src/main.zig @@ -100,7 +100,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Start the timer and unmask interrupts — the kernel now has a heartbeat. arch.startTimer(); arch.enableInterrupts(); - con.write("danos: timer interrupts enabled\n"); + con.print("danos: timer online ({d} Hz tick, LAPIC {d} MHz measured)\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000 }); // In a test build (`zig build -Dtest-case=`), run that case and stop. // Normal builds fall through to the idle halt. diff --git a/src/tests.zig b/src/tests.zig index f2b8bb1..4139d2c 100644 --- a/src/tests.zig +++ b/src/tests.zig @@ -49,6 +49,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { smoke(boot_info); } else if (eql(case, "timer")) { timer(); + } else if (eql(case, "clock")) { + clock(); } else if (eql(case, "vmm")) { vmm(); } else if (eql(case, "heap")) { @@ -204,6 +206,27 @@ fn heapTest() void { result(); } +/// Verify the calibrated clock: a plausible measured LAPIC frequency, the +/// configured tick rate, and monotonic uptime that advances with real ticks. +fn clock() void { + log("DANOS-TEST-BEGIN: clock\n", .{}); + + // Calibration produced a sane LAPIC frequency (roughly 1 MHz .. 100 GHz). + const lapic = arch.lapicHz(); + check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000); + + // Wait for ~5 real ticks and confirm uptime advanced by about that many ms + // (tick rate is 1000 Hz, so 1 tick == 1 ms). + const start_ticks = arch.ticks(); + const start_ms = arch.uptimeMs(); + var spins: u64 = 0; + while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1; + const elapsed_ms = arch.uptimeMs() - start_ms; + check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100); + + result(); +} + fn faultInvalidOpcode() void { log("DANOS-TEST-BEGIN: fault-ud\n", .{}); asm volatile ("ud2"); diff --git a/test/qemu_test.py b/test/qemu_test.py index 79a4ea2..f504212 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -66,6 +66,9 @@ CASES = [ {"name": "timer", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + {"name": "clock", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, {"name": "vmm", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"},