From cb2f49cb48ffeec32ab54c80ee7dbe662d03495e Mon Sep 17 00:00:00 2001 From: Daniel Samson Date: Fri, 3 Jul 2026 11:24:26 +0100 Subject: [PATCH] added physical memory manager --- docs/README.md | 11 ++- docs/frame-allocator.md | 111 +++++++++++++++++++++++++++++ docs/memory-map.md | 8 +-- src/main.zig | 16 +++++ src/pmm.zig | 151 ++++++++++++++++++++++++++++++++++++++++ 5 files changed, 290 insertions(+), 7 deletions(-) create mode 100644 docs/frame-allocator.md create mode 100644 src/pmm.zig diff --git a/docs/README.md b/docs/README.md index 16ef352..100ae4b 100644 --- a/docs/README.md +++ b/docs/README.md @@ -16,7 +16,10 @@ rather than restate it. Roughly in the order things happen at runtime: 4. **[memory-map.md](memory-map.md) — the memory map.** How the loader learns what physical RAM exists and hands it to the kernel in danos's own neutral format, rather than leaking UEFI's memory descriptors across the boundary. -5. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and +5. **[frame-allocator.md](frame-allocator.md) — the physical frame allocator.** The + bitmap allocator that hands out and reclaims 4 KiB physical frames from that + map — the primitive page tables and the heap will be built on. +6. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and how `while (true) hlt` parks the CPU safely once there's nothing left to do. Cutting across all of these: @@ -30,7 +33,8 @@ Cutting across all of these: The boot flow ties them together: UEFI runs the loader ([efi.md](efi.md)), which queries the **GOP** to pick a graphics mode ([gop.md](gop.md)), hands the kernel a **framebuffer** to draw into ([framebuffer.md](framebuffer.md)) and a **memory -map** of physical RAM ([memory-map.md](memory-map.md)); the kernel runs — its +map** of physical RAM ([memory-map.md](memory-map.md)); the kernel turns that map +into a **frame allocator** ([frame-allocator.md](frame-allocator.md)), runs — its CPU-specific bits behind the [arch](arch.md) boundary — and when it has finished, or panics, it **halts** ([halting.md](halting.md)). @@ -39,8 +43,9 @@ or panics, it **halts** ([halting.md](halting.md)). | Area | Code | |------|------| | Bootloader (UEFI app) | `src/efi.zig` → `BOOTX64.efi` | -| Kernel entry, panic, memory-map read-out | `src/main.zig` | +| Kernel entry, panic, bring-up | `src/main.zig` | | Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` | +| Physical frame allocator | `src/pmm.zig` | | Framebuffer text console | `src/console.zig` | | Arch-specific kernel code (`halt`, linker script) | `src/arch/x86_64/` | | Build + `run-efi` (QEMU/OVMF) | `build.zig` | diff --git a/docs/frame-allocator.md b/docs/frame-allocator.md new file mode 100644 index 0000000..8f6e2f9 --- /dev/null +++ b/docs/frame-allocator.md @@ -0,0 +1,111 @@ +# The physical frame allocator + +Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a +way to *hand out* that RAM: give me a free page of physical memory, and later, +here's one back. That's the **physical frame allocator** (a "physical memory +manager", hence `src/pmm.zig`). It deals only in fixed 4 KiB **frames** — the +natural unit because that's the granularity the CPU's paging hardware maps — and +it is the primitive everything above it stands on: page tables, the kernel heap, +per-process memory all ultimately ask the frame allocator for pages. + +It's **generic kernel code**: it operates on the neutral `danos.MemoryRegion` +array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB +page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.) + +## Why a bitmap + +There are a few classic designs; danos starts with the simplest that still +supports freeing: + +- **Bitmap** (chosen): one bit per frame, `1 = used`, `0 = free`. Freeing is + trivial (clear a bit), it's very compact, and you can later extend it to + allocate *contiguous* runs by scanning for consecutive zero bits. Allocation is + a linear scan, but that's cheap and easy to reason about. +- **Intrusive free-list / stack**: store the "next free frame" pointer inside each + free frame; O(1) alloc and free. Elegant, but it can't satisfy contiguous + multi-frame requests and can't answer "is *this* frame free?". +- **Buddy allocator**: great for contiguous power-of-two blocks, but more + machinery than a first allocator needs. + +Compactness matters less than clarity here, but it's a nice property: 128 MiB of +RAM is 32768 frames — a **4 KiB bitmap, a single frame**. Even 64 GiB needs only +2 MiB of bitmap. + +## How it works + +State lives in `src/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`, +and a `next_hint` marking where the next allocation scan should start. + +### init(map) — building it from the memory map + +1. **Size it.** Find the highest address across all `usable` regions; + `total_frames = highest / page_size`. Reserved and MMIO spans above that + (remember the ~12 GiB of MMIO from [memory-map.md](memory-map.md)) sit *outside* + the bitmap and are simply never allocatable. +2. **Place it (the bootstrap).** The bitmap needs storage before an allocator + exists — a chicken-and-egg. Solution: pick the first `usable` region big enough + to hold the bitmap and put it there, addressing it directly as a pointer. That + last part relies on the firmware's **identity mapping** still being in effect + (physical address == virtual address), which holds until the kernel installs + its own page tables. +3. **Mark, then free.** Set the whole bitmap to `used` (`0xff`), then walk the + `usable` regions clearing their bits. Doing it in that direction means every + gap, reserved span, and hole is unallocatable *by default* — we only ever hand + back memory the firmware explicitly called usable. +4. **Take back the essentials.** Re-reserve the frames the bitmap itself occupies + (they're inside a usable region we just freed), plus **frame 0**, so an address + of `0` can keep meaning "no frame". + +### alloc() → ?u64 + +Scan the bitmap from `next_hint` (wrapping once) for the first free bit, mark it +used, advance the hint, and return `frame * page_size`. Returns `null` when no +frame is free — genuine out-of-memory. The hint avoids rescanning the low, +long-since-allocated frames on every call. + +### free(addr) + +Clear the frame's bit and, if it's below `next_hint`, pull the hint back so the +reclaimed frame gets reused soon. Bogus or double frees (a frame already marked +free, or one out of range) are ignored rather than corrupting the used count. + +## Correctness points worth remembering + +- **Generic walk.** Because `MemoryRegion` is danos's own type, the map is a plain + slice — none of the variable descriptor-stride from the raw UEFI map. +- **Identity mapping assumption.** Placing the bitmap by physical address only + works while the firmware's identity map is live. When danos sets up its own + paging, the bitmap (and any other physical pointer) will need an explicit + mapping. This is a deliberate, documented dependency of this stage. +- **Frame 0 is reserved** so `0` stays a safe "none" sentinel — and the bitmap is + never placed there. (An early bug did exactly that: a `usable` region at physical + address 0 collided with a `0`-means-not-found sentinel and tripped a panic. The + fix was an optional plus starting the bitmap at least one page in.) +- **Everything non-usable is unallocatable by construction** — the "mark all used, + then free usable" order gives that for free, so the kernel image, the loader's + buffers, MMIO and firmware memory can never be handed out. + +## Verifying it + +`kmain` brings the allocator up and self-tests it. Booted in QEMU with 128 MiB: + +``` +danos: frame allocator online + free frames: 19751 (77 MiB) <- matches the map's 77 MiB usable + alloc x3 : 0x2000 0x3000 0x4000 <- frame 0 reserved, bitmap at 0x1000, so allocs start at 0x2000 + after free : 19751 frames free <- three freed, count restored +``` + +The `free frames` MiB agreeing with the memory map's `usable RAM`, the three +distinct consecutive addresses, and the count returning to its start after freeing +are the three signals that init, alloc and free are all correct. + +## What's next (not done here) + +- **Contiguous allocation** — scan for N consecutive free bits — for callers that + need physically adjacent frames. +- **Consumers**: the virtual memory manager / page tables and then the kernel heap + will be the first real users, each asking `alloc()` for frames. +- **Reclaiming `reclaimable`** (UEFI boot-services) memory, and eventually the + `reserved` `loader_data` (kernel image, boot buffers) once nothing needs it — + see the deferred list in [memory-map.md](memory-map.md). diff --git a/docs/memory-map.md b/docs/memory-map.md index 8036d81..e473899 100644 --- a/docs/memory-map.md +++ b/docs/memory-map.md @@ -131,12 +131,12 @@ kernel with a **device-tree blob**; the AArch64 entry code will parse its array. The kernel's memory code — the frame allocator and everything above it — never knows the difference. -## What's next (not done here) +## What's next -This is plumbing plus classification only. Still to come: +This page is plumbing plus classification only. The map's first consumer, the +**physical frame allocator**, is built directly on the `usable` regions here — +see [frame-allocator.md](frame-allocator.md). Still to come after that: -- A **physical frame allocator** that consumes `usable` regions and hands out - 4 KiB frames — the foundation everything else stands on. - Reclaiming `reclaimable` regions, and carefully freeing `reserved` `loader_data` (kernel image, these buffers) once the kernel is done reading them. - Paging / the kernel's own page tables, then a heap. diff --git a/src/main.zig b/src/main.zig index c906a59..0640639 100644 --- a/src/main.zig +++ b/src/main.zig @@ -2,6 +2,7 @@ const std = @import("std"); const danos = @import("danos"); const arch = @import("arch"); const console = @import("console.zig"); +const pmm = @import("pmm.zig"); const BootInfo = danos.BootInfo; /// The calling convention used to enter the kernel. Pinned to SysV explicitly: @@ -45,6 +46,21 @@ fn kmain(boot_info: *const BootInfo) noreturn { con.print(" mem regions: {d}\n", .{regions.len}); con.print(" usable RAM : {d} MiB\n", .{usable_pages * danos.page_size / (1024 * 1024)}); + // Bring up the physical frame allocator over that map, and prove it works: + // allocate three frames, then hand them back. + pmm.init(boot_info.memory_map); + const s = pmm.stats(); + con.print("\ndanos: frame allocator online\n", .{}); + con.print(" free frames: {d} ({d} MiB)\n", .{ s.free_frames, s.free_frames * danos.page_size / (1024 * 1024) }); + const f0 = pmm.alloc(); + const f1 = pmm.alloc(); + const f2 = pmm.alloc(); + con.print(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 }); + if (f0) |p| pmm.free(p); + if (f1) |p| pmm.free(p); + if (f2) |p| pmm.free(p); + con.print(" after free : {d} frames free\n", .{pmm.stats().free_frames}); + con.write("\nkernel initialised; nothing left to do, halting.\n"); arch.halt(); diff --git a/src/pmm.zig b/src/pmm.zig new file mode 100644 index 0000000..01635fb --- /dev/null +++ b/src/pmm.zig @@ -0,0 +1,151 @@ +//! Physical memory manager: a bitmap frame allocator. It hands out and reclaims +//! 4 KiB physical frames — the primitive every later memory feature (page +//! tables, the heap) is built on top of. +//! +//! This is generic kernel code: it works on the neutral `danos.MemoryRegion` +//! array the loader hands over (see docs/memory-map.md), so it carries no UEFI +//! and nothing architecture-specific beyond the 4 KiB page. + +const std = @import("std"); +const danos = @import("danos"); + +const page_size = danos.page_size; + +/// One bit per frame, covering physical RAM from 0 up to the highest usable +/// address: 1 = used/unavailable, 0 = free. The bitmap itself lives in a frame +/// we carve out of usable memory during init. +var bitmap: []u8 = &.{}; +var total_frames: usize = 0; +var used_frames: usize = 0; +/// Where the next allocation scan begins, so we don't rescan from frame 0 every +/// time. Pulled back on free() so reclaimed low frames get reused. +var next_hint: usize = 0; + +pub const Stats = struct { + total_frames: usize, + used_frames: usize, + free_frames: usize, +}; + +pub fn stats() Stats { + return .{ + .total_frames = total_frames, + .used_frames = used_frames, + .free_frames = total_frames - used_frames, + }; +} + +inline fn bit(frame: usize) u3 { + return @intCast(frame & 7); +} +inline fn isUsed(frame: usize) bool { + return (bitmap[frame >> 3] >> bit(frame)) & 1 != 0; +} +inline fn setUsed(frame: usize) void { + bitmap[frame >> 3] |= @as(u8, 1) << bit(frame); +} +inline fn setFree(frame: usize) void { + bitmap[frame >> 3] &= ~(@as(u8, 1) << bit(frame)); +} + +fn regions(map: danos.MemoryMap) []const danos.MemoryRegion { + return @as([*]const danos.MemoryRegion, @ptrFromInt(map.regions))[0..map.len]; +} + +/// Build the allocator from the loader's memory map. Relies on the firmware's +/// identity mapping still being in effect (a physical address is usable directly +/// as a pointer) — true until the kernel installs its own page tables. +pub fn init(map: danos.MemoryMap) void { + const regs = regions(map); + + // 1. Size the bitmap to cover every frame up to the highest usable address. + // Reserved/MMIO spans above that are simply outside the map and never + // allocatable. + var highest: u64 = 0; + for (regs) |r| { + if (r.kind != .usable) continue; + const end = r.base + r.pages * page_size; + if (end > highest) highest = end; + } + total_frames = @intCast(highest / page_size); + if (total_frames == 0) @panic("pmm: no usable memory"); + const bitmap_bytes = (total_frames + 7) / 8; + const bitmap_pages = (bitmap_bytes + page_size - 1) / page_size; + + // 2. Park the bitmap in the first usable region large enough to hold it. + // Start at least one page in, so we never place it on frame 0 (which is + // kept reserved as the "none" address, and is an awkward pointer besides). + var storage: ?u64 = null; + for (regs) |r| { + if (r.kind != .usable) continue; + const base = if (r.base == 0) page_size else r.base; + const skipped = (base - r.base) / page_size; + if (r.pages - skipped >= bitmap_pages) { + storage = base; + break; + } + } + const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap"); + bitmap = @as([*]u8, @ptrFromInt(bitmap_base))[0..bitmap_bytes]; + + // 3. Start with everything marked used, then free the usable regions. Doing + // it this way means every gap, reserved span and MMIO hole is unallocatable + // by default — we only ever hand back memory the firmware called usable. + @memset(bitmap, 0xff); + used_frames = total_frames; + for (regs) |r| { + if (r.kind != .usable) continue; + var f: usize = @intCast(r.base / page_size); + const end = f + @as(usize, @intCast(r.pages)); + while (f < end and f < total_frames) : (f += 1) { + setFree(f); + used_frames -= 1; + } + } + + // 4. Take back the frames the bitmap occupies, and frame 0 — so a 0 result + // stays reserved to mean "no frame". + reserve(bitmap_base, bitmap_pages); + reserve(0, 1); +} + +/// Mark `count` frames from physical `base` as used, counting only those that +/// were actually free. +fn reserve(base: u64, count: usize) void { + var f: usize = @intCast(base / page_size); + const end = f + count; + while (f < end and f < total_frames) : (f += 1) { + if (!isUsed(f)) { + setUsed(f); + used_frames += 1; + } + } +} + +/// Allocate one physical frame, or null if none are free. The address is +/// page-aligned; the frame's contents are undefined. +pub fn alloc() ?u64 { + var scanned: usize = 0; + var f = next_hint; + while (scanned < total_frames) : (scanned += 1) { + if (f >= total_frames) f = 0; + if (!isUsed(f)) { + setUsed(f); + used_frames += 1; + next_hint = f + 1; + return @as(u64, f) * page_size; + } + f += 1; + } + return null; // out of physical memory +} + +/// Return a frame obtained from alloc() to the pool. Bogus or double frees are +/// ignored rather than corrupting the count. +pub fn free(addr: u64) void { + const f: usize = @intCast(addr / page_size); + if (f >= total_frames or !isUsed(f)) return; + setFree(f); + used_frames -= 1; + if (f < next_hint) next_hint = f; +}