From 724de7bbd05cd8d41bba838695cc2f13d7377a28 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 8 Jul 2026 22:41:42 +0100 Subject: [PATCH] M2 step 3: route every physical dereference through the physmap paging.init now builds the physmap (physToVirt(phys)) alongside the low identity map, so both addressing modes resolve during the transition. tableAt (the page-table walk hinge), the pmm bitmap, the framebuffer, LAPIC/IOAPIC/HPET/PM-timer/SPCR MMIO, the ACPI table walk, AML OperationRegions, the ACPI power registers, the user-ELF frame fills, and the AP trampoline arm/disarm all reach physical memory through the physmap. Hal.mapMmio now maps into the physmap and returns the virtual address, so the device layer never learns the layout. boot_info and its pointees are converted at kmain entry. The kernel still links and runs low; identity is the safety net until it's removed. Suite 27/27. Co-Authored-By: Claude Fable 5 --- src/device/acpi.zig | 25 ++++++++------- src/device/aml/aml.zig | 4 ++- src/device/aml/interp.zig | 10 +++--- src/device/device.zig | 5 ++- src/device/power.zig | 6 ++-- src/kernel/arch/x86_64/apic.zig | 12 +++++-- src/kernel/arch/x86_64/cpu.zig | 7 +++++ src/kernel/arch/x86_64/ioapic.zig | 4 ++- src/kernel/arch/x86_64/paging.zig | 52 ++++++++++++++++++++++++++++--- src/kernel/arch/x86_64/serial.zig | 6 +++- src/kernel/arch/x86_64/smp.zig | 22 ++++++++----- src/kernel/console.zig | 7 ++++- src/kernel/main.zig | 12 ++++--- src/kernel/pmm.zig | 14 ++++++--- src/kernel/tests.zig | 6 ++-- src/kernel/usermode.zig | 10 +++--- 16 files changed, 143 insertions(+), 59 deletions(-) diff --git a/src/device/acpi.zig b/src/device/acpi.zig index da1a84c..827ea1d 100644 --- a/src/device/acpi.zig +++ b/src/device/acpi.zig @@ -147,7 +147,7 @@ var aml_block_count: usize = 0; fn addAmlBlock(sdt_phys: u64) void { if (aml_block_count >= aml_block_phys.len or sdt_phys == 0) return; - const h: *const SystemDescriptorTableHeader = @ptrFromInt(sdt_phys); + const h: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(sdt_phys)); if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return; aml_block_phys[aml_block_count] = sdt_phys + @sizeOf(SystemDescriptorTableHeader); aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader); @@ -376,14 +376,14 @@ pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void { dsdt_phys = 0; aml_block_count = 0; - const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(rsdp_phys); + const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(danos.physToVirt(rsdp_phys)); if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature; // Revision 0 checksums only the first 20 bytes (the v1.0 RSDP). - if (!checksumOk(@ptrFromInt(rsdp_phys), 20)) return error.BadRsdpChecksum; + if (!checksumOk(@ptrFromInt(danos.physToVirt(rsdp_phys)), 20)) return error.BadRsdpChecksum; if (rsdp.revision >= 2) { - const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(rsdp_phys); - if (!checksumOk(@ptrFromInt(rsdp_phys), xsdp.length)) return error.BadXsdpChecksum; + const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(danos.physToVirt(rsdp_phys)); + if (!checksumOk(@ptrFromInt(danos.physToVirt(rsdp_phys)), xsdp.length)) return error.BadXsdpChecksum; try walkRoot(u64, xsdp.extended_system_descriptor_table_address, dt, hal); } else { try walkRoot(u32, rsdp.root_system_description_table_address, dt, hal); @@ -393,7 +393,7 @@ pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void { // read the sleep types from it. var blocks: [aml_block_phys.len][]const u8 = undefined; for (0..aml_block_count) |i| { - blocks[i] = @as([*]const u8, @ptrFromInt(aml_block_phys[i]))[0..aml_block_len[i]]; + blocks[i] = @as([*]const u8, @ptrFromInt(danos.physToVirt(aml_block_phys[i])))[0..aml_block_len[i]]; } const active = blocks[0..aml_block_count]; if (aml.parse(dt.allocator, active)) |pr| { @@ -412,11 +412,11 @@ pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void { /// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch /// each SDT it points at. A bad individual table is skipped, not fatal. fn walkRoot(comptime Entry: type, root_phys: u64, dt: *DeviceTree, hal: Hal) !void { - const header: *const SystemDescriptorTableHeader = @ptrFromInt(root_phys); - if (!checksumOk(@ptrFromInt(root_phys), header.length)) return error.BadRootChecksum; + const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(root_phys)); + if (!checksumOk(@ptrFromInt(danos.physToVirt(root_phys)), header.length)) return error.BadRootChecksum; const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry); - const base: [*]const u8 = @ptrFromInt(root_phys); + const base: [*]const u8 = @ptrFromInt(danos.physToVirt(root_phys)); const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader)); for (entries[0..count]) |ent| { @@ -427,7 +427,7 @@ fn walkRoot(comptime Entry: type, root_phys: u64, dt: *DeviceTree, hal: Hal) !vo /// Dispatch a single SDT on its signature. fn handleTable(dt: *DeviceTree, hal: Hal, sdt_phys: u64) !void { - const header: *const SystemDescriptorTableHeader = @ptrFromInt(sdt_phys); + const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(sdt_phys)); const sig = header.signature; if (std.mem.eql(u8, &sig, &APIC)) { try parseMadt(dt, header); @@ -1080,8 +1080,9 @@ fn pciConfigPtr(alloc: McfgAllocation, hal: Hal, bus: u8, dev: u8, func: u8) [*] (@as(u64, bus - alloc.start_bus) << 20) + (@as(u64, dev) << 15) + (@as(u64, func) << 12); - hal.mapMmio(phys, phys, true); // identity-map this config page (writable) - return @ptrFromInt(phys); + // Map the config page (writable, for BAR sizing) and use the virtual + // address the HAL hands back. + return @ptrFromInt(hal.mapMmio(phys, danos.page_size, true)); } /// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer. diff --git a/src/device/aml/aml.zig b/src/device/aml/aml.zig index e04380a..4ec6ab1 100644 --- a/src/device/aml/aml.zig +++ b/src/device/aml/aml.zig @@ -173,7 +173,9 @@ test "parses a nested namespace and finds the sleep package" { try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b); } -fn noMap(_: u64, _: u64, _: bool) void {} +fn noMap(phys: u64, _: u64, _: bool) u64 { + return phys; +} fn noRead(_: u8, _: u16) u32 { return 0; } diff --git a/src/device/aml/interp.zig b/src/device/aml/interp.zig index 3647198..318baa8 100644 --- a/src/device/aml/interp.zig +++ b/src/device/aml/interp.zig @@ -21,7 +21,7 @@ const Namespace = nsp.Namespace; /// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio). pub const Hal = struct { - mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void, + mapMmio: *const fn (phys: u64, len: u64, writable: bool) u64, pioRead: *const fn (width: u8, port: u16) u32, pioWrite: *const fn (width: u8, port: u16, value: u32) void, }; @@ -665,8 +665,8 @@ pub const Interp = struct { fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 { switch (space) { 0 => { // SystemMemory - self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true); - const p: *align(1) const volatile u8 = @ptrFromInt(addr); + const virt = self.hal.mapMmio(addr & ~@as(u64, 0xFFF), 0x1000, true); + const p: *align(1) const volatile u8 = @ptrFromInt(virt + (addr & 0xFFF)); return p.*; }, 1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO @@ -677,8 +677,8 @@ pub const Interp = struct { fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void { switch (space) { 0 => { - self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true); - const p: *align(1) volatile u8 = @ptrFromInt(addr); + const virt = self.hal.mapMmio(addr & ~@as(u64, 0xFFF), 0x1000, true); + const p: *align(1) volatile u8 = @ptrFromInt(virt + (addr & 0xFFF)); p.* = value; }, 1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value), diff --git a/src/device/device.zig b/src/device/device.zig index 2684ca3..782d0ac 100644 --- a/src/device/device.zig +++ b/src/device/device.zig @@ -18,7 +18,10 @@ const std = @import("std"); /// layer touches hardware without importing `arch` — the same discipline that lets /// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4). pub const Hal = struct { - mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void, + /// Map a physical MMIO range and return the virtual address to reach it at. + /// The device layer dereferences the returned address and never learns how + /// the kernel places it (identity, physmap, a window — the kernel's choice). + mapMmio: *const fn (phys: u64, len: u64, writable: bool) u64, pioRead: *const fn (width: u8, port: u16) u32, pioWrite: *const fn (width: u8, port: u16, value: u32) void, }; diff --git a/src/device/power.zig b/src/device/power.zig index 53574fb..4ad7796 100644 --- a/src/device/power.zig +++ b/src/device/power.zig @@ -78,8 +78,7 @@ fn sleepValue(slp_typ: u8) u32 { fn readReg(hal: Hal, reg: acpi.RegAccess) u32 { if (reg.mmio) { - hal.mapMmio(reg.address, reg.address, true); - const p: *align(1) volatile u32 = @ptrFromInt(reg.address); + const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(reg.address, 4, true)); return p.*; } return hal.pioRead(reg.width, @intCast(reg.address)); @@ -87,8 +86,7 @@ fn readReg(hal: Hal, reg: acpi.RegAccess) u32 { fn writeReg(hal: Hal, reg: acpi.RegAccess, value: u32) void { if (reg.mmio) { - hal.mapMmio(reg.address, reg.address, true); - const p: *align(1) volatile u32 = @ptrFromInt(reg.address); + const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(reg.address, 4, true)); p.* = value; } else { hal.pioWrite(reg.width, @intCast(reg.address), value); diff --git a/src/kernel/arch/x86_64/apic.zig b/src/kernel/arch/x86_64/apic.zig index e18d9b7..04621c7 100644 --- a/src/kernel/arch/x86_64/apic.zig +++ b/src/kernel/arch/x86_64/apic.zig @@ -9,6 +9,7 @@ //! map). Every interrupt must be acknowledged with an end-of-interrupt write, or //! the LAPIC won't deliver the next one. +const danos = @import("danos"); const io = @import("io.zig"); const paging = @import("paging.zig"); @@ -118,7 +119,8 @@ pub fn init() void { if (cfg_pic_present) remapAndMaskPic(); const msr = io.rdmsr(ia32_apic_base_msr); - base = @intCast(msr & 0xFFFFF000); // physical base is bits 12+ + // Reach the LAPIC through the physmap (paging.init maps its page there). + base = @intCast(danos.physToVirt(msr & 0xFFFFF000)); // physical base is bits 12+ io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable @@ -299,8 +301,10 @@ fn hpetWrite64(off: usize, value: u64) void { } /// Map + enable the HPET and return its tick frequency, or null if unusable. +/// Maps the HPET into the physmap and switches cfg_hpet_base to that virtual +/// address, so the register accessors reach it without the identity map. fn hpetHz() ?u64 { - paging.map(cfg_hpet_base & ~@as(u64, 0xFFF), cfg_hpet_base & ~@as(u64, 0xFFF), true); + cfg_hpet_base = paging.mapMmio(cfg_hpet_base, 0x400, true); const caps = hpetRead64(0x00); const period_fs = caps >> 32; // femtoseconds per tick if (period_fs == 0) return null; @@ -319,7 +323,9 @@ fn readHpet() u64 { fn readPmTimer() u64 { const pt = cfg_pm_timer.?; - if (pt.mmio) return @as(*volatile u32, @ptrFromInt(pt.address)).*; + // MMIO PM timer via the physmap (mapMmio is idempotent); the common case is + // a legacy I/O port. + if (pt.mmio) return @as(*volatile u32, @ptrFromInt(paging.mapMmio(pt.address, 4, false))).*; return io.inl(@intCast(pt.address)); } diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index 40bb985..a14e988 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -71,6 +71,13 @@ pub fn mapPage(virt: u64, phys: u64, writable: bool) void { paging.map(virt, phys, writable); } +/// Map a device MMIO range and return the virtual address to reach it at. This +/// is the device layer's `Hal.mapMmio` — it hands back a physmap pointer and +/// never exposes how the mapping is placed. +pub fn mapMmio(phys: u64, len: u64, writable: bool) u64 { + return paging.mapMmio(phys, len, writable); +} + /// Remove a kernel mapping. pub fn unmapPage(virt: u64) void { paging.unmap(virt); diff --git a/src/kernel/arch/x86_64/ioapic.zig b/src/kernel/arch/x86_64/ioapic.zig index e785432..debf2cc 100644 --- a/src/kernel/arch/x86_64/ioapic.zig +++ b/src/kernel/arch/x86_64/ioapic.zig @@ -52,7 +52,9 @@ fn writeEntry(n: u32, low: u32, high: u32) void { /// Map the I/O APIC and mask every redirection entry — the safe quiescent state. pub fn init() void { if (base == 0) return; - paging.map(base & ~@as(u64, 0xFFF), base & ~@as(u64, 0xFFF), true); + // Reach the I/O APIC through the physmap; switch `base` to that virtual + // address so the register accessors work without the identity map. + base = paging.mapMmio(base, 0x1000, true); max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1; var n: u32 = 0; while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0); diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index cbd4cb4..6d4ff75 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -30,8 +30,14 @@ const pf_w: u32 = 2; var kernel_pml4: u64 = 0; var alloc_frame: *const fn () ?u64 = undefined; +/// Dereference a page-table frame by its physical address, via the physmap. +/// This is the single hinge for the higher-half move: page tables hold physical +/// frame addresses (pmm gives out physical frames, and CR3/PTEs must be +/// physical), but the kernel reaches them at `physmap_base + phys`. Valid under +/// both the loader's bootstrap tables and the kernel's own, which share the +/// physmap base. fn tableAt(phys: u64) *[512]u64 { - return @ptrFromInt(phys); + return @ptrFromInt(danos.physToVirt(phys)); } fn allocTable() u64 { @@ -71,8 +77,19 @@ fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void { } } +/// Map [phys_base, phys_base+len) into the physmap (at physToVirt(phys)) with +/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent +/// window onto physical memory once the low identity map goes away. +fn mapRangePhysmap(pml4: u64, phys_base: u64, len: u64, flags: u64) void { + var addr = phys_base & ~@as(u64, page_size - 1); + const end = phys_base + len; + while (addr < end) : (addr += page_size) { + mapPage(pml4, danos.physToVirt(addr), addr, flags); + } +} + fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion { - return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len]; + return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(mm.regions)))[0..mm.len]; } /// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we @@ -88,16 +105,22 @@ pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) voi enableNx(); const pml4 = allocTable(); - // 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays - // unmapped unless mapped explicitly below. + // 1. All RAM in the physmap (physToVirt(phys)) RW + NX, plus — during the + // higher-half transition — a low identity map so any not-yet-converted + // physical deref still resolves. Non-RAM (MMIO) is skipped here and + // mapped explicitly below. The identity half is removed in a later step. for (regions(boot_info.memory_map)) |r| { if (r.kind == .mmio) continue; + mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute); mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute); } - // 2. The framebuffer and the Local APIC (device memory we need), RW + NX. + // 2. The framebuffer and the Local APIC (device memory we need), RW + NX — + // in the physmap and (transitionally) identity. const fb = boot_info.framebuffer; + mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute); mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute); + mapPage(pml4, danos.physToVirt(0xFEE00000), 0xFEE00000, present | writable | no_execute); mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute); // 3. Overlay the kernel's own segments with their real ELF permissions, @@ -130,6 +153,25 @@ pub fn map(virt: u64, phys: u64, writable_page: bool) void { invalidate(virt); } +/// Map a device MMIO range into the physmap and return the virtual address to +/// use for it (physToVirt(phys)). The single way the kernel (and the device +/// layer, via the HAL) reaches memory-mapped registers once the identity map is +/// gone: physmap pages are RW + NX, so a driver never executes device memory. +/// Idempotent for already-mapped ranges. `len` 0 maps one page. +pub fn mapMmio(phys: u64, len: u64, writable_page: bool) u64 { + var flags: u64 = present | no_execute; + if (writable_page) flags |= writable; + const first = phys & ~@as(u64, page_size - 1); + const last = phys + (if (len == 0) 1 else len) - 1; + var addr = first; + while (addr <= (last & ~@as(u64, page_size - 1))) : (addr += page_size) { + const virt = danos.physToVirt(addr); + mapPage(kernel_pml4, virt, addr, flags); + invalidate(virt); + } + return danos.physToVirt(phys); +} + /// Like `descend`, but also sets the U/S bit on the intermediate entry (new or /// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves /// existing entries untouched. Only used under user-exclusive virtual ranges, so diff --git a/src/kernel/arch/x86_64/serial.zig b/src/kernel/arch/x86_64/serial.zig index 9b2ce07..990e903 100644 --- a/src/kernel/arch/x86_64/serial.zig +++ b/src/kernel/arch/x86_64/serial.zig @@ -9,6 +9,8 @@ //! optimistically to COM1 (harmless if absent); the framebuffer console is the //! always-present log. +const paging = @import("paging.zig"); + /// How the UART registers are reached: legacy I/O ports or memory-mapped. const Access = enum { port, mmio }; @@ -61,7 +63,9 @@ pub fn init() void { /// re-run the UART setup there. Called after discovery when an SPCR entry exists. pub fn reconfigure(is_mmio: bool, addr: u64) void { access = if (is_mmio) .mmio else .port; - base = addr; + // An MMIO UART is reached through the physmap; an I/O-port UART keeps its + // port number unchanged. + base = if (is_mmio) paging.mapMmio(addr, 0x100, true) else addr; init(); } diff --git a/src/kernel/arch/x86_64/smp.zig b/src/kernel/arch/x86_64/smp.zig index 1b849cc..2b92ba4 100644 --- a/src/kernel/arch/x86_64/smp.zig +++ b/src/kernel/arch/x86_64/smp.zig @@ -13,6 +13,7 @@ //! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic //! scheduler entry and joins the run loop — mechanism here, policy there. +const danos = @import("danos"); const io = @import("io.zig"); const gdt = @import("gdt.zig"); const tss = @import("tss.zig"); @@ -75,22 +76,27 @@ pub fn trampolinePage() u64 { /// Arm the trampoline for a wake: make its page executable (W^X exception for the /// duration of the climb) and copy the blob in. fn arm() void { + // The AP executes this page at its physical address (identity) while it + // climbs from real to long mode, so it needs a low identity mapping that is + // executable — the one deliberate, transient W^X exception. The BSP writes + // the blob into the frame through the physmap. paging.setExecutable(tramp_phys); const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" }); const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" }); const len = @intFromPtr(end) - @intFromPtr(start); - const dst: [*]u8 = @ptrFromInt(tramp_phys); + const dst: [*]u8 = @ptrFromInt(danos.physToVirt(tramp_phys)); @memcpy(dst[0..len], start[0..len]); } -/// Disarm after a wake: wipe the page and restore it to inert RW+NX, so no -/// executable code (nor any stale bytes) lingers between wakes. Safe to run once the -/// woken core has reported in — it's long past the trampoline by then, in the kernel -/// image; a core that never answered is dead and can't be mid-climb. +/// Disarm after a wake: wipe the page through the physmap and remove its low +/// identity mapping, so no executable code (nor any stale bytes, nor any +/// low-half mapping) lingers between wakes. Safe once the woken core has +/// reported in — it's long past the trampoline by then, in the kernel image; a +/// core that never answered is dead and can't be mid-climb. fn disarm() void { - const dst: [*]u8 = @ptrFromInt(tramp_phys); + const dst: [*]u8 = @ptrFromInt(danos.physToVirt(tramp_phys)); @memset(dst[0..page_size], 0); - paging.map(tramp_phys, tramp_phys, true); // RW + NX, like every other RAM frame + paging.unmap(tramp_phys); // drop the transient low identity mapping } /// Address of a patchable trampoline parameter, by symbol name: the copied blob's @@ -100,7 +106,7 @@ fn disarm() void { fn param(comptime name: []const u8) *align(1) volatile u64 { const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" })); const sym = @intFromPtr(@extern([*]const u8, .{ .name = name })); - return @ptrFromInt(tramp_phys + (sym - start)); + return @ptrFromInt(danos.physToVirt(tramp_phys + (sym - start))); } /// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it diff --git a/src/kernel/console.zig b/src/kernel/console.zig index 149fdfe..5cb3d38 100644 --- a/src/kernel/console.zig +++ b/src/kernel/console.zig @@ -67,8 +67,13 @@ pub const Console = struct { bg: u32 = 0x0000_0000, // black pub fn init(fb: danos.Framebuffer) Console { + // Reach the framebuffer through the physmap, so the pointer stays valid + // once the low identity map is gone. The base is mapped by both the + // loader's bootstrap tables and paging.init. + var mapped = fb; + if (fb.base != 0) mapped.base = danos.physToVirt(fb.base); return .{ - .fb = fb, + .fb = mapped, .cols = fb.width / glyph_w, .rows = fb.height / glyph_h, }; diff --git a/src/kernel/main.zig b/src/kernel/main.zig index b4a16e7..5e13062 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -41,7 +41,11 @@ var ap_trampoline_page: u64 = 0; /// pointer to the handoff data. There is no runtime, no stack unwinding, and no /// caller to return to, so this never returns. export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn { - kmain(boot_info); + // The loader passes a low physical pointer (its own stack). Reach it — and + // everything it points at — through the physmap, so it stays valid once the + // low identity map is gone. The physmap base is the same under the loader's + // bootstrap tables and the kernel's own. + kmain(@ptrFromInt(danos.physToVirt(@intFromPtr(boot_info)))); } fn kmain(boot_info: *const BootInfo) noreturn { @@ -82,7 +86,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Summarise the physical memory the loader handed us. The array is danos's // own MemoryRegion, so this is a plain slice — no firmware layout in sight. - const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len]; + const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(boot_info.memory_map.regions)))[0..boot_info.memory_map.len]; var usable_pages: u64 = 0; var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM for (regions) |r| { @@ -141,7 +145,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { // mapped) and maps PCIe config space on demand via the VMM. A failure here is // not fatal yet — log it and carry on. const hal = platform.Hal{ - .mapMmio = arch.mapPage, + .mapMmio = arch.mapMmio, .pioRead = arch.pioRead, .pioWrite = arch.pioWrite, }; @@ -253,7 +257,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { // idle). init becomes a real schedulable process in M3. if (boot_info.init_len != 0) { status("starting /sbin/init...\n"); - const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len]; + const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; scheduler.setPreemption(false); const code = usermode.runInitElf(image); scheduler.setPreemption(true); diff --git a/src/kernel/pmm.zig b/src/kernel/pmm.zig index 4349a56..7c62282 100644 --- a/src/kernel/pmm.zig +++ b/src/kernel/pmm.zig @@ -49,12 +49,16 @@ inline fn setFree(frame: usize) void { } fn regions(map: danos.MemoryMap) []const danos.MemoryRegion { - return @as([*]const danos.MemoryRegion, @ptrFromInt(map.regions))[0..map.len]; + return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(map.regions)))[0..map.len]; } -/// Build the allocator from the loader's memory map. Relies on the firmware's -/// identity mapping still being in effect (a physical address is usable directly -/// as a pointer) — true until the kernel installs its own page tables. +/// Build the allocator from the loader's memory map. Reaches physical memory +/// (the region array, the bitmap's own storage) through the physmap, which the +/// loader's bootstrap tables already provide — so this works before the kernel +/// installs its own tables. Invariant: the bitmap lands in the first usable +/// region (lowest address), which must sit under the bootstrap physmap's reach +/// (4 GiB); it always does, as both this and the page-table allocator scan from +/// low addresses up. pub fn init(map: danos.MemoryMap) void { const regs = regions(map); @@ -87,7 +91,7 @@ pub fn init(map: danos.MemoryMap) void { } } const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap"); - bitmap = @as([*]u8, @ptrFromInt(bitmap_base))[0..bitmap_bytes]; + bitmap = @as([*]u8, @ptrFromInt(danos.physToVirt(bitmap_base)))[0..bitmap_bytes]; // 3. Start with everything marked used, then free the usable regions. Doing // it this way means every gap, reserved span and MMIO hole is unallocatable diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index c31aabd..60b4432 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -110,7 +110,7 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { fn platformHal() platform.Hal { return .{ - .mapMmio = arch.mapPage, + .mapMmio = arch.mapMmio, .pioRead = arch.pioRead, .pioWrite = arch.pioWrite, }; @@ -552,7 +552,7 @@ fn smpTest() void { const tramp = arch.trampolinePage(); check("trampoline frame reserved", tramp != 0); if (tramp != 0) { - const bytes: [*]const u8 = @ptrFromInt(tramp); + const bytes: [*]const u8 = @ptrFromInt(danos.physToVirt(tramp)); var zeroed = true; for (0..4096) |b| { if (bytes[b] != 0) zeroed = false; @@ -753,7 +753,7 @@ fn initTest(boot_info: *const BootInfo) void { result(); return; } - const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len]; + const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; sched.setPreemption(false); // see userTest: pins the run to this core's rsp0 const code = usermode.runInitElf(image); sched.setPreemption(true); diff --git a/src/kernel/usermode.zig b/src/kernel/usermode.zig index 0294436..2971731 100644 --- a/src/kernel/usermode.zig +++ b/src/kernel/usermode.zig @@ -128,10 +128,10 @@ pub fn run(blob: []const u8) RunError!void { return error.OutOfMemory; }; - // Fill the code frame through its identity mapping (supervisor RW): the - // user-facing mapping is read-only, and this also sidesteps CR0.WP/SMAP. - // The tail is padded with int3 so a stray jump traps instead of sliding. - const code: [*]u8 = @ptrFromInt(code_frame); + // Fill the code frame through the physmap (supervisor RW): the user-facing + // mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is + // padded with int3 so a stray jump traps instead of sliding. + const code: [*]u8 = @ptrFromInt(danos.physToVirt(code_frame)); @memcpy(code[0..blob.len], blob); @memset(code[blob.len..page_size], 0xCC); @@ -251,7 +251,7 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru /// with the segment's W^X permissions. Records the page for rollback. fn loadPage(image: []const u8, seg: Segment, page_index: u64) InitError!void { const frame = pmm.alloc() orelse return error.OutOfMemory; - const dst: [*]u8 = @ptrFromInt(frame); + const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame)); @memset(dst[0..page_size], 0); const page_off = page_index * page_size; if (page_off < seg.filesz) {