docs+code: spell out aspace/vaddr/paddr per coding standards
Expand the abbreviations flagged in docs/coding-standards.md (names spelled
out in full unless an acronym) across the kernel, runtime, ABI, tests, and
docs:
aspace -> address_space (AspaceRef -> AddressSpaceRef, retainAspace ->
retainAddressSpace, loaded_aspace -> loaded_address_space, the
liveAspaceCount/aspaceDestroyCount test hooks, etc.)
vaddr -> virtual_address
paddr -> physical_address
The kernel test case and its serial markers are renamed to match:
aspace-refcount -> address-space-refcount (kernel dispatch string and
test/qemu_test.py case name kept in sync). Prose in docs uses the natural
"address space"/"virtual address"; backticked field/identifier references
use the code spelling.
Also expand the bare "AS" abbreviation in three ABI comments and reframe the
set_thread_pointer ABI/handler docs to lead with the arch-neutral concept
(user-space TLS thread pointer; x86_64 IA32_FS_BASE, aarch64 TPIDR_EL0)
rather than x86 FS-first, matching scheduler.zig's existing framing.
Foreign ABI names preserved: the ELF p_vaddr field and mmap/mmio remain.
Verified: zig build, zig build test, and the full 25-case QEMU guardrail
suite all green.
This commit is contained in:
+78
-77
@@ -59,7 +59,7 @@ pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pa
|
||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||
/// process bump-allocates from `heap_arena_base` upward via a per-address-space cursor
|
||||
/// (`scheduler.aspaceMmapNextPtr`, shared by its threads); a 1 GiB window is far more
|
||||
/// (`scheduler.addressSpaceMmapNextPtr`, shared by its threads); a 1 GiB window is far more
|
||||
/// than any user heap needs today.
|
||||
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
||||
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
||||
@@ -71,7 +71,7 @@ pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
||||
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
||||
/// device pages user-accessible widens no kernel mapping. Per-address-space cursor
|
||||
/// (`scheduler.aspaceDeviceMapNextPtr`).
|
||||
/// (`scheduler.addressSpaceDeviceMapNextPtr`).
|
||||
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
|
||||
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||
|
||||
@@ -164,7 +164,7 @@ fn fail(state: *architecture.CpuState) void {
|
||||
|
||||
fn system_call(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
const user = t.aspace != 0;
|
||||
const user = t.address_space != 0;
|
||||
if (user) {
|
||||
// A condemned process (process_kill caught it running) dies at its next
|
||||
// kernel entry — before it can spawn, claim, or message anything else.
|
||||
@@ -317,7 +317,7 @@ fn systemIpcReplyWait(state: *architecture.CpuState) void {
|
||||
fn systemIpcSend(state: *architecture.CpuState) void {
|
||||
const me = scheduler.current();
|
||||
const endpoint = ipc.resolveHandle(me, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const r = ipc.send(endpoint, me.aspace, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), me.id);
|
||||
const r = ipc.send(endpoint, me.address_space, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), me.id);
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
}
|
||||
|
||||
@@ -327,7 +327,7 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||
const maximum = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
||||
@@ -351,14 +351,14 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
} else fail(state);
|
||||
}
|
||||
|
||||
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
||||
/// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into
|
||||
/// this address space (strong-uncacheable) and return the register base address.
|
||||
/// The claim is the capability — a process can only map hardware it owns.
|
||||
fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const resource_index = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
// Read the broker table under the lock: ring-3 device_register (M19) now
|
||||
// mutates it concurrently on other cores, so a lock-free read here could
|
||||
// see a torn resource (and a torn length used to panic the arithmetic
|
||||
@@ -387,11 +387,11 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
// same as mmap (docs/threading-plan.md M7).
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const cursor = scheduler.aspaceDeviceMapNextPtr(t.aspace) orelse return fail(state);
|
||||
const cursor = scheduler.addressSpaceDeviceMapNextPtr(t.address_space) orelse return fail(state);
|
||||
if (cursor.* == 0) cursor.* = device_arena_base; // seed the arena lazily
|
||||
const base_v = cursor.*;
|
||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len, write_combining);
|
||||
architecture.mapUserDeviceInto(t.address_space, base_v, r.start, r.len, write_combining);
|
||||
cursor.* = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||
}
|
||||
@@ -421,7 +421,7 @@ pub fn resolveIoPort(t: *scheduler.Task, device_id: u64, resource_index: u64, of
|
||||
/// is fine. See docs/drivers.md.
|
||||
fn systemIoRead(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const width = architecture.systemCallArg(state, 3);
|
||||
const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state);
|
||||
architecture.setSystemCallResult(state, architecture.pioRead(@intCast(width), port));
|
||||
@@ -432,14 +432,14 @@ fn systemIoRead(state: *architecture.CpuState) void {
|
||||
/// gate as `io_read`.
|
||||
fn systemIoWrite(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const width = architecture.systemCallArg(state, 3);
|
||||
const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state);
|
||||
architecture.pioWrite(@intCast(width), port, @intCast(architecture.systemCallArg(state, 4)));
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): grant `len` bytes (rounded up to
|
||||
/// dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): grant `len` bytes (rounded up to
|
||||
/// whole pages) of DMA-capable memory — physically contiguous, zeroed, pinned, and
|
||||
/// strong-uncacheable (coherent) — mapping it into the caller's DMA arena and handing
|
||||
/// back both the virtual address to touch and the physical address to program into the
|
||||
@@ -451,7 +451,7 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const flags = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or len == 0) return fail(state);
|
||||
if (t.address_space == 0 or len == 0) return fail(state);
|
||||
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
const max_phys: u64 = if (flags & abi.dma_below_4g != 0) (@as(u64, 4) << 30) else ~@as(u64, 0);
|
||||
@@ -467,14 +467,14 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
// Zero through the physmap (the frames aren't mapped in the caller yet), then map.
|
||||
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
|
||||
@memset(kernel_view[0 .. pages * page_size], 0);
|
||||
architecture.mapUserDmaInto(t.aspace, base_v, phys, pages * page_size);
|
||||
architecture.mapUserDmaInto(t.address_space, base_v, phys, pages * page_size);
|
||||
|
||||
t.dma_map_next = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
||||
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
||||
}
|
||||
|
||||
/// dma_free(vaddr, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||
/// it can never unmap-and-free the caller's stack, heap, or an MMIO grant; only pages
|
||||
/// actually mapped are freed (an unmapped hole is skipped). Teardown also reclaims any
|
||||
/// DMA pages left mapped at exit (they carry no `device_grant`, so `freeSubtree` frees
|
||||
@@ -483,21 +483,21 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
const base_v = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base_v + i * page_size;
|
||||
if (architecture.translate(t.aspace, va)) |phys| {
|
||||
architecture.unmapUserPageInto(t.aspace, va);
|
||||
if (architecture.translate(t.address_space, va)) |phys| {
|
||||
architecture.unmapUserPageInto(t.address_space, va);
|
||||
pmm.free(phys);
|
||||
}
|
||||
}
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
||||
/// shm_create(len) -> virtual_address (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
||||
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
|
||||
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
|
||||
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
|
||||
@@ -508,7 +508,7 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
fn systemShmCreate(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or len == 0) return fail(state);
|
||||
if (t.address_space == 0 or len == 0) return fail(state);
|
||||
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
|
||||
@@ -533,20 +533,20 @@ fn systemShmCreate(state: *architecture.CpuState) void {
|
||||
return fail(state);
|
||||
}
|
||||
|
||||
architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size);
|
||||
architecture.mapUserSharedInto(t.address_space, base_v, phys, pages * page_size);
|
||||
t.shm_map_next = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v); // vaddr for the CPU
|
||||
architecture.setSystemCallResult(state, base_v); // virtual_address for the CPU
|
||||
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
|
||||
}
|
||||
|
||||
/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller
|
||||
/// shm_map(cap) -> virtual_address: map the shared region named by a capability handle the caller
|
||||
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
|
||||
/// creator sees — returning the virtual address. The handle already holds a reference (taken
|
||||
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
|
||||
fn systemShmMap(state: *architecture.CpuState) void {
|
||||
const cap = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
|
||||
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||
@@ -554,12 +554,12 @@ fn systemShmMap(state: *architecture.CpuState) void {
|
||||
const size = shm.pages * page_size;
|
||||
if (base_v + size > shm_arena_end) return fail(state);
|
||||
|
||||
architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size);
|
||||
architecture.mapUserSharedInto(t.address_space, base_v, shm.phys, size);
|
||||
t.shm_map_next = base_v + size;
|
||||
architecture.setSystemCallResult(state, base_v);
|
||||
}
|
||||
|
||||
/// shm_physical(cap) -> paddr: the guest-physical base of a shared region the caller holds a
|
||||
/// shm_physical(cap) -> physical_address: the guest-physical base of a shared region the caller holds a
|
||||
/// capability for. The frames are contiguous (allocated by `allocContiguous`), so a single
|
||||
/// physical base + length describes the whole region — which is exactly what a driver needs
|
||||
/// to hand a shm surface to a device (virtio-gpu `attach_backing`). Only a holder of the
|
||||
@@ -567,7 +567,7 @@ fn systemShmMap(state: *architecture.CpuState) void {
|
||||
fn systemShmPhysical(state: *architecture.CpuState) void {
|
||||
const cap = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||
architecture.setSystemCallResult(state, shm.phys);
|
||||
}
|
||||
@@ -588,10 +588,10 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||
const parent_id = architecture.systemCallArg(state, 0);
|
||||
const descriptor_ptr = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
|
||||
var descriptor: device_abi.DeviceDescriptor = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
if (!ipc.copyFromUser(t.address_space, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
|
||||
// Under the big kernel lock: the broker's table is also mutated by the
|
||||
// death sweep (releaseAllOwnedBy) and read by enumerate on other cores —
|
||||
@@ -675,7 +675,7 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
|
||||
const arg = architecture.systemCallArg(state, 2);
|
||||
const exit_handle = architecture.systemCallArg(state, 3);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // kernel tasks own no address space to share
|
||||
if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share
|
||||
if (entry == 0 or entry >= user_half_end) return fail(state);
|
||||
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
|
||||
// The endpoint the thread notifies on exit (how join waits), or none.
|
||||
@@ -683,17 +683,17 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
|
||||
null
|
||||
else
|
||||
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||
const tid = spawnThreadSupervised(t.aspace, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state);
|
||||
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state);
|
||||
architecture.setSystemCallResult(state, tid);
|
||||
}
|
||||
|
||||
/// Spawn a thread sharing `aspace`, taking the exit-endpoint reference under the **same**
|
||||
/// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same**
|
||||
/// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before
|
||||
/// its reference exists. Returns the new thread id, or null on resource exhaustion.
|
||||
fn spawnThreadSupervised(aspace: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 {
|
||||
fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const tid = scheduler.spawnUserLocked(aspace, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null;
|
||||
const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null;
|
||||
if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death
|
||||
return tid;
|
||||
}
|
||||
@@ -708,13 +708,14 @@ fn systemThreadSelf(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, scheduler.currentId());
|
||||
}
|
||||
|
||||
/// set_thread_pointer(addr) -> 0: set the caller's FS base (its user-space TLS thread
|
||||
/// pointer). The kernel never uses FS; the scheduler restores this per task across context
|
||||
/// switches (docs/threading-plan.md M10). `addr` must be a user-half address.
|
||||
/// set_thread_pointer(addr) -> 0: set the caller's user-space TLS thread pointer. The
|
||||
/// arch layer maps it to IA32_FS_BASE on x86_64, `TPIDR_EL0` on aarch64; the kernel
|
||||
/// never reads it, and the scheduler restores it per task across context switches
|
||||
/// (docs/threading-plan.md M10). `addr` must be a user-half address.
|
||||
fn systemSetThreadPointer(state: *architecture.CpuState) void {
|
||||
const addr = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // kernel tasks have no user TLS
|
||||
if (t.address_space == 0) return fail(state); // kernel tasks have no user TLS
|
||||
if (addr >= user_half_end) return fail(state);
|
||||
const flags = sync.enter();
|
||||
scheduler.setThreadPointerLocked(addr);
|
||||
@@ -728,7 +729,7 @@ fn systemSetThreadPointer(state: *architecture.CpuState) void {
|
||||
fn systemThreadJoin(state: *architecture.CpuState) void {
|
||||
const tid: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // kernel tasks don't join
|
||||
if (t.address_space == 0) return fail(state); // kernel tasks don't join
|
||||
const flags = sync.enter();
|
||||
scheduler.joinThreadLocked(tid);
|
||||
sync.leave(flags);
|
||||
@@ -745,12 +746,12 @@ fn systemFutexWait(state: *architecture.CpuState) void {
|
||||
const expected: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||
const timeout_ns = architecture.systemCallArg(state, 2);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state);
|
||||
|
||||
const flags = sync.enter();
|
||||
var word_bytes: [4]u8 = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, addr, &word_bytes)) {
|
||||
if (!ipc.copyFromUser(t.address_space, addr, &word_bytes)) {
|
||||
sync.leave(flags);
|
||||
return fail(state);
|
||||
}
|
||||
@@ -774,10 +775,10 @@ fn systemFutexWake(state: *architecture.CpuState) void {
|
||||
const addr = architecture.systemCallArg(state, 0);
|
||||
const count: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state);
|
||||
const flags = sync.enter();
|
||||
const woken = scheduler.futexWakeLocked(t.aspace, addr, count);
|
||||
const woken = scheduler.futexWakeLocked(t.address_space, addr, count);
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, woken);
|
||||
}
|
||||
@@ -792,7 +793,7 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||
const maximum = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
const sz = @sizeOf(abi.ProcessDescriptor);
|
||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||
const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr);
|
||||
@@ -805,7 +806,7 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
||||
/// cannot be a weapon (ids are never reused, so a stale one just misses).
|
||||
fn systemProcessKill(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
const r = killProcess(t.id, @intCast(id));
|
||||
@@ -938,7 +939,7 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
||||
if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||
if (target.address_space == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||
if (target.supervisor != caller_id) return -ipc.EPERM;
|
||||
target.exit_reason = .killed;
|
||||
if (target.state == .running) {
|
||||
@@ -1008,7 +1009,7 @@ var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exi
|
||||
/// secret between cooperating processes. -ENOSPC when the table is full.
|
||||
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
@@ -1028,7 +1029,7 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||
/// delivered immediately on bind, coalesced into one notification.
|
||||
fn systemSignalBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
@@ -1048,7 +1049,7 @@ fn systemSignalBind(state: *architecture.CpuState) void {
|
||||
/// targets accumulate the signal in their pending mask.
|
||||
fn systemProcessSignal(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
const signal = architecture.systemCallArg(state, 1);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
@@ -1056,7 +1057,7 @@ fn systemProcessSignal(state: *architecture.CpuState) void {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH);
|
||||
if (target.aspace == 0) return failErr(state, ipc.ESRCH);
|
||||
if (target.address_space == 0) return failErr(state, ipc.ESRCH);
|
||||
if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM);
|
||||
target.pending_signals |= @as(u32, 1) << @intCast(signal);
|
||||
if (target.signal_endpoint) |raw| {
|
||||
@@ -1093,7 +1094,7 @@ fn timerSweepLocked() void {
|
||||
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
|
||||
fn systemTimerBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const ms = architecture.systemCallArg(state, 1);
|
||||
const flags = sync.enter();
|
||||
@@ -1110,7 +1111,7 @@ fn systemTimerBind(state: *architecture.CpuState) void {
|
||||
|
||||
fn systemProcessExitReason(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
const r = exitReasonOf(t.id, @intCast(id));
|
||||
@@ -1136,7 +1137,7 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
||||
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
||||
fn systemIrqBind(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
||||
@@ -1156,7 +1157,7 @@ fn systemIrqBind(state: *architecture.CpuState) void {
|
||||
fn systemMsiBind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||
if (owner != t.id) return fail(state); // not claimed by this process
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||
@@ -1175,7 +1176,7 @@ fn systemMsiBind(state: *architecture.CpuState) void {
|
||||
/// more arrives until the driver says it has serviced the hardware.
|
||||
fn systemIrqAck(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (t.address_space == 0) return fail(state);
|
||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||
return fail(state);
|
||||
|
||||
@@ -1263,7 +1264,7 @@ fn systemKlogRead(state: *architecture.CpuState) void {
|
||||
fn systemMmap(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
|
||||
if (t.address_space == 0) return fail(state); // not a user process — nothing to map into
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
|
||||
|
||||
@@ -1275,7 +1276,7 @@ fn systemMmap(state: *architecture.CpuState) void {
|
||||
const base = reserve: {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const cursor = scheduler.aspaceMmapNextPtr(t.aspace) orelse return fail(state);
|
||||
const cursor = scheduler.addressSpaceMmapNextPtr(t.address_space) orelse return fail(state);
|
||||
if (cursor.* == 0) cursor.* = heap_arena_base; // seed the arena lazily
|
||||
const b = cursor.*;
|
||||
if (b + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||
@@ -1295,8 +1296,8 @@ fn systemMmap(state: *architecture.CpuState) void {
|
||||
var i: usize = 0;
|
||||
while (i < mapped) : (i += 1) {
|
||||
const va = base + i * page_size;
|
||||
if (architecture.translate(t.aspace, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.aspace, va);
|
||||
if (architecture.translate(t.address_space, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.address_space, va);
|
||||
pmm.free(physical);
|
||||
}
|
||||
}
|
||||
@@ -1305,7 +1306,7 @@ fn systemMmap(state: *architecture.CpuState) void {
|
||||
};
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||
architecture.mapUserPageInto(t.aspace, base + mapped * page_size, frame, true, false); // RW + NX
|
||||
architecture.mapUserPageInto(t.address_space, base + mapped * page_size, frame, true, false); // RW + NX
|
||||
sync.leave(flags);
|
||||
}
|
||||
architecture.setSystemCallResult(state, base); // the cursor was already advanced at reserve
|
||||
@@ -1320,14 +1321,14 @@ fn systemMunmap(state: *architecture.CpuState) void {
|
||||
const base = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or base % page_size != 0) return fail(state);
|
||||
if (t.address_space == 0 or base % page_size != 0) return fail(state);
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base + i * page_size;
|
||||
if (architecture.translate(t.aspace, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.aspace, va);
|
||||
if (architecture.translate(t.address_space, va)) |physical| {
|
||||
architecture.unmapUserPageInto(t.address_space, va);
|
||||
pmm.free(physical);
|
||||
}
|
||||
}
|
||||
@@ -1391,7 +1392,7 @@ const maximum_segments = 16;
|
||||
const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
||||
|
||||
const Segment = struct {
|
||||
vaddr: u64,
|
||||
virtual_address: u64,
|
||||
memsz: u64,
|
||||
filesz: u64,
|
||||
off: u64,
|
||||
@@ -1439,7 +1440,7 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
||||
if (w and x) return error.BadSegment; // W^X, even for init
|
||||
|
||||
const seg = Segment{
|
||||
.vaddr = phdr.p_vaddr,
|
||||
.virtual_address = phdr.p_vaddr,
|
||||
.memsz = phdr.p_memsz,
|
||||
.filesz = phdr.p_filesz,
|
||||
.off = phdr.p_offset,
|
||||
@@ -1448,9 +1449,9 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
||||
};
|
||||
// No overlap with any earlier segment (page-granular, since mapping is).
|
||||
for (segs[0..count]) |other| {
|
||||
const a_end = seg.vaddr + seg.pages() * page_size;
|
||||
const b_end = other.vaddr + other.pages() * page_size;
|
||||
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
|
||||
const a_end = seg.virtual_address + seg.pages() * page_size;
|
||||
const b_end = other.virtual_address + other.pages() * page_size;
|
||||
if (seg.virtual_address < b_end and other.virtual_address < a_end) return error.BadSegment;
|
||||
}
|
||||
total_pages += seg.pages();
|
||||
if (total_pages > maximum_pages) return error.ProgramTooBig;
|
||||
@@ -1461,17 +1462,17 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
||||
|
||||
// The entry point must land inside an executable segment.
|
||||
for (segs[0..count]) |seg| {
|
||||
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
|
||||
if (seg.executable and ehdr.e_entry >= seg.virtual_address and ehdr.e_entry < seg.virtual_address + seg.memsz)
|
||||
return .{ .count = count, .entry = ehdr.e_entry };
|
||||
}
|
||||
return error.BadEntry;
|
||||
}
|
||||
|
||||
/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed
|
||||
/// Load one page of a segment into address space `address_space`: a fresh frame, zeroed
|
||||
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
||||
/// On a later failure the whole address space is torn down, which frees every
|
||||
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||
fn loadPageInto(address_space: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0);
|
||||
@@ -1480,7 +1481,7 @@ fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) I
|
||||
const n = @min(page_size, seg.filesz - page_off);
|
||||
@memcpy(destination[0..n], image[seg.off + page_off ..][0..n]);
|
||||
}
|
||||
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
||||
architecture.mapUserPageInto(address_space, seg.virtual_address + page_off, frame, seg.writable, seg.executable);
|
||||
}
|
||||
|
||||
/// Build the System V AMD64 process-entry block at the top of a process's stack
|
||||
@@ -1567,11 +1568,11 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
||||
errdefer architecture.destroyAddressSpace(aspace);
|
||||
const address_space = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
||||
errdefer architecture.destroyAddressSpace(address_space);
|
||||
|
||||
for (segs[0..parsed.count]) |seg| {
|
||||
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
||||
for (0..seg.pages()) |i| try loadPageInto(address_space, image, seg, i);
|
||||
}
|
||||
|
||||
// The stack: `user_stack_pages` zeroed pages below stack_top_virtual, RW + NX.
|
||||
@@ -1585,10 +1586,10 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
||||
const page_virtual = stack_base_virtual + i * page_size;
|
||||
if (i == parameters.user_stack_pages - 1)
|
||||
user_sp = buildEntryStack(stack_page, page_virtual, argv);
|
||||
architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX
|
||||
architecture.mapUserPageInto(address_space, page_virtual, stack_frame, true, false); // RW + NX
|
||||
}
|
||||
|
||||
const child = scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
||||
const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
||||
return error.OutOfMemory;
|
||||
// The child holds a reference to its exit endpoint from birth to death. Taken
|
||||
// only now, after nothing can fail; the lock is still held, so the child
|
||||
|
||||
Reference in New Issue
Block a user