Compare commits
2
Commits
59104dd988
...
a5fe63c1dd
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a5fe63c1dd | ||
|
|
6b3ae0c997 |
@@ -299,6 +299,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "hpet", "system/drivers/hpet/hpet.zig");
|
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "hpet", "system/drivers/hpet/hpet.zig");
|
||||||
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "bus", "system/drivers/bus/bus.zig");
|
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "bus", "system/drivers/bus/bus.zig");
|
||||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||||
|
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||||
|
|
||||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||||
@@ -316,6 +317,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
mk_run.addFileArg(bus_exe.getEmittedBin());
|
mk_run.addFileArg(bus_exe.getEmittedBin());
|
||||||
mk_run.addArg("device-manager");
|
mk_run.addArg("device-manager");
|
||||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("args-echo");
|
||||||
|
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||||
|
|
||||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||||
|
|||||||
@@ -164,8 +164,11 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
|||||||
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
||||||
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
||||||
booted with an emulated `intel-iommu`.
|
booted with an emulated `intel-iommu`.
|
||||||
- **`system_spawn`** — a user-space supervisor starts a driver: `system_spawn(name)`
|
- **`system_spawn`** — a user-space supervisor starts a driver:
|
||||||
loads a binary bundled in the initial-ramdisk as a fresh ring-3 process. This is what
|
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||||
|
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||||
|
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||||
|
([sysv.md](sysv.md)). This is what
|
||||||
turned the device manager from "log the match" into "run the driver": the kernel now
|
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||||
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||||
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||||
|
|||||||
+5
-3
@@ -37,8 +37,9 @@ kernel ──spawns──► init (PID 1) ──spawns──► device-manag
|
|||||||
```
|
```
|
||||||
|
|
||||||
The kernel launches exactly one process — `init` — and hands it nothing but the raw
|
The kernel launches exactly one process — `init` — and hands it nothing but the raw
|
||||||
ability to start more (`system_spawn(name)`, which loads a binary bundled in the
|
ability to start more (`system_spawn(name, arguments)`, which loads a binary bundled
|
||||||
initial-ramdisk as a fresh ring-3 process). Everything else is a user-space decision:
|
in the initial-ramdisk as a fresh ring-3 process — `name` becoming its argv[0],
|
||||||
|
the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Everything else is a user-space decision:
|
||||||
|
|
||||||
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||||
supervisor**. It spawns the system services danos brings up at boot — today `vfs` and
|
supervisor**. It spawns the system services danos brings up at boot — today `vfs` and
|
||||||
@@ -52,7 +53,8 @@ initial-ramdisk as a fresh ring-3 process). Everything else is a user-space deci
|
|||||||
is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads
|
is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads
|
||||||
what each driver *binds* (a manifest under `/system/drivers`, or the driver
|
what each driver *binds* (a manifest under `/system/drivers`, or the driver
|
||||||
describing its own match).
|
describing its own match).
|
||||||
3. **Spawn** — `system_spawn(driver_name)` starts the matched driver, which then claims
|
3. **Spawn** — `system_spawn(driver_name, arguments)` starts the matched driver (the
|
||||||
|
arguments can carry *which* device it matched), which then claims
|
||||||
its device and runs the event loop below.
|
its device and runs the event loop below.
|
||||||
|
|
||||||
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
||||||
|
|||||||
+15
-4
@@ -80,11 +80,22 @@ inline). `build.zig` adds `isr.s` to the arch module.
|
|||||||
## Reporting a fault
|
## Reporting a fault
|
||||||
|
|
||||||
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
||||||
hook. The generic kernel installs a reporter (`onException` in `main.zig`) that
|
hook. The generic kernel installs a reporter (`onException` in `kernel.zig`) that
|
||||||
prints, in red, the exception name and vector, the error code, the faulting RIP
|
prints the exception name and vector, the error code, the faulting RIP
|
||||||
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
||||||
**CR2**. Then it halts. There's no fault *recovery* yet, so every exception is
|
**CR2**. What happens next depends on where the fault came from:
|
||||||
terminal; the point is that it's now **visible** instead of a silent reset.
|
|
||||||
|
- **User mode (CPL 3): kill the process, keep the machine.** The kernel is intact
|
||||||
|
(the CPU trapped onto the task's kernel stack), so the faulting process is
|
||||||
|
killed — address space, IRQ bindings, and IPC handles reclaimed; a client it
|
||||||
|
owed a reply to is failed with `-EPEER` — and the core reschedules. A crashing
|
||||||
|
driver takes itself down, never the OS. This is fault recovery step 2 of
|
||||||
|
[resilience.md](resilience.md). NMI, double fault, and machine check are
|
||||||
|
excluded: they report machine trouble regardless of what was running.
|
||||||
|
- **Kernel mode: halt this core.** The trusted base itself is broken, so there is
|
||||||
|
nothing safe to kill; the fault is still *contained* to the core (an
|
||||||
|
application-processor fault leaves the rest of the system running), and the
|
||||||
|
report makes it **visible** instead of a silent reset.
|
||||||
|
|
||||||
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
||||||
caught.
|
caught.
|
||||||
|
|||||||
+9
-3
@@ -1,6 +1,10 @@
|
|||||||
# Resilience: fault isolation and live restart
|
# Resilience: fault isolation and live restart
|
||||||
|
|
||||||
A design/research note, not built yet. This is the property danos is really chasing:
|
Steps 1–2 of the ordering below are **built**: user-mode isolation, and fault →
|
||||||
|
kill the process → keep the core (`onException` in `system/kernel/kernel.zig`; the
|
||||||
|
`fault-recovery` test proves a crashing ring-3 process dies alone while the system
|
||||||
|
keeps running). The supervisor notification and restart policy (steps 3+) are
|
||||||
|
still design. This is the property danos is really chasing:
|
||||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||||
@@ -111,9 +115,11 @@ Honest boundaries:
|
|||||||
## Suggested ordering
|
## Suggested ordering
|
||||||
|
|
||||||
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
||||||
path for everything else).
|
path for everything else). **Done.**
|
||||||
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
||||||
"confine to the process and report it."
|
"confine to the process and report it." **Done** (the kill and reclaim; the
|
||||||
|
supervisor notification waits for step 3's supervisor). A killed server's
|
||||||
|
pending client is unblocked with `-EPEER` rather than hung.
|
||||||
3. **A minimal supervisor server** that can (re)start a process.
|
3. **A minimal supervisor server** that can (re)start a process.
|
||||||
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
||||||
table.
|
table.
|
||||||
|
|||||||
@@ -65,6 +65,36 @@ function-pointer type and the kernel's `_start` both carry
|
|||||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||||
the handoff it governs.
|
the handoff it governs.
|
||||||
|
|
||||||
|
## The process-entry stack (argc/argv)
|
||||||
|
|
||||||
|
The SysV ABI also fixes what a *fresh process* finds on its stack — and danos
|
||||||
|
follows it, so its own runtime and any future C libc read arguments the same way.
|
||||||
|
At the first user instruction, `rsp` is 16-byte aligned and points at (addresses
|
||||||
|
growing upward):
|
||||||
|
|
||||||
|
```
|
||||||
|
rsp → argc u64
|
||||||
|
argv[0] … argv[argc-1] pointers into the strings area below
|
||||||
|
NULL argv terminator
|
||||||
|
NULL envp terminator (no environment yet)
|
||||||
|
{AT_PAGESZ, page size} auxiliary vector
|
||||||
|
{AT_NULL, 0} auxiliary-vector terminator
|
||||||
|
argv string bytes NUL-terminated
|
||||||
|
───────────────────────── stack top (stack_top_virtual)
|
||||||
|
```
|
||||||
|
|
||||||
|
The kernel builds this block at the top of the process's stack — 8 pages (32 KiB,
|
||||||
|
`parameters.user_stack_pages`) mapped RW+NX below a fixed top, with the page below
|
||||||
|
them left unmapped as a **guard**, so a stack overflow faults (killing only that
|
||||||
|
process) instead of silently corrupting the image
|
||||||
|
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||||
|
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||||
|
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||||
|
(`library/runtime/start.zig`) hands the block to `rt_start`, which exposes it as
|
||||||
|
`runtime.argumentCount()` / `runtime.argument(i)`. A C runtime's `crt0` would walk
|
||||||
|
the identical layout unmodified — that's the compatibility being bought. The
|
||||||
|
`args` test proves the round trip.
|
||||||
|
|
||||||
## Where else it surfaces
|
## Where else it surfaces
|
||||||
|
|
||||||
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
||||||
|
|||||||
@@ -26,5 +26,10 @@ pub const dma = @import("dma.zig");
|
|||||||
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||||
pub const panic = start.panic;
|
pub const panic = start.panic;
|
||||||
|
|
||||||
|
/// Process arguments (argc/argv, parsed from the kernel-built entry stack):
|
||||||
|
/// `argument(0)` is the path or name this binary was spawned as.
|
||||||
|
pub const argumentCount = start.argumentCount;
|
||||||
|
pub const argument = start.argument;
|
||||||
|
|
||||||
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||||
pub const allocator = heap.allocator;
|
pub const allocator = heap.allocator;
|
||||||
|
|||||||
@@ -5,25 +5,52 @@
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const system = @import("system.zig");
|
const system = @import("system.zig");
|
||||||
|
|
||||||
/// The kernel enters at `_start` with rsp 16-aligned, but a SystemV function expects
|
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||||
/// rsp ≡ 8 (mod 16) on entry (as if reached by `call`). The `call` below pushes
|
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||||
/// the 8-byte return address, satisfying the ABI before any Zig frame runs; the
|
/// auxiliary vector, then the strings (see system/kernel/process.zig,
|
||||||
/// `ud2` is a safety net if `rt_start` ever returns.
|
/// `buildEntryStack`). Capture that address in rdi — the first SysV argument —
|
||||||
|
/// before `call` disturbs the stack; the call's pushed return address also puts
|
||||||
|
/// rsp ≡ 8 (mod 16), satisfying the ABI before any Zig frame runs. The `ud2` is a
|
||||||
|
/// safety net if `rt_start` ever returns.
|
||||||
pub export fn _start() callconv(.naked) noreturn {
|
pub export fn _start() callconv(.naked) noreturn {
|
||||||
asm volatile (
|
asm volatile (
|
||||||
|
\\mov %%rsp, %%rdi
|
||||||
\\call rt_start
|
\\call rt_start
|
||||||
\\ud2
|
\\ud2
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The first Zig frame. The heap is lazy (first alloc grows it), so there is no
|
// The process-entry block, recorded by `rt_start` for the accessors below.
|
||||||
/// runtime init to order here — just hand control to the program's `main`.
|
var argument_count: usize = 0;
|
||||||
export fn rt_start() callconv(.c) noreturn {
|
var argument_vector: [*]const u64 = undefined;
|
||||||
|
|
||||||
|
/// The first Zig frame, entered with `stack` pointing at the kernel-built entry
|
||||||
|
/// block. Record argc/argv for the accessors, then hand control to the program's
|
||||||
|
/// `main`. The heap is lazy (first alloc grows it), so there is no other runtime
|
||||||
|
/// init to order here.
|
||||||
|
export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||||
|
argument_count = stack[0];
|
||||||
|
argument_vector = stack + 1;
|
||||||
const root = @import("root"); // the user binary's root source file
|
const root = @import("root"); // the user binary's root source file
|
||||||
root.main();
|
root.main();
|
||||||
system.exit(0);
|
system.exit(0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Number of process arguments (argc). At least 1: argument 0 is the path or
|
||||||
|
/// name this binary was spawned as.
|
||||||
|
pub fn argumentCount() usize {
|
||||||
|
return argument_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Process argument `index` (0 = the program's own path/name), or an empty slice
|
||||||
|
/// if out of range. The bytes live in the entry block at the top of the stack
|
||||||
|
/// page, NUL-terminated, valid for the process's lifetime.
|
||||||
|
pub fn argument(index: usize) []const u8 {
|
||||||
|
if (index >= argument_count) return "";
|
||||||
|
const string: [*:0]const u8 = @ptrFromInt(argument_vector[index]);
|
||||||
|
return std.mem.span(string);
|
||||||
|
}
|
||||||
|
|
||||||
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||||
pub const panic = std.debug.FullPanic(struct {
|
pub const panic = std.debug.FullPanic(struct {
|
||||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||||
|
|||||||
@@ -44,11 +44,31 @@ pub fn exit(code: usize) noreturn {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||||
/// process, returning true on success. This is how a supervisor (the device manager)
|
/// process, returning true on success. The child's argv[0] is `name`. This is how
|
||||||
/// launches a driver it matched — danos-native, not POSIX (a spawn/exec family comes
|
/// a supervisor (the device manager) launches a driver it matched — danos-native,
|
||||||
/// with the process work later).
|
/// not POSIX (a spawn/exec family comes with the process work later).
|
||||||
pub fn spawn(name: []const u8) bool {
|
pub fn spawn(name: []const u8) bool {
|
||||||
return sc.systemCall2(.system_spawn, @intFromPtr(name.ptr), name.len) == 0;
|
return sc.systemCall4(.system_spawn, @intFromPtr(name.ptr), name.len, 0, 0) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like `spawn`, but hands the child command-line arguments: they arrive as
|
||||||
|
/// argv[1..] on its System V entry stack (argv[0] is still `name`). Marshalled to
|
||||||
|
/// the kernel as one NUL-separated blob; the combined arguments must fit
|
||||||
|
/// `blob` (the kernel caps the blob at 256 bytes and argc at 8 anyway).
|
||||||
|
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) bool {
|
||||||
|
var blob: [256]u8 = undefined;
|
||||||
|
var len: usize = 0;
|
||||||
|
for (arguments, 0..) |argument, i| {
|
||||||
|
if (i != 0) {
|
||||||
|
if (len >= blob.len) return false;
|
||||||
|
blob[len] = 0;
|
||||||
|
len += 1;
|
||||||
|
}
|
||||||
|
if (len + argument.len > blob.len) return false;
|
||||||
|
@memcpy(blob[len..][0..argument.len], argument);
|
||||||
|
len += argument.len;
|
||||||
|
}
|
||||||
|
return sc.systemCall4(.system_spawn, @intFromPtr(name.ptr), name.len, @intFromPtr(&blob), len) == 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||||
|
|||||||
@@ -3,9 +3,11 @@
|
|||||||
//! triple-faults and silently resets the machine. With it, the CPU vectors into
|
//! triple-faults and silently resets the machine. With it, the CPU vectors into
|
||||||
//! our stubs, which capture the register state and hand it to a dispatcher.
|
//! our stubs, which capture the register state and hand it to a dispatcher.
|
||||||
//!
|
//!
|
||||||
//! Vectors split in two: 0-31 are CPU exceptions (terminal — reported and
|
//! Vectors split in two: 0-31 are CPU exceptions, handed to the `on_fault` hook
|
||||||
//! halted); 32+ are device interrupts (a registered handler runs, the APIC is
|
//! and never returned from (the kernel's handler kills a faulting user process
|
||||||
//! acknowledged, and we return to the interrupted code).
|
//! and reschedules, or halts the core for a kernel-mode fault); 32+ are device
|
||||||
|
//! interrupts (a registered handler runs, the APIC is acknowledged, and we
|
||||||
|
//! return to the interrupted code).
|
||||||
|
|
||||||
const gdt = @import("gdt.zig");
|
const gdt = @import("gdt.zig");
|
||||||
const tss = @import("tss.zig");
|
const tss = @import("tss.zig");
|
||||||
@@ -162,8 +164,9 @@ pub fn loadOnThisCpu() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
|
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
|
||||||
/// assembly stubs can `call` it by name. Exceptions are terminal; device
|
/// assembly stubs can `call` it by name. Exceptions never return here (on_fault
|
||||||
/// interrupts run their handler, get acknowledged, and return.
|
/// kills the faulting process or halts the core); device interrupts run their
|
||||||
|
/// handler, get acknowledged, and return.
|
||||||
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
|
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
|
||||||
if (state.vector < 32) {
|
if (state.vector < 32) {
|
||||||
on_fault(state); // CPU exception — never returns
|
on_fault(state); // CPU exception — never returns
|
||||||
|
|||||||
@@ -46,6 +46,7 @@ pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
|||||||
pub const ENOENT: i64 = 4; // no such registered service
|
pub const ENOENT: i64 = 4; // no such registered service
|
||||||
pub const ENOSPC: i64 = 5; // handle table or registry full
|
pub const ENOSPC: i64 = 5; // handle table or registry full
|
||||||
pub const ENOMEM: i64 = 6; // out of memory
|
pub const ENOMEM: i64 = 6; // out of memory
|
||||||
|
pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed)
|
||||||
|
|
||||||
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
|
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
|
||||||
/// message from a client — there is no reply owed. The low bits carry the source
|
/// message from a client — there is no reply owed. The low bits carry the source
|
||||||
|
|||||||
@@ -284,7 +284,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
|||||||
if (boot_information.init_len != 0) {
|
if (boot_information.init_len != 0) {
|
||||||
status("starting /system/services/init...\n");
|
status("starting /system/services/init...\n");
|
||||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||||
process.spawnProcess(image, 4) catch |err| {
|
process.spawnProcess(image, 4, &.{"/system/services/init"}) catch |err| {
|
||||||
statusPrint("/system/services/init failed to load: {s}\n", .{@errorName(err)});
|
statusPrint("/system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||||
};
|
};
|
||||||
} else {
|
} else {
|
||||||
@@ -385,13 +385,42 @@ fn kib(frames: u64) u64 {
|
|||||||
return frames * abi.page_size / (1024);
|
return frames * abi.page_size / (1024);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Report a CPU exception and halt **this core**. There's no fault recovery yet, so
|
/// Whether a ring-3 exception is attributable to the process that raised it — and
|
||||||
/// the faulting core is terminal — but the fault is *contained* to it: on an
|
/// therefore recoverable by killing that process. NMI (2), double fault (8), and
|
||||||
/// application processor only that core stops, and the rest of the system keeps
|
/// machine check (18) report machine or kernel trouble even when they arrive with a
|
||||||
/// running (full recovery — kill the task, keep the core — is the resilience track,
|
/// user CS (an NMI interrupts whatever happens to be running), so they stay terminal.
|
||||||
/// see docs/resilience.md). The report names the core so an AP fault is attributed,
|
fn recoverableFault(vector: u64) bool {
|
||||||
/// and goes to every output sink plus a POST code and a persistent breadcrumb.
|
return switch (vector) {
|
||||||
|
2, 8, 18 => false,
|
||||||
|
else => true,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Report a CPU exception. Two outcomes (docs/resilience.md):
|
||||||
|
///
|
||||||
|
/// **A fault taken in user mode kills the faulting process, not the machine.** The
|
||||||
|
/// kernel is intact — the CPU trapped onto the task's kernel stack — so the process
|
||||||
|
/// is killed, everything it held (address space, IRQ bindings, IPC handles, device
|
||||||
|
/// grants' frames) is reclaimed, and the core reschedules. A crashing driver takes
|
||||||
|
/// itself down, never the OS.
|
||||||
|
///
|
||||||
|
/// **Everything else halts this core.** A kernel-mode fault means the trusted base
|
||||||
|
/// itself is broken — there is nothing safe to kill — and NMI/#DF/#MC report machine
|
||||||
|
/// trouble regardless of CS (`recoverableFault`). Even then the fault is *contained*:
|
||||||
|
/// on an application processor only that core stops and the rest keep running. The
|
||||||
|
/// report names the core so an AP fault is attributed, and goes to every output sink
|
||||||
|
/// plus a POST code and a persistent breadcrumb. (A ring-3 fault on a *borrowed*
|
||||||
|
/// kernel thread — process.run, the user-pf isolation probe — also lands here: there
|
||||||
|
/// is no scheduled process to kill.)
|
||||||
fn onException(state: *const architecture.CpuState) noreturn {
|
fn onException(state: *const architecture.CpuState) noreturn {
|
||||||
|
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
||||||
|
statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||||
|
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||||
|
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||||
|
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||||
|
process.killCurrentProcess(); // reclaims everything, reschedules; never returns
|
||||||
|
}
|
||||||
|
|
||||||
log.checkpoint(cp_exception);
|
log.checkpoint(cp_exception);
|
||||||
const core = scheduler.currentCpuIndex();
|
const core = scheduler.currentCpuIndex();
|
||||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
||||||
|
|||||||
+181
-38
@@ -24,6 +24,7 @@ const elf = std.elf;
|
|||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const device_abi = @import("device-abi");
|
const device_abi = @import("device-abi");
|
||||||
|
const parameters = @import("parameters");
|
||||||
const architecture = @import("architecture");
|
const architecture = @import("architecture");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
@@ -40,9 +41,18 @@ const SystemCall = abi.SystemCall;
|
|||||||
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
||||||
/// the identity map (low indices) and the vmm test address (index 128), so
|
/// the identity map (low indices) and the vmm test address (index 128), so
|
||||||
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
|
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
|
||||||
/// An ELF image may occupy [code_virtual, stack_virtual); the stack page sits above.
|
/// An ELF image may occupy [code_virtual, stack_virtual); the stack sits above.
|
||||||
pub const code_virtual: u64 = 0x0000_7000_0000_0000;
|
pub const code_virtual: u64 = 0x0000_7000_0000_0000;
|
||||||
pub const stack_virtual: u64 = 0x0000_7000_0020_0000;
|
|
||||||
|
/// The stack region, above the image. The page at `stack_virtual` is **never
|
||||||
|
/// mapped** — it is the guard page: a process that overflows its stack walks into
|
||||||
|
/// it and faults (killing only that process) rather than silently corrupting the
|
||||||
|
/// top of its own image. The stack proper is `parameters.user_stack_pages` pages
|
||||||
|
/// at [stack_base_virtual, stack_top_virtual), RW + NX, with the System V entry
|
||||||
|
/// block (argc/argv) at the very top.
|
||||||
|
pub const stack_virtual: u64 = 0x0000_7000_0020_0000; // guard page (unmapped)
|
||||||
|
pub const stack_base_virtual: u64 = stack_virtual + page_size;
|
||||||
|
pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pages * page_size;
|
||||||
|
|
||||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||||
@@ -73,6 +83,19 @@ pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per proc
|
|||||||
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
||||||
const maximum_mmap_pages = 256;
|
const maximum_mmap_pages = 256;
|
||||||
|
|
||||||
|
/// Ceiling on a process's argv entries, including argv[0]. Arguments are spawn
|
||||||
|
/// parameters ("you are the driver for device 12"), not bulk data — IPC carries
|
||||||
|
/// that — so the bound is small and everything fits the single stack page.
|
||||||
|
pub const maximum_arguments = 8;
|
||||||
|
|
||||||
|
/// Ceiling on the `system_spawn` extra-arguments blob (argv[1..], NUL-separated).
|
||||||
|
pub const maximum_argument_bytes = 256;
|
||||||
|
|
||||||
|
/// Auxiliary-vector entry types (System V AMD64 process entry). Only what the
|
||||||
|
/// kernel emits today; a C runtime scans the vector until the null terminator.
|
||||||
|
const auxiliary_vector_null: u64 = 0; // AT_NULL — end of the vector
|
||||||
|
const auxiliary_vector_page_size: u64 = 6; // AT_PAGESZ
|
||||||
|
|
||||||
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
|
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
|
||||||
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
|
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
|
||||||
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
|
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
|
||||||
@@ -120,18 +143,10 @@ fn system_call(state: *architecture.CpuState) void {
|
|||||||
switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) {
|
switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) {
|
||||||
.exit => {
|
.exit => {
|
||||||
exit_code = architecture.systemCallArg(state, 0);
|
exit_code = architecture.systemCallArg(state, 0);
|
||||||
// A scheduled process drops its endpoint references, frees its address
|
// A scheduled process tears down fully (terminateCurrent); a borrowed
|
||||||
// space, and reschedules; a borrowed test thread unwinds back to the
|
// test thread unwinds back to the kernel that entered it.
|
||||||
// kernel that entered it.
|
|
||||||
if (scheduler.currentIsUserProcess()) {
|
if (scheduler.currentIsUserProcess()) {
|
||||||
// Unbind before closeHandles: dropping the last reference destroys the
|
terminateCurrent();
|
||||||
// Endpoint, and a still-bound GSI would have an ISR call
|
|
||||||
// notifyFromIsr on freed memory the next time the device fired.
|
|
||||||
// unbindAll also leaves the line masked, so a dead driver's device
|
|
||||||
// goes quiet rather than storming.
|
|
||||||
releaseIrqs(scheduler.current());
|
|
||||||
ipc.closeHandles(scheduler.current());
|
|
||||||
scheduler.exitUser();
|
|
||||||
} else architecture.userExit();
|
} else architecture.userExit();
|
||||||
},
|
},
|
||||||
.yield => {
|
.yield => {
|
||||||
@@ -400,40 +415,94 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
|||||||
architecture.setSystemCallResult(state, id);
|
architecture.setSystemCallResult(state, id);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// system_spawn(name_ptr, name_len) -> 0 on success, -1 on failure. Load the binary
|
/// system_spawn(name_ptr, name_len, arguments_ptr, arguments_len) -> 0 on success,
|
||||||
/// bundled in the initial-ramdisk under `name` as a fresh ring-3 process. This is the
|
/// -1 on failure. Load the binary bundled in the initial-ramdisk under `name` as a
|
||||||
/// mechanism a user-space supervisor (the device manager) uses to start a driver it
|
/// fresh ring-3 process. `name` becomes the child's argv[0] (and its task name, so
|
||||||
/// matched: discovery and policy stay in user space, the kernel only spawns.
|
/// a fault report can say which binary died); `arguments` is an optional
|
||||||
|
/// NUL-separated blob that becomes argv[1..] — how a supervisor parameterises what
|
||||||
|
/// it starts ("you are the driver for device 12"). 0/0 means no extra arguments.
|
||||||
|
/// This is the mechanism a user-space supervisor (the device manager) uses to start
|
||||||
|
/// a driver it matched: discovery and policy stay in user space, the kernel only
|
||||||
|
/// spawns.
|
||||||
///
|
///
|
||||||
/// Ungated for now — any process may spawn any bundled binary. A capability (only a
|
/// Ungated for now — any process may spawn any bundled binary. A capability (only a
|
||||||
/// supervisor holds the right to spawn) belongs here once the model grows one; see
|
/// supervisor holds the right to spawn) belongs here once the model grows one; see
|
||||||
/// docs/driver-model.md. The name is bounds-checked into the user half exactly like
|
/// docs/driver-model.md. Both buffers are bounds-checked into the user half exactly
|
||||||
/// `debug_write`, and an unknown name or a load failure returns -1.
|
/// like `debug_write`, and an unknown name or a load failure returns -1.
|
||||||
fn systemSpawn(state: *architecture.CpuState) void {
|
fn systemSpawn(state: *architecture.CpuState) void {
|
||||||
const ptr = architecture.systemCallArg(state, 0);
|
const ptr = architecture.systemCallArg(state, 0);
|
||||||
const len = architecture.systemCallArg(state, 1);
|
const len = architecture.systemCallArg(state, 1);
|
||||||
|
const arguments_ptr = architecture.systemCallArg(state, 2);
|
||||||
|
const arguments_len = architecture.systemCallArg(state, 3);
|
||||||
if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
||||||
|
if (arguments_len > maximum_argument_bytes) return fail(state);
|
||||||
|
if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state);
|
||||||
const image = ramdisk_image orelse return fail(state);
|
const image = ramdisk_image orelse return fail(state);
|
||||||
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
||||||
|
|
||||||
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
|
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
|
||||||
|
var argv: [maximum_arguments][]const u8 = undefined;
|
||||||
|
argv[0] = name;
|
||||||
|
var argc: usize = 1;
|
||||||
|
if (arguments_len != 0) {
|
||||||
|
const blob = @as([*]const u8, @ptrFromInt(arguments_ptr))[0..arguments_len];
|
||||||
|
var pieces = std.mem.tokenizeScalar(u8, blob, 0);
|
||||||
|
while (pieces.next()) |piece| {
|
||||||
|
if (argc == maximum_arguments) return fail(state);
|
||||||
|
argv[argc] = piece;
|
||||||
|
argc += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!std.mem.eql(u8, item.name, name)) continue;
|
if (!std.mem.eql(u8, item.name, name)) continue;
|
||||||
spawnProcess(item.blob, 4) catch return fail(state);
|
spawnProcess(item.blob, 4, argv[0..argc]) catch return fail(state);
|
||||||
architecture.setSystemCallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
fail(state); // no bundled binary by that name
|
fail(state); // no bundled binary by that name
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed
|
/// Processes killed by a CPU fault rather than a clean exit. Evidence for the
|
||||||
/// (which is what frees the endpoints an ISR would otherwise notify into).
|
/// fault-recovery test, and a health signal a supervisor can consult later.
|
||||||
fn releaseIrqs(t: *scheduler.Task) void {
|
pub var fault_kill_count: u64 = 0;
|
||||||
const flags = sync.enter();
|
|
||||||
defer sync.leave(flags);
|
/// Tear down the current user process and reschedule; never returns. Shared by the
|
||||||
irq.releaseOwner(t.id);
|
/// exit system call and the fault path (`killCurrentProcess`). The order matters:
|
||||||
|
/// - IRQ bindings are dropped before the handle table closes: dropping the last
|
||||||
|
/// endpoint reference destroys the Endpoint, and a still-bound GSI would have an
|
||||||
|
/// ISR call notifyFromIsr on freed memory the next time the device fired.
|
||||||
|
/// `releaseOwner` also leaves the line masked, so a dead driver's device goes
|
||||||
|
/// quiet rather than storming.
|
||||||
|
/// - A client this task still owes a reply to (it died between receive and reply)
|
||||||
|
/// is failed with -EPEER rather than left blocked forever — a dead server must
|
||||||
|
/// not hang its callers.
|
||||||
|
pub fn terminateCurrent() noreturn {
|
||||||
|
const t = scheduler.current();
|
||||||
|
{
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
irq.releaseOwner(t.id);
|
||||||
|
if (t.ipc_client) |client| {
|
||||||
|
t.ipc_client = null;
|
||||||
|
client.ipc_status = -ipc.EPEER;
|
||||||
|
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
||||||
|
}
|
||||||
|
ipc.closeHandles(t);
|
||||||
|
}
|
||||||
|
scheduler.exitUser();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Kill the current user process in response to a CPU fault it raised in ring 3.
|
||||||
|
/// The fault is confined to the process — the kernel trapped it on the task's own
|
||||||
|
/// kernel stack and is intact — so everything the process held is reclaimed and the
|
||||||
|
/// core reschedules. The system keeps running; only the faulting process dies
|
||||||
|
/// (docs/resilience.md: fault -> kill -> continue).
|
||||||
|
pub fn killCurrentProcess() noreturn {
|
||||||
|
fault_kill_count += 1;
|
||||||
|
terminateCurrent();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
||||||
@@ -617,15 +686,15 @@ pub fn run(blob: []const u8) RunError!void {
|
|||||||
@memset(code[blob.len..page_size], 0xCC);
|
@memset(code[blob.len..page_size], 0xCC);
|
||||||
|
|
||||||
architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X
|
architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X
|
||||||
architecture.mapUserPage(stack_virtual, stack_frame, true, false); // RW + NX
|
architecture.mapUserPage(stack_base_virtual, stack_frame, true, false); // RW + NX (one page; the probe barely stacks)
|
||||||
resetRecords();
|
resetRecords();
|
||||||
|
|
||||||
architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_virtual + page_size);
|
architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_base_virtual + page_size);
|
||||||
|
|
||||||
// Back via the exit system_call; the interrupt gate left IF clear.
|
// Back via the exit system_call; the interrupt gate left IF clear.
|
||||||
architecture.enableInterrupts();
|
architecture.enableInterrupts();
|
||||||
architecture.unmapPage(code_virtual);
|
architecture.unmapPage(code_virtual);
|
||||||
architecture.unmapPage(stack_virtual);
|
architecture.unmapPage(stack_base_virtual);
|
||||||
pmm.free(code_frame);
|
pmm.free(code_frame);
|
||||||
pmm.free(stack_frame);
|
pmm.free(stack_frame);
|
||||||
}
|
}
|
||||||
@@ -636,6 +705,7 @@ pub const InitError = error{
|
|||||||
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
|
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
|
||||||
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
|
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
|
||||||
BadEntry, // e_entry not inside an executable segment
|
BadEntry, // e_entry not inside an executable segment
|
||||||
|
BadArguments, // no argv[0], too many entries, or too many bytes for the entry stack
|
||||||
ProgramTooBig, // more pages than the loader's budget
|
ProgramTooBig, // more pages than the loader's budget
|
||||||
OutOfMemory,
|
OutOfMemory,
|
||||||
};
|
};
|
||||||
@@ -736,13 +806,74 @@ fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) I
|
|||||||
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Build the System V AMD64 process-entry block at the top of a process's stack
|
||||||
|
/// page and return the initial user stack pointer. At the first user instruction,
|
||||||
|
/// rsp is 16-byte aligned and points at (addresses growing upward):
|
||||||
|
///
|
||||||
|
/// argc, argv[0..argc-1], NULL, NULL (empty envp), auxiliary vector, strings
|
||||||
|
///
|
||||||
|
/// — the layout every C runtime's startup code walks, so danos's own runtime and a
|
||||||
|
/// future libc port read arguments identically (docs/sysv.md). `page` is the kernel
|
||||||
|
/// (physmap) view of the stack's **top** frame and `page_user_base` that frame's
|
||||||
|
/// user address (stack_top_virtual - page_size); the pointers written into it are
|
||||||
|
/// user addresses inside that page. The caller has validated the sizes
|
||||||
|
/// (`entryStackBytes`), so this cannot overrun.
|
||||||
|
fn buildEntryStack(page: [*]u8, page_user_base: u64, argv: []const []const u8) u64 {
|
||||||
|
// The strings live at the very top of the page, packed from the end downward.
|
||||||
|
var string_offset: usize = page_size;
|
||||||
|
var pointers: [maximum_arguments]u64 = undefined;
|
||||||
|
var i: usize = argv.len;
|
||||||
|
while (i > 0) {
|
||||||
|
i -= 1;
|
||||||
|
string_offset -= argv[i].len + 1;
|
||||||
|
@memcpy(page[string_offset..][0..argv[i].len], argv[i]);
|
||||||
|
page[string_offset + argv[i].len] = 0; // NUL-terminated, as C expects
|
||||||
|
pointers[i] = page_user_base + string_offset;
|
||||||
|
}
|
||||||
|
|
||||||
|
// The vector sits below the strings: argc, the argv pointers, the argv
|
||||||
|
// terminator, an empty envp (terminator only), then the auxiliary vector.
|
||||||
|
const word_count = 1 + argv.len + 1 + 1 + 4;
|
||||||
|
const vector_offset = (string_offset - word_count * 8) & ~@as(usize, 15); // entry rsp % 16 == 0
|
||||||
|
const words: [*]u64 = @ptrCast(@alignCast(page + vector_offset));
|
||||||
|
var w: usize = 0;
|
||||||
|
words[w] = argv.len; // argc
|
||||||
|
w += 1;
|
||||||
|
for (pointers[0..argv.len]) |pointer| {
|
||||||
|
words[w] = pointer;
|
||||||
|
w += 1;
|
||||||
|
}
|
||||||
|
words[w] = 0; // argv terminator
|
||||||
|
words[w + 1] = 0; // envp: no environment yet, just the terminator
|
||||||
|
words[w + 2] = auxiliary_vector_page_size;
|
||||||
|
words[w + 3] = page_size;
|
||||||
|
words[w + 4] = auxiliary_vector_null; // end of the auxiliary vector
|
||||||
|
words[w + 5] = 0;
|
||||||
|
return page_user_base + vector_offset;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bytes the entry block for `argv` occupies at the top of the stack page:
|
||||||
|
/// strings (each NUL-terminated), vector words, and the alignment slack.
|
||||||
|
fn entryStackBytes(argv: []const []const u8) usize {
|
||||||
|
var string_bytes: usize = 0;
|
||||||
|
for (argv) |argument| string_bytes += argument.len + 1;
|
||||||
|
return string_bytes + (1 + argv.len + 1 + 1 + 4) * 8 + 16;
|
||||||
|
}
|
||||||
|
|
||||||
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
||||||
/// ring-3 process at `priority`. Returns immediately — the process runs
|
/// ring-3 process at `priority`, entered with `argv` on its stack per the System V
|
||||||
/// preemptively on its own page tables alongside everything else, and its exit
|
/// convention (`buildEntryStack`). `argv[0]` is required — it names the process:
|
||||||
/// is handled by the system_call layer. The whole build (address space + ELF load +
|
/// the path or initial-ramdisk name it was spawned as. It is also recorded on the
|
||||||
/// task) runs under the kernel lock so it appears atomically and can't race
|
/// task, so a fault report can say *which* binary died, not just its id.
|
||||||
/// pmm/heap on another core.
|
/// Returns immediately — the process runs preemptively on its own page tables
|
||||||
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
|
/// alongside everything else, and its exit is handled by the system_call layer.
|
||||||
|
/// The whole build (address space + ELF load + task) runs under the kernel lock so
|
||||||
|
/// it appears atomically and can't race pmm/heap on another core.
|
||||||
|
pub fn spawnProcess(image: []const u8, priority: u3, argv: []const []const u8) InitError!void {
|
||||||
|
if (argv.len == 0 or argv.len > maximum_arguments) return error.BadArguments;
|
||||||
|
// The entry block must leave most of the page as actual stack.
|
||||||
|
if (entryStackBytes(argv) > page_size / 2) return error.BadArguments;
|
||||||
|
|
||||||
var segs: [maximum_segments]Segment = undefined;
|
var segs: [maximum_segments]Segment = undefined;
|
||||||
const parsed = try parseSegments(image, &segs);
|
const parsed = try parseSegments(image, &segs);
|
||||||
|
|
||||||
@@ -755,10 +886,22 @@ pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
|
|||||||
for (segs[0..parsed.count]) |seg| {
|
for (segs[0..parsed.count]) |seg| {
|
||||||
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
||||||
}
|
}
|
||||||
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
|
||||||
architecture.mapUserPageInto(aspace, stack_virtual, stack_frame, true, false); // RW + NX
|
|
||||||
|
|
||||||
if (!scheduler.spawnUserLocked(aspace, parsed.entry, stack_virtual + page_size, priority))
|
// The stack: `user_stack_pages` zeroed pages below stack_top_virtual, RW + NX.
|
||||||
|
// The page below them (`stack_virtual`) stays unmapped as the overflow guard.
|
||||||
|
// The entry block goes at the top of the highest page.
|
||||||
|
var user_sp: u64 = 0;
|
||||||
|
for (0..parameters.user_stack_pages) |i| {
|
||||||
|
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
|
const stack_page: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(stack_frame));
|
||||||
|
@memset(stack_page[0..page_size], 0); // no stale frame contents leak into user space
|
||||||
|
const page_virtual = stack_base_virtual + i * page_size;
|
||||||
|
if (i == parameters.user_stack_pages - 1)
|
||||||
|
user_sp = buildEntryStack(stack_page, page_virtual, argv);
|
||||||
|
architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, priority, argv[0]))
|
||||||
return error.OutOfMemory;
|
return error.OutOfMemory;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -70,8 +70,24 @@ pub const Task = struct {
|
|||||||
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
||||||
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
||||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||||
|
// The process's name — argv[0] as it was spawned (a boot-volume path for init,
|
||||||
|
// an initial-ramdisk name for everything else); empty for kernel tasks. Fixed
|
||||||
|
// storage, so the fault path can name the dead without touching the heap.
|
||||||
|
// Zero-initialised (not `undefined`): an undefined default is materialised as
|
||||||
|
// a 0xAA fill, which would move the whole static task pool out of .bss.
|
||||||
|
name_buffer: [maximum_task_name]u8 = .{0} ** maximum_task_name,
|
||||||
|
name_length: u8 = 0,
|
||||||
|
|
||||||
|
/// The task's name (argv[0] at spawn), or empty for a kernel task.
|
||||||
|
pub fn name(self: *const Task) []const u8 {
|
||||||
|
return self.name_buffer[0..self.name_length];
|
||||||
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// Capacity of `Task.name_buffer` — matches the longest name `system_spawn`
|
||||||
|
/// accepts, so a spawned name is never truncated.
|
||||||
|
pub const maximum_task_name = 64;
|
||||||
|
|
||||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||||
pub const ipc_maximum_handles = 16;
|
pub const ipc_maximum_handles = 16;
|
||||||
@@ -258,12 +274,13 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
||||||
/// in user mode at `entry` on `user_sp`. It gets a fresh kernel stack for
|
/// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]).
|
||||||
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
|
/// It gets a fresh kernel stack for syscalls/interrupts, and its first switch-in
|
||||||
|
/// lands in `user_task_trampoline`.
|
||||||
/// Returns false (creating nothing) if the table is full or out of memory.
|
/// Returns false (creating nothing) if the table is full or out of memory.
|
||||||
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
||||||
/// across the whole spawn, so the address space and the task appear atomically).
|
/// across the whole spawn, so the address space and the task appear atomically).
|
||||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority) bool {
|
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority, task_name: []const u8) bool {
|
||||||
const t = freeSlot() orelse return false;
|
const t = freeSlot() orelse return false;
|
||||||
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
||||||
t.* = .{
|
t.* = .{
|
||||||
@@ -275,6 +292,9 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority
|
|||||||
.user_ip = entry,
|
.user_ip = entry,
|
||||||
.user_sp = user_sp,
|
.user_sp = user_sp,
|
||||||
};
|
};
|
||||||
|
const name_length = @min(task_name.len, maximum_task_name);
|
||||||
|
@memcpy(t.name_buffer[0..name_length], task_name[0..name_length]);
|
||||||
|
t.name_length = @intCast(name_length);
|
||||||
next_id += 1;
|
next_id += 1;
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||||
t.kstack_top = top;
|
t.kstack_top = top;
|
||||||
|
|||||||
+128
-5
@@ -118,6 +118,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
userMemTest();
|
userMemTest();
|
||||||
} else if (eql(case, "user-pf")) {
|
} else if (eql(case, "user-pf")) {
|
||||||
userPfTest();
|
userPfTest();
|
||||||
|
} else if (eql(case, "fault-recovery")) {
|
||||||
|
faultRecoveryTest(boot_information);
|
||||||
|
} else if (eql(case, "args")) {
|
||||||
|
argsTest(boot_information);
|
||||||
} else if (eql(case, "init")) {
|
} else if (eql(case, "init")) {
|
||||||
initTest(boot_information);
|
initTest(boot_information);
|
||||||
} else if (eql(case, "process")) {
|
} else if (eql(case, "process")) {
|
||||||
@@ -1160,8 +1164,8 @@ fn processTest(boot_information: *const BootInformation) void {
|
|||||||
scheduler.spawn(procWorker, 4); // kernel task at the processes' priority
|
scheduler.spawn(procWorker, 4); // kernel task at the processes' priority
|
||||||
|
|
||||||
var spawned: u32 = 0;
|
var spawned: u32 = 0;
|
||||||
if (process.spawnProcess(image, 4)) spawned += 1 else |_| {}
|
if (process.spawnProcess(image, 4, &.{"/system/services/init"})) spawned += 1 else |_| {}
|
||||||
if (process.spawnProcess(image, 4)) spawned += 1 else |_| {}
|
if (process.spawnProcess(image, 4, &.{"/system/services/init"})) spawned += 1 else |_| {}
|
||||||
|
|
||||||
// Wait (real time) for several heartbeats across the two processes. Each
|
// Wait (real time) for several heartbeats across the two processes. Each
|
||||||
// process sleeps ~1 s between beats, so a few seconds yields several.
|
// process sleeps ~1 s between beats, so a few seconds yields several.
|
||||||
@@ -1190,6 +1194,87 @@ fn userPfTest() void {
|
|||||||
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
|
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Spawn a real scheduled ring-3 process whose body is the user-pf blob (a read of
|
||||||
|
/// a kernel-only page, then a spin), so its first instruction raises #PF. Built by
|
||||||
|
/// hand — address space, code page RO+X, stack page RW+NX — because the blob is a
|
||||||
|
/// raw code fragment, not an ELF `spawnProcess` could load. Returns false if any
|
||||||
|
/// allocation fails.
|
||||||
|
fn spawnFaultingProcess() bool {
|
||||||
|
const blob = process.pfBlob();
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
|
||||||
|
const aspace = architecture.createAddressSpace() orelse return false;
|
||||||
|
const code_frame = pmm.alloc() orelse {
|
||||||
|
architecture.destroyAddressSpace(aspace);
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
|
||||||
|
// stray jump traps instead of sliding.
|
||||||
|
const code: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(code_frame));
|
||||||
|
@memset(code[0..abi.page_size], 0xCC);
|
||||||
|
@memcpy(code[0..blob.len], blob);
|
||||||
|
architecture.mapUserPageInto(aspace, process.code_virtual, code_frame, false, true); // RO + X
|
||||||
|
|
||||||
|
const stack_frame = pmm.alloc() orelse {
|
||||||
|
architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX
|
||||||
|
|
||||||
|
if (!scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe")) {
|
||||||
|
architecture.destroyAddressSpace(aspace);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Fault recovery (docs/resilience.md step 2): a scheduled ring-3 process that
|
||||||
|
/// faults must be killed — counted, resources reclaimed — while the rest of the
|
||||||
|
/// system keeps running. init heartbeats before and after the kill are the proof
|
||||||
|
/// the OS survived; the old behaviour (halt the core) would freeze the beat and
|
||||||
|
/// time the harness out.
|
||||||
|
fn faultRecoveryTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: fault-recovery\n", .{});
|
||||||
|
check("bootloader handed over /system/services/init", boot_information.init_len != 0);
|
||||||
|
if (boot_information.init_len == 0) {
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||||
|
|
||||||
|
process.write_count = 0;
|
||||||
|
process.fault_kill_count = 0;
|
||||||
|
const spawned = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |_| false;
|
||||||
|
check("init spawned as the surviving process", spawned);
|
||||||
|
|
||||||
|
// A first heartbeat proves init runs before the fault.
|
||||||
|
scheduler.setPriority(1); // drop below the processes so they get the core
|
||||||
|
var deadline = architecture.millis() + 8000;
|
||||||
|
while (process.write_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("init heartbeat before the fault", process.write_count >= 1);
|
||||||
|
|
||||||
|
check("faulting process spawned", spawnFaultingProcess());
|
||||||
|
|
||||||
|
// The kill: the faulting process #PFs on its first instruction and the kernel
|
||||||
|
// reaps it instead of halting.
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
deadline = architecture.millis() + 5000;
|
||||||
|
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("faulting process was killed (not the machine)", process.fault_kill_count == 1);
|
||||||
|
|
||||||
|
// Life after the kill: init must keep beating on the same core.
|
||||||
|
const beats_at_kill = process.write_count;
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
deadline = architecture.millis() + 8000;
|
||||||
|
while (process.write_count < beats_at_kill + 2 and architecture.millis() < deadline) scheduler.yield();
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
check("init kept heartbeating after the kill", process.write_count >= beats_at_kill + 2);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
|
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
|
||||||
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
|
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
|
||||||
/// — the same call the normal boot path makes — then confirm it beats. init
|
/// — the same call the normal boot path makes — then confirm it beats. init
|
||||||
@@ -1204,7 +1289,7 @@ fn initTest(boot_information: *const BootInformation) void {
|
|||||||
}
|
}
|
||||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||||
process.write_count = 0;
|
process.write_count = 0;
|
||||||
const spawned = if (process.spawnProcess(image, 4)) true else |err| blk: {
|
const spawned = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |err| blk: {
|
||||||
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
|
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
|
||||||
break :blk false;
|
break :blk false;
|
||||||
};
|
};
|
||||||
@@ -1251,7 +1336,7 @@ fn initialRamdiskTest(boot_information: *const BootInformation) void {
|
|||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (process.spawnProcess(item.blob, 4)) spawned += 1 else |err| {
|
if (process.spawnProcess(item.blob, 4, &.{item.name})) spawned += 1 else |err| {
|
||||||
log("DANOS-INITRD-ERR: {s}: {s}\n", .{ item.name, @errorName(err) });
|
log("DANOS-INITRD-ERR: {s}: {s}\n", .{ item.name, @errorName(err) });
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1311,6 +1396,44 @@ fn vfsTest(boot_information: *const BootInformation) void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||||
|
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||||
|
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||||
|
/// blob. Instance 2 parses the kernel-built System V entry stack via the runtime
|
||||||
|
/// and echoes its whole argv in one write, which must arrive exactly as sent.
|
||||||
|
fn argsTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: args\n", .{});
|
||||||
|
check("bootloader handed over an initial_ramdisk", boot_information.initial_ramdisk_len != 0);
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
process.setInitialRamdisk(image); // args-echo respawns itself through system_spawn
|
||||||
|
|
||||||
|
process.write_count = 0;
|
||||||
|
process.write_from_user = false;
|
||||||
|
check("args-echo spawned from the initial_ramdisk", spawnNamed(rd, "args-echo"));
|
||||||
|
|
||||||
|
// Wait for the *second* instance's echo (the first writes nothing).
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 8000;
|
||||||
|
while (process.write_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
const expected = "args: args-echo alpha beta-42\n";
|
||||||
|
const echoed = process.write_len == expected.len and eql(process.write_buffer[0..process.write_len], expected);
|
||||||
|
if (!echoed and process.write_len > 0) log("DANOS-ARGS: got \"{s}\"\n", .{process.write_buffer[0..process.write_len]});
|
||||||
|
check("argv arrived intact (argv[0] = name, argv[1..] = spawn arguments)", echoed);
|
||||||
|
check("echo came from user mode (CPL 3)", process.write_from_user);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
/// Spawn the initial_ramdisk binary named `name` as a ring-3 process. Returns false if it
|
/// Spawn the initial_ramdisk binary named `name` as a ring-3 process. Returns false if it
|
||||||
/// isn't in the image or fails to load.
|
/// isn't in the image or fails to load.
|
||||||
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
||||||
@@ -1318,7 +1441,7 @@ fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
|||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (eql(item.name, name)) {
|
if (eql(item.name, name)) {
|
||||||
return if (process.spawnProcess(item.blob, 4)) true else |_| false;
|
return if (process.spawnProcess(item.blob, 4, &.{item.name})) true else |_| false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
|
|||||||
@@ -22,6 +22,12 @@ pub const maximum_tasks = 16;
|
|||||||
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
/// Each task's kernel stack (also each AP's bring-up stack), in bytes.
|
||||||
pub const kernel_stack_size = 16 * 1024;
|
pub const kernel_stack_size = 16 * 1024;
|
||||||
|
|
||||||
|
/// Each user process's stack, in pages (32 KiB). Mapped just below a fixed top;
|
||||||
|
/// the System V entry block (argc/argv) occupies the top of the highest page, and
|
||||||
|
/// the page below the mapping is left unmapped as a guard, so an overflow faults
|
||||||
|
/// (killing only that process) instead of silently corrupting the image.
|
||||||
|
pub const user_stack_pages = 8;
|
||||||
|
|
||||||
/// Each core's IST (double-fault) stack, in bytes. The BSP's is static; an AP's is
|
/// Each core's IST (double-fault) stack, in bytes. The BSP's is static; an AP's is
|
||||||
/// heap-allocated at bring-up.
|
/// heap-allocated at bring-up.
|
||||||
pub const ist_stack_size = 16 * 1024;
|
pub const ist_stack_size = 16 * 1024;
|
||||||
|
|||||||
@@ -0,0 +1,55 @@
|
|||||||
|
//! args-echo — a test fixture for process arguments (bundled in the
|
||||||
|
//! initial-ramdisk, spawned only by the `args` test case). Run with no arguments,
|
||||||
|
//! it respawns itself *with* some via `spawnWithArguments` — exercising the
|
||||||
|
//! system_spawn argument blob. Run with arguments, it burns more stack than one
|
||||||
|
//! page could hold (proving the multi-page stack: on a single-page stack the
|
||||||
|
//! recursion would hit the guard and the process would be killed before echoing),
|
||||||
|
//! then echoes its whole argv in one `debug_write` the kernel test asserts on —
|
||||||
|
//! proving the kernel-built System V entry stack (argc, argv pointers,
|
||||||
|
//! NUL-terminated strings) and the runtime's parsing of it, end to end.
|
||||||
|
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
|
||||||
|
/// Recurse with a real frame each level: `depth` levels of ~0.5 KiB, touched
|
||||||
|
/// through a volatile pointer so no optimiser can flatten the frames away.
|
||||||
|
fn burnStack(depth: usize) u8 {
|
||||||
|
var frame: [512]u8 = undefined;
|
||||||
|
const touch: *volatile [512]u8 = &frame;
|
||||||
|
touch[0] = @truncate(depth);
|
||||||
|
touch[511] = touch[0];
|
||||||
|
if (depth == 0) return touch[511];
|
||||||
|
return touch[0] +% burnStack(depth - 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
if (runtime.argumentCount() <= 1) {
|
||||||
|
// First instance: spawn the second with real arguments, then exit.
|
||||||
|
_ = runtime.system.spawnWithArguments("args-echo", &.{ "alpha", "beta-42" });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ~16 x 0.5 KiB frames: comfortably past one page, well inside the 32 KiB stack.
|
||||||
|
_ = burnStack(16);
|
||||||
|
|
||||||
|
// Second instance: echo "args: <argv0> <argv1> ..." for the test to match.
|
||||||
|
var buffer: [128]u8 = undefined;
|
||||||
|
const prefix = "args:";
|
||||||
|
@memcpy(buffer[0..prefix.len], prefix);
|
||||||
|
var len: usize = prefix.len;
|
||||||
|
for (0..runtime.argumentCount()) |i| {
|
||||||
|
const argument = runtime.argument(i);
|
||||||
|
if (len + 1 + argument.len + 1 > buffer.len) break;
|
||||||
|
buffer[len] = ' ';
|
||||||
|
len += 1;
|
||||||
|
@memcpy(buffer[len..][0..argument.len], argument);
|
||||||
|
len += argument.len;
|
||||||
|
}
|
||||||
|
buffer[len] = '\n';
|
||||||
|
len += 1;
|
||||||
|
_ = runtime.system.write(buffer[0..len]);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const panic = runtime.panic;
|
||||||
|
comptime {
|
||||||
|
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||||
|
}
|
||||||
@@ -199,6 +199,17 @@ CASES = [
|
|||||||
{"name": "user-pf",
|
{"name": "user-pf",
|
||||||
"expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*IP\s*: 0x00007000000000",
|
"expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*IP\s*: 0x00007000000000",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# Fault recovery: a scheduled ring-3 process that faults is killed — resources
|
||||||
|
# reclaimed, core kept — while init's heartbeat proves the OS survived.
|
||||||
|
{"name": "fault-recovery",
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned
|
||||||
|
# name, argv[1..] = the system_spawn argument blob) and echoes back intact.
|
||||||
|
{"name": "args",
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# The real user binary: the bootloader ships /system/services/init off the ESP, the
|
# The real user binary: the bootloader ships /system/services/init off the ESP, the
|
||||||
# kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.
|
# kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.
|
||||||
{"name": "init",
|
{"name": "init",
|
||||||
|
|||||||
Reference in New Issue
Block a user