Compare commits
3
Commits
e4da4e0610
...
f3bc23cb81
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f3bc23cb81 | ||
|
|
8d4a7cf240 | ||
|
|
c4f16a5448 |
@@ -115,7 +115,8 @@ fn driverArtifact(comptime package: []const u8, comptime artifact: []const u8) S
|
|||||||
|
|
||||||
/// The production ship table — what a plain `zig build` image contains,
|
/// The production ship table — what a plain `zig build` image contains,
|
||||||
/// beyond the specials the build fn adds around it (init, discovery, the
|
/// beyond the specials the build fn adds around it (init, discovery, the
|
||||||
/// /etc data files; the /test fixtures join only under -Dtest-case).
|
/// /system/configuration data files; the /test fixtures join only under
|
||||||
|
/// -Dtest-case).
|
||||||
/// Selecting what goes into a build = selecting rows: a package in no row is
|
/// Selecting what goes into a build = selecting rows: a package in no row is
|
||||||
/// not just unshipped, its build file is never even loaded
|
/// not just unshipped, its build file is never even loaded
|
||||||
/// (docs/build-packages-plan.md).
|
/// (docs/build-packages-plan.md).
|
||||||
@@ -265,9 +266,9 @@ pub fn build(b: *std.Build) void {
|
|||||||
// (receives the root's -Dserial as a dependency option — its liveness
|
// (receives the root's -Dserial as a dependency option — its liveness
|
||||||
// heartbeat is a serial/test-build diagnostic the QEMU harness asserts
|
// heartbeat is a serial/test-build diagnostic the QEMU harness asserts
|
||||||
// on; a flashable image leaves it out), discovery (the -Ddiscovery pick),
|
// on; a flashable image leaves it out), discovery (the -Ddiscovery pick),
|
||||||
// and the /etc data files. Each binary builds itself against the domain
|
// and the /system/configuration data files. Each binary builds itself
|
||||||
// packages via build-support's shared recipe; the root just takes
|
// against the domain packages via build-support's shared recipe; the root
|
||||||
// artifacts (docs/build-packages-plan.md).
|
// just takes artifacts (docs/build-packages-plan.md).
|
||||||
var bundled_list: std.ArrayListUnmanaged(images.BundledBinary) = .empty;
|
var bundled_list: std.ArrayListUnmanaged(images.BundledBinary) = .empty;
|
||||||
bundled_list.append(b.allocator, .{
|
bundled_list.append(b.allocator, .{
|
||||||
.path = "system/services/init",
|
.path = "system/services/init",
|
||||||
@@ -300,15 +301,16 @@ pub fn build(b: *std.Build) void {
|
|||||||
.binary = b.dependency(row.package, .{}).artifact(row.artifact).getEmittedBin(),
|
.binary = b.dependency(row.package, .{}).artifact(row.artifact).getEmittedBin(),
|
||||||
}) catch @panic("OOM");
|
}) catch @panic("OOM");
|
||||||
}
|
}
|
||||||
// Data files, not binaries: packing them under /etc makes the kernel
|
// Data files, not binaries: packing them under /system/configuration rides
|
||||||
// auto-mount /etc as a read-only initrd tree (system/kernel/vfs.zig
|
// the kernel's read-only initrd mount of /system (system/kernel/vfs.zig
|
||||||
// setInitialRamdisk) — the device manager reads its registry and init its
|
// setInitialRamdisk) — the device manager reads its registry and init its
|
||||||
// service list with no filesystem service running. -Ddiagnose selects the
|
// service list with no filesystem service running. -Ddiagnose selects the
|
||||||
// init.csv variant that omits the display stack (so the kernel's boot
|
// init.csv variant that omits the display stack (so the kernel's boot
|
||||||
// transcript stays on screen); both bundle at the same /etc/init.csv path.
|
// transcript stays on screen); both bundle at the same
|
||||||
const init_csv_source = if (diagnose) "etc/init-diagnose.csv" else "etc/init.csv";
|
// /system/configuration/init.csv path.
|
||||||
bundled_list.append(b.allocator, .{ .path = "etc/devices.csv", .binary = b.path("etc/devices.csv") }) catch @panic("OOM");
|
const init_csv_source = if (diagnose) "system/configuration/init-diagnose.csv" else "system/configuration/init.csv";
|
||||||
bundled_list.append(b.allocator, .{ .path = "etc/init.csv", .binary = b.path(init_csv_source) }) catch @panic("OOM");
|
bundled_list.append(b.allocator, .{ .path = "system/configuration/devices.csv", .binary = b.path("system/configuration/devices.csv") }) catch @panic("OOM");
|
||||||
|
bundled_list.append(b.allocator, .{ .path = "system/configuration/init.csv", .binary = b.path(init_csv_source) }) catch @panic("OOM");
|
||||||
// A no-option build assumes neither -Dtest-case nor -Ddiagnose: it ships the
|
// A no-option build assumes neither -Dtest-case nor -Ddiagnose: it ships the
|
||||||
// production set only. The userspace test fixtures under /test join in only
|
// production set only. The userspace test fixtures under /test join in only
|
||||||
// for a test build — which the QEMU harness signals by passing
|
// for a test build — which the QEMU harness signals by passing
|
||||||
@@ -330,6 +332,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
"args-echo",
|
"args-echo",
|
||||||
"process-test",
|
"process-test",
|
||||||
"thread-test", // the multi-threaded fixture (its package sets .threaded)
|
"thread-test", // the multi-threaded fixture (its package sets .threaded)
|
||||||
|
"user-memory-test", // aims deliberately bad user pointers at the checked copy layer
|
||||||
}) |fixture| {
|
}) |fixture| {
|
||||||
const package = b.lazyDependency(fixture, .{}) orelse
|
const package = b.lazyDependency(fixture, .{}) orelse
|
||||||
@panic("a test fixture package is missing under test/system/services");
|
@panic("a test fixture package is missing under test/system/services");
|
||||||
|
|||||||
@@ -74,6 +74,7 @@
|
|||||||
.@"args-echo" = .{ .path = "test/system/services/args-echo", .lazy = true },
|
.@"args-echo" = .{ .path = "test/system/services/args-echo", .lazy = true },
|
||||||
.@"process-test" = .{ .path = "test/system/services/process-test", .lazy = true },
|
.@"process-test" = .{ .path = "test/system/services/process-test", .lazy = true },
|
||||||
.@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true },
|
.@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true },
|
||||||
|
.@"user-memory-test" = .{ .path = "test/system/services/user-memory-test", .lazy = true },
|
||||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||||
//.example = .{
|
//.example = .{
|
||||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||||
|
|||||||
+1
-1
@@ -83,7 +83,7 @@ pub fn addImageSteps(b: *std.Build, options: Options) std.Build.LazyPath {
|
|||||||
// holding the EFI stub, the kernel, and the whole /system tree of user
|
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||||
// danos fat driver mounts the same image at /mnt/usb.
|
// danos fat driver mounts the same image at /volumes/usb.
|
||||||
const fat_image = addBootImage(b, options.kernel.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
const fat_image = addBootImage(b, options.kernel.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||||
|
|||||||
+2
-2
@@ -42,8 +42,8 @@ pub fn addRunSteps(b: *std.Build, fat_image_serial: std.Build.LazyPath) void {
|
|||||||
const vars_out = vars_copy.addOutputFileArg("OVMF_VARS.4m.fd");
|
const vars_out = vars_copy.addOutputFileArg("OVMF_VARS.4m.fd");
|
||||||
|
|
||||||
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||||
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
// scratch area — a dev/host artifact, kept out of the boot volume we mount.
|
||||||
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
// (/system/logs on the volume belongs to the guest's own logger.) One
|
||||||
// timestamped file per run.
|
// timestamped file per run.
|
||||||
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
|
|||||||
@@ -32,8 +32,8 @@ the logging/USB-lifecycle track).
|
|||||||
## Status
|
## Status
|
||||||
|
|
||||||
- [x] **Phase 0** — baseline: suite green on `main` (106/106, 2026-07-31; `zig build` + `zig build test` clean at 9a32380), plan committed
|
- [x] **Phase 0** — baseline: suite green on `main` (106/106, 2026-07-31; `zig build` + `zig build test` clean at 9a32380), plan committed
|
||||||
- [ ] **PM** — path-migration flag-day (`/etc`→`/system/configuration`, `/var/log`→`/system/logs`, `/mnt/usb`→`/volumes/usb`)
|
- [x] **PM** — path-migration flag-day (`/etc`→`/system/configuration`, `/var/log`→`/system/logs`, `/mnt/usb`→`/volumes/usb`; vfs carve-out for the two writable `/system` subtrees, FAT's `/var` mount split in two; suite 106/106)
|
||||||
- [ ] **H1** — the `user-memory` module; nine stragglers converted; leaf U/S+W checks
|
- [x] **H1** — the `user-memory` module; nine stragglers converted; leaf U/S+W checks (plus physmap-coverage confirmation, so an `mmio_map`'d buffer cannot fault ring 0 — this also closes the same hazard on the IPC path; `fs_resolve`'s out-capacity bound made overflow-safe; suite 107/107)
|
||||||
- [ ] **merge** group 1 → main, push
|
- [ ] **merge** group 1 → main, push
|
||||||
- [ ] **P1** — envelope module + `Define`; vfs `NodeKind.protocol` + open-reply-capability; client `Channel`
|
- [ ] **P1** — envelope module + `Define`; vfs `NodeKind.protocol` + open-reply-capability; client `Channel`
|
||||||
- [ ] **P2** — registry in init; `/protocol` reserved; ServiceId flag-day (11 binds, 17 lookups)
|
- [ ] **P2** — registry in init; `/protocol` reserved; ServiceId flag-day (11 binds, 17 lookups)
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
//! The "csv" library domain: shared CSV helpers (comment stripping, field
|
//! The "csv" library domain: shared CSV helpers (comment stripping, field
|
||||||
//! iteration) for the /etc/*.csv config files — the device registry and the
|
//! iteration) for the /system/configuration/*.csv config files — the device registry and the
|
||||||
//! init service list both parse them.
|
//! init service list both parse them.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
|||||||
+2
-2
@@ -1,5 +1,5 @@
|
|||||||
//! Minimal CSV helpers shared by the `/etc/*.csv` config files — the device
|
//! Minimal CSV helpers shared by the `/system/configuration/*.csv` config files — the device
|
||||||
//! registry (`/etc/devices.csv`) and the init service list (`/etc/init.csv`).
|
//! registry (`/system/configuration/devices.csv`) and the init service list (`/system/configuration/init.csv`).
|
||||||
//! Freestanding, no allocator: returned fields are slices into the source line,
|
//! Freestanding, no allocator: returned fields are slices into the source line,
|
||||||
//! so the source must outlive them. `#` starts a comment (whole-line or trailing);
|
//! so the source must outlive them. `#` starts a comment (whole-line or trailing);
|
||||||
//! whitespace around a field is trimmed, so columns may be padded for alignment.
|
//! whitespace around a field is trimmed, so columns may be padded for alignment.
|
||||||
|
|||||||
@@ -97,7 +97,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
|
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
// The device registry: parse /etc/devices.csv into match rules and bind a
|
// The device registry: parse /system/configuration/devices.csv into match rules and bind a
|
||||||
// reported device to a driver. Pure logic (no hardware, no syscalls), so it
|
// reported device to a driver. Pure logic (no hardware, no syscalls), so it
|
||||||
// unit-tests on the host; the device manager imports it.
|
// unit-tests on the host; the device manager imports it.
|
||||||
_ = b.addModule("device-registry", .{
|
_ = b.addModule("device-registry", .{
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
.kernel = .{ .path = "../kernel" },
|
.kernel = .{ .path = "../kernel" },
|
||||||
// driver speaks device-manager-protocol; block/usb their transfer protocols.
|
// driver speaks device-manager-protocol; block/usb their transfer protocols.
|
||||||
.protocol = .{ .path = "../protocol" },
|
.protocol = .{ .path = "../protocol" },
|
||||||
// device-registry parses /etc/devices.csv with the shared csv helpers.
|
// device-registry parses /system/configuration/devices.csv with the shared csv helpers.
|
||||||
.csv = .{ .path = "../csv" },
|
.csv = .{ .path = "../csv" },
|
||||||
},
|
},
|
||||||
.paths = .{""},
|
.paths = .{""},
|
||||||
|
|||||||
@@ -126,7 +126,7 @@ pub const DeviceDescriptor = extern struct {
|
|||||||
// names with the pci-class module.
|
// names with the pci-class module.
|
||||||
pci_class: u64,
|
pci_class: u64,
|
||||||
// Numeric identity beyond the class triple, mirrored in the bus report's
|
// Numeric identity beyond the class triple, mirrored in the bus report's
|
||||||
// ChildAdded so /etc/devices.csv can bind on it: `vendor`/`device` are the PCI
|
// ChildAdded so /system/configuration/devices.csv can bind on it: `vendor`/`device` are the PCI
|
||||||
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
||||||
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
||||||
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
//! The device registry: parse `/etc/devices.csv` into match rules and bind a
|
//! The device registry: parse `/system/configuration/devices.csv` into match rules and bind a
|
||||||
//! reported device to a driver. This is the data-driven replacement for the
|
//! reported device to a driver. This is the data-driven replacement for the
|
||||||
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
||||||
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
||||||
@@ -11,7 +11,7 @@
|
|||||||
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
||||||
//!
|
//!
|
||||||
//! The file format (docs/device-driver-development/device-manager.md, and the
|
//! The file format (docs/device-driver-development/device-manager.md, and the
|
||||||
//! `/etc/devices.csv` header itself): one rule per line, nine comma-separated
|
//! `/system/configuration/devices.csv` header itself): one rule per line, nine comma-separated
|
||||||
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
||||||
//!
|
//!
|
||||||
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||||
@@ -226,7 +226,7 @@ fn parseLine(line: []const u8) Line {
|
|||||||
} };
|
} };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Parse a whole `/etc/devices.csv` into `out_rules`. The string fields of the
|
/// Parse a whole `/system/configuration/devices.csv` into `out_rules`. The string fields of the
|
||||||
/// returned rules point into `source`, which must outlive them.
|
/// returned rules point into `source`, which must outlive them.
|
||||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||||
|
|||||||
@@ -305,7 +305,7 @@ pub fn makePath(path: []const u8) bool {
|
|||||||
while (end < path.len and path[end] != '/') end += 1;
|
while (end < path.len and path[end] != '/') end += 1;
|
||||||
const prefix = path[0..end];
|
const prefix = path[0..end];
|
||||||
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
||||||
// Best-effort per prefix: components at or above a mount point ("/mnt")
|
// Best-effort per prefix: components at or above a mount point ("/volumes")
|
||||||
// are router names, not filesystem nodes — they neither exist as nodes
|
// are router names, not filesystem nodes — they neither exist as nodes
|
||||||
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
||||||
if (!exists(prefix)) _ = makeDirectory(prefix);
|
if (!exists(prefix)) _ = makeDirectory(prefix);
|
||||||
@@ -348,8 +348,9 @@ pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
/// the backend as `rewrite` + the mount-relative tail. How one volume serves
|
||||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
/// several mounts ("/volumes/usb" from its root, "/system/logs" from its
|
||||||
|
/// /system/logs subtree).
|
||||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||||
return fsMount(target, backend, rewrite);
|
return fsMount(target, backend, rewrite);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -12,7 +12,7 @@
|
|||||||
pub const version: u16 = 1;
|
pub const version: u16 = 1;
|
||||||
|
|
||||||
/// Which bus a `child_added` came from — stated by the reporting bus driver so
|
/// Which bus a `child_added` came from — stated by the reporting bus driver so
|
||||||
/// the manager's /etc/devices.csv matcher knows how to read the report's identity
|
/// the manager's /system/configuration/devices.csv matcher knows how to read the report's identity
|
||||||
/// (a PCI class triple vs a USB class triple are the same 24 bits but different
|
/// (a PCI class triple vs a USB class triple are the same 24 bits but different
|
||||||
/// namespaces) and which `bus` column a rule must name to bind it. `unknown` is
|
/// namespaces) and which `bus` column a rule must name to bind it. `unknown` is
|
||||||
/// the zero default, so an un-upgraded reporter fails to match rather than
|
/// the zero default, so an un-upgraded reporter fails to match rather than
|
||||||
@@ -96,7 +96,7 @@ pub const ChildAdded = extern struct {
|
|||||||
/// for an unregistered leaf (a USB port before the descriptor track).
|
/// for an unregistered leaf (a USB port before the descriptor track).
|
||||||
device_id: u64 = no_device,
|
device_id: u64 = no_device,
|
||||||
/// The vendor id (PCI vendor / USB idVendor), or 0 when the bus has no such
|
/// The vendor id (PCI vendor / USB idVendor), or 0 when the bus has no such
|
||||||
/// concept (ACPI). Carried so the manager's /etc/devices.csv matcher can bind
|
/// concept (ACPI). Carried so the manager's /system/configuration/devices.csv matcher can bind
|
||||||
/// on vendor — a level the bus-native `identity` (a class triple) cannot express.
|
/// on vendor — a level the bus-native `identity` (a class triple) cannot express.
|
||||||
vendor: u16 = 0,
|
vendor: u16 = 0,
|
||||||
/// The device id (PCI device / USB idProduct), or 0. The most specific numeric
|
/// The device id (PCI device / USB idProduct), or 0. The most specific numeric
|
||||||
|
|||||||
+1
-1
@@ -295,7 +295,7 @@ pub const ServiceId = enum(u32) {
|
|||||||
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||||
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
||||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/volumes/usb) to it
|
||||||
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
||||||
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||||
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# /etc/devices.csv — the device→driver registry.
|
# /system/configuration/devices.csv — the device→driver registry.
|
||||||
#
|
#
|
||||||
# The device manager reads this at boot and binds each device a bus driver
|
# The device manager reads this at boot and binds each device a bus driver
|
||||||
# reports to the driver named here. It is AUTHORITATIVE: a device that no row
|
# reports to the driver named here. It is AUTHORITATIVE: a device that no row
|
||||||
|
@@ -1,9 +1,10 @@
|
|||||||
# /etc/init.csv — diagnose variant (-Ddiagnose), bundled at /etc/init.csv.
|
# /system/configuration/init.csv — diagnose variant (-Ddiagnose), bundled at
|
||||||
|
# /system/configuration/init.csv.
|
||||||
#
|
#
|
||||||
# The display stack (display, display-demo) is omitted so the kernel's timestamped
|
# The display stack (display, display-demo) is omitted so the kernel's timestamped
|
||||||
# on-screen boot transcript is never suppressed — the bring-up timeline (USB,
|
# on-screen boot transcript is never suppressed — the bring-up timeline (USB,
|
||||||
# storage, logger) stays readable on real hardware with no serial. See etc/init.csv
|
# storage, logger) stays readable on real hardware with no serial. See
|
||||||
# for the format; this file must otherwise track it.
|
# system/configuration/init.csv for the format; this file must otherwise track it.
|
||||||
#
|
#
|
||||||
# service args...
|
# service args...
|
||||||
/system/services/input
|
/system/services/input
|
||||||
|
@@ -1,4 +1,4 @@
|
|||||||
# /etc/init.csv — the services init (PID 1) starts at boot, in order.
|
# /system/configuration/init.csv — the services init (PID 1) starts at boot, in order.
|
||||||
#
|
#
|
||||||
# init reads this at startup and spawns each service supervised (restarting it on
|
# init reads this at startup and spawns each service supervised (restarting it on
|
||||||
# a crash, up to a cap). Startup order is top->bottom; shutdown is the reverse, so
|
# a crash, up to a cap). Startup order is top->bottom; shutdown is the reverse, so
|
||||||
@@ -9,7 +9,7 @@
|
|||||||
# '#' starts a comment (whole-line or trailing); blank lines are ignored. The
|
# '#' starts a comment (whole-line or trailing); blank lines are ignored. The
|
||||||
# first field is the service binary path; any fields after it are the service's
|
# first field is the service binary path; any fields after it are the service's
|
||||||
# argv. Drivers are absent on purpose — the device manager discovers hardware and
|
# argv. Drivers are absent on purpose — the device manager discovers hardware and
|
||||||
# spawns those (see /etc/devices.csv).
|
# spawns those (see /system/configuration/devices.csv).
|
||||||
#
|
#
|
||||||
# service args...
|
# service args...
|
||||||
/system/services/input
|
/system/services/input
|
||||||
|
@@ -21,7 +21,7 @@ const logging = @import("logging");
|
|||||||
const device_manager_protocol = @import("device-manager-protocol");
|
const device_manager_protocol = @import("device-manager-protocol");
|
||||||
const pci_class = @import("pci-class");
|
const pci_class = @import("pci-class");
|
||||||
|
|
||||||
/// Log a discovered function as its would-be /etc/devices.csv columns (bus, base,
|
/// Log a discovered function as its would-be /system/configuration/devices.csv columns (bus, base,
|
||||||
/// class, prog_if, vendor, device, subsystem) followed by the human-readable
|
/// class, prog_if, vendor, device, subsystem) followed by the human-readable
|
||||||
/// class/subclass/prog-IF names — so a row for a new driver reads straight off the
|
/// class/subclass/prog-IF names — so a row for a new driver reads straight off the
|
||||||
/// boot log. `subsystem` prints as `*` when the function has none, matching the CSV
|
/// boot log. `subsystem` prints as `*` when the function has none, matching the CSV
|
||||||
@@ -154,7 +154,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
|||||||
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
||||||
descriptor.pci_class = class_triple;
|
descriptor.pci_class = class_triple;
|
||||||
// Vendor/device from the first config dword (0x00): low half vendor, high half
|
// Vendor/device from the first config dword (0x00): low half vendor, high half
|
||||||
// device. These carry to the manager's /etc/devices.csv matcher so a function
|
// device. These carry to the manager's /system/configuration/devices.csv matcher so a function
|
||||||
// can bind on its exact 1AF4:1050 identity, not just its class triple.
|
// can bind on its exact 1AF4:1050 identity, not just its class triple.
|
||||||
const vendor_device = configRead(bus, dev, function, 0x00);
|
const vendor_device = configRead(bus, dev, function, 0x00);
|
||||||
descriptor.vendor = @truncate(vendor_device);
|
descriptor.vendor = @truncate(vendor_device);
|
||||||
|
|||||||
@@ -459,7 +459,7 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
|
|||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
// The devices.csv columns (bus=usb, and the class triple as base/class/prog_if)
|
// The devices.csv columns (bus=usb, and the class triple as base/class/prog_if)
|
||||||
// then the human-readable interface name — a would-be /etc/devices.csv row read
|
// then the human-readable interface name — a would-be /system/configuration/devices.csv row read
|
||||||
// straight off the boot log.
|
// straight off the boot log.
|
||||||
std.log.info("port {d} interface {d} bus=usb base={X:0>2} class={X:0>2} prog_if={X:0>2} — {s} registered as device {d}", .{
|
std.log.info("port {d} interface {d} bus=usb base={X:0>2} class={X:0>2} prog_if={X:0>2} — {s} registered as device {d}", .{
|
||||||
port,
|
port,
|
||||||
|
|||||||
@@ -205,7 +205,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
|
||||||
// Config space is resource 0. The registry (/etc/devices.csv) bound this driver by the
|
// Config space is resource 0. The registry (/system/configuration/devices.csv) bound this driver by the
|
||||||
// exact virtio-gpu identity (vendor 0x1AF4 / device 0x1050), so there is no re-confirm to
|
// exact virtio-gpu identity (vendor 0x1AF4 / device 0x1050), so there is no re-confirm to
|
||||||
// do here any more — map config space and enable memory-space decode + bus mastering (the
|
// do here any more — map config space and enable memory-space decode + bus mastering (the
|
||||||
// device DMAs the ring and backing out of RAM; pci-bus only preserves whatever the firmware
|
// device DMAs the ring and backing out of RAM; pci-bus only preserves whatever the firmware
|
||||||
|
|||||||
@@ -257,6 +257,16 @@ pub fn translate(root: u64, virtual: u64) ?u64 {
|
|||||||
return paging.translateIn(root, virtual);
|
return paging.translateIn(root, virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `translate` for an address the kernel is about to touch *on a process's
|
||||||
|
/// behalf*: the walk additionally demands the permission ring 3 would need — the
|
||||||
|
/// leaf user-accessible (U/S set at every level), and writable (R/W at every
|
||||||
|
/// level) when `for_write`. Null means "the process itself could not do this",
|
||||||
|
/// which the checked copy layer (system/kernel/user-memory.zig) turns into
|
||||||
|
/// -EFAULT instead of a kernel dereference.
|
||||||
|
pub fn translateUser(root: u64, virtual: u64, for_write: bool) ?u64 {
|
||||||
|
return paging.translateUserIn(root, virtual, for_write);
|
||||||
|
}
|
||||||
|
|
||||||
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
||||||
/// W^X: code read-only + executable, data writable + no-execute.
|
/// W^X: code read-only + executable, data writable + no-execute.
|
||||||
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||||
|
|||||||
@@ -524,6 +524,58 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
|||||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `translateIn` with the ring-3 permission bits enforced: the walk accumulates
|
||||||
|
/// the protection flags of every level it descends through and refuses the
|
||||||
|
/// translation unless the *effective* permission allows the access ring 3 would
|
||||||
|
/// be allowed — U/S set at every level, and (for `for_write`) R/W set at every
|
||||||
|
/// level too. A bit cleared anywhere on the path denies, which is exactly how
|
||||||
|
/// the MMU combines them, so a checked kernel copy sees the same permissions the
|
||||||
|
/// process itself does.
|
||||||
|
///
|
||||||
|
/// This is the walk `system/kernel/user-memory.zig` copies through, and the
|
||||||
|
/// reason a kernel copy can never be steered at a kernel-only mapping or made to
|
||||||
|
/// write a read-only user page (a process's own text, say).
|
||||||
|
///
|
||||||
|
/// Huge pages: a 2 MiB PDE leaf resolves like `translateIn`, with its own U/S and
|
||||||
|
/// R/W folded into the accumulator first. A PDPTE with PS set (a 1 GiB leaf) is
|
||||||
|
/// refused rather than descended into — danos never builds one, and denying is
|
||||||
|
/// the safe direction for a permission-checked walk.
|
||||||
|
pub fn translateUserIn(pml4: u64, virtual: u64, for_write: bool) ?u64 {
|
||||||
|
// Start all-ones and AND in each level: a cleared bit at any level denies.
|
||||||
|
var effective: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
if (pml4e & present == 0) return null;
|
||||||
|
effective &= pml4e;
|
||||||
|
|
||||||
|
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||||
|
if (pdpte & present == 0) return null;
|
||||||
|
if (pdpte & page_size_bit != 0) return null; // 1 GiB leaf: never built here, refuse
|
||||||
|
effective &= pdpte;
|
||||||
|
|
||||||
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
|
if (pde & present == 0) return null;
|
||||||
|
effective &= pde;
|
||||||
|
if (pde & page_size_bit != 0) { // 2 MiB huge leaf: frame base is bits 51:21
|
||||||
|
if (!permits(effective, for_write)) return null;
|
||||||
|
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
|
if (pte & present == 0) return null;
|
||||||
|
effective &= pte;
|
||||||
|
if (!permits(effective, for_write)) return null;
|
||||||
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether accumulated walk flags allow a ring-3 access: user-accessible always,
|
||||||
|
/// and writable when the access is a store.
|
||||||
|
fn permits(effective: u64, for_write: bool) bool {
|
||||||
|
if (effective & user == 0) return false;
|
||||||
|
if (for_write and effective & writable == 0) return false;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
fn invalidate(virtual: u64) void {
|
fn invalidate(virtual: u64) void {
|
||||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||||
// inline asm won't form directly, so stage the address in a register first.
|
// inline asm won't form directly, so stage the address in a register first.
|
||||||
|
|||||||
@@ -132,13 +132,31 @@ fn record(node: *platform.Device, parent_id: u64) u64 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
||||||
/// available (which may exceed `out.len`).
|
/// available (which may exceed `out.len`). For kernel callers with a buffer big
|
||||||
|
/// enough to take the whole table in one go.
|
||||||
pub fn enumerate(out: []device_abi.DeviceDescriptor) usize {
|
pub fn enumerate(out: []device_abi.DeviceDescriptor) usize {
|
||||||
const n = @min(count, out.len);
|
_ = enumerateFrom(0, out);
|
||||||
@memcpy(out[0..n], devices[0..n]);
|
|
||||||
return count;
|
return count;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// How many devices the table holds — the total `device_enumerate` reports back
|
||||||
|
/// however few of them fit in the caller's buffer.
|
||||||
|
pub fn deviceCount() usize {
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy up to `out.len` descriptors starting at table index `start`, returning how
|
||||||
|
/// many were filled (0 once `start` reaches the end). The chunked form: the
|
||||||
|
/// `device_enumerate` system call bounces the table out through a small kernel
|
||||||
|
/// buffer, one chunk at a time, because a descriptor is far too big to stage a
|
||||||
|
/// whole user-requested array of them on a 16 KiB kernel stack.
|
||||||
|
pub fn enumerateFrom(start: usize, out: []device_abi.DeviceDescriptor) usize {
|
||||||
|
if (start >= count) return 0;
|
||||||
|
const n = @min(count - start, out.len);
|
||||||
|
@memcpy(out[0..n], devices[start..][0..n]);
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
||||||
/// out of range or already claimed.
|
/// out of range or already claimed.
|
||||||
pub fn claim(id: u64, owner: u32) bool {
|
pub fn claim(id: u64, owner: u32) bool {
|
||||||
|
|||||||
@@ -17,18 +17,20 @@
|
|||||||
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
||||||
//! waiting for work use a normal WaitQueue.
|
//! waiting for work use a normal WaitQueue.
|
||||||
//!
|
//!
|
||||||
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
|
//! Trust model: every side of a copy that names a *user* address space goes
|
||||||
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
|
//! through system/kernel/user-memory.zig — user-half bound, page presence, and
|
||||||
//! copy is a later security-track item, matching the existing debug_write gap.
|
//! the leaf permissions ring 3 itself would face (U/S to read, U/S + R/W to
|
||||||
|
//! write). A kernel-side buffer is trusted and translated as-is. An unmapped or
|
||||||
|
//! wrongly-permissioned page fails the operation; it never faults ring 0.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const architecture = @import("architecture");
|
|
||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
|
const user_memory = @import("user-memory.zig");
|
||||||
|
|
||||||
const page_size = abi.page_size;
|
const page_size = abi.page_size;
|
||||||
const Task = scheduler.Task;
|
const Task = scheduler.Task;
|
||||||
@@ -83,8 +85,9 @@ const PostSlot = struct {
|
|||||||
bytes: [POST_MAXIMUM]u8 = undefined,
|
bytes: [POST_MAXIMUM]u8 = undefined,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// End of the user (low) canonical half — user buffers must lie below it.
|
/// End of the user (low) canonical half — user buffers must lie below it. One
|
||||||
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
/// definition, in the module that owns the user-memory contract.
|
||||||
|
const user_half_end: u64 = user_memory.user_half_end;
|
||||||
|
|
||||||
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
||||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||||
@@ -300,18 +303,22 @@ pub fn abandonSenderLocked(t: *Task) void {
|
|||||||
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
|
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
|
||||||
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
|
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
|
||||||
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
||||||
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
|
/// buffers must lie in the low half and carry the permission ring 3 would need for
|
||||||
/// unmapped or out of range. Handles page-straddling buffers.
|
/// their side of the copy — readable to send from, writable to receive into.
|
||||||
|
/// Returns false — never #PFs — if any page is unmapped, out of range, or
|
||||||
|
/// wrongly permissioned. Handles page-straddling buffers.
|
||||||
|
///
|
||||||
|
/// This is the process↔process case, which `user-memory` deliberately does not
|
||||||
|
/// cover (it knows one user address space at a time); both sides resolve through
|
||||||
|
/// `user_memory.resolve`, so the permission rules are the same ones.
|
||||||
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
|
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
|
||||||
const source_root = if (source_as != 0) source_as else architecture.kernelPageTable();
|
if (source_as != 0 and !user_memory.userRangeOk(source_va, len)) return false;
|
||||||
const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable();
|
if (destination_as != 0 and !user_memory.userRangeOk(destination_va, len)) return false;
|
||||||
if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false;
|
|
||||||
if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false;
|
|
||||||
|
|
||||||
var off: usize = 0;
|
var off: usize = 0;
|
||||||
while (off < len) {
|
while (off < len) {
|
||||||
const s = architecture.translate(source_root, source_va + off) orelse return false;
|
const s = user_memory.resolve(source_as, source_va + off, false) orelse return false;
|
||||||
const d = architecture.translate(destination_root, destination_va + off) orelse return false;
|
const d = user_memory.resolve(destination_as, destination_va + off, true) orelse return false;
|
||||||
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
||||||
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
||||||
const n = @min(@min(s_left, d_left), len - off);
|
const n = @min(@min(s_left, d_left), len - off);
|
||||||
@@ -323,27 +330,10 @@ fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_v
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel
|
/// The checked copy-in, re-exported from its home in `user-memory` so the many
|
||||||
/// buffer `destination`, walking the user page tables through the physmap. Returns false if
|
/// `ipc.copyFromUser` call sites keep reading naturally. New code should reach
|
||||||
/// the range escapes the user half or any source page is unmapped — so a bad user
|
/// for `user-memory` directly — it is where the write direction lives too.
|
||||||
/// pointer *fails the system_call* rather than faulting the kernel (danos has no
|
pub const copyFromUser = user_memory.copyFromUser;
|
||||||
/// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt
|
|
||||||
/// the machine). The correct way to pull a fixed-size struct in from user space, and
|
|
||||||
/// a single fetch: no TOCTOU against a hostile pointer.
|
|
||||||
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
|
||||||
if (user_as == 0) return false; // not a user address space
|
|
||||||
if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false;
|
|
||||||
var off: usize = 0;
|
|
||||||
while (off < destination.len) {
|
|
||||||
const s = architecture.translate(user_as, user_va + off) orelse return false;
|
|
||||||
const s_left = page_size - ((user_va + off) & (page_size - 1));
|
|
||||||
const n = @min(s_left, destination.len - off);
|
|
||||||
const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(s));
|
|
||||||
@memcpy(destination[off..][0..n], source[0..n]);
|
|
||||||
off += n;
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- the two IPC operations -------------------------------------------------
|
// --- the two IPC operations -------------------------------------------------
|
||||||
|
|
||||||
|
|||||||
@@ -161,7 +161,7 @@ test "append/read round trip" {
|
|||||||
defer std.testing.allocator.destroy(ring);
|
defer std.testing.allocator.destroy(ring);
|
||||||
ring.* = .{};
|
ring.* = .{};
|
||||||
|
|
||||||
_ = ring.append(7, "/system/services/fat", .info, 123, "mounted /mnt/usb", false);
|
_ = ring.append(7, "/system/services/fat", .info, 123, "mounted /volumes/usb", false);
|
||||||
_ = ring.append(0, "kernel", .raw, 456, "wall clock online", false);
|
_ = ring.append(0, "kernel", .raw, 456, "wall clock online", false);
|
||||||
|
|
||||||
const first = parseAt(ring, ring.tail);
|
const first = parseAt(ring, ring.tail);
|
||||||
@@ -169,7 +169,7 @@ test "append/read round trip" {
|
|||||||
try std.testing.expectEqual(abi.KlogLevel.info, first.header.level);
|
try std.testing.expectEqual(abi.KlogLevel.info, first.header.level);
|
||||||
try std.testing.expectEqual(@as(u64, 123), first.header.timestamp_ns);
|
try std.testing.expectEqual(@as(u64, 123), first.header.timestamp_ns);
|
||||||
try std.testing.expectEqualStrings("/system/services/fat", first.nameSlice());
|
try std.testing.expectEqualStrings("/system/services/fat", first.nameSlice());
|
||||||
try std.testing.expectEqualStrings("mounted /mnt/usb", first.messageSlice());
|
try std.testing.expectEqualStrings("mounted /volumes/usb", first.messageSlice());
|
||||||
|
|
||||||
const second = parseAt(ring, first.next(ring.tail));
|
const second = parseAt(ring, first.next(ring.tail));
|
||||||
try std.testing.expectEqual(@as(u32, 0), second.header.pid);
|
try std.testing.expectEqual(@as(u32, 0), second.header.pid);
|
||||||
|
|||||||
+145
-41
@@ -31,6 +31,7 @@ const scheduler = @import("scheduler.zig");
|
|||||||
const console = @import("console.zig");
|
const console = @import("console.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const ipc = @import("ipc-synchronous.zig");
|
const ipc = @import("ipc-synchronous.zig");
|
||||||
|
const user_memory = @import("user-memory.zig");
|
||||||
const devices_broker = @import("devices-broker.zig");
|
const devices_broker = @import("devices-broker.zig");
|
||||||
const irq = @import("irq.zig");
|
const irq = @import("irq.zig");
|
||||||
const iommu = @import("iommu.zig");
|
const iommu = @import("iommu.zig");
|
||||||
@@ -110,6 +111,14 @@ pub const maximum_arguments = 8;
|
|||||||
/// Ceiling on the `system_spawn` extra-arguments blob (argv[1..], NUL-separated).
|
/// Ceiling on the `system_spawn` extra-arguments blob (argv[1..], NUL-separated).
|
||||||
pub const maximum_argument_bytes = 256;
|
pub const maximum_argument_bytes = 256;
|
||||||
|
|
||||||
|
/// Longest path `fs_resolve` accepts, longest prefix `fs_mount`/`fs_unmount`
|
||||||
|
/// accept, and longest backend rewrite prefix. Each is also the size of the
|
||||||
|
/// kernel staging buffer the argument is copied into, which is why they are
|
||||||
|
/// named here rather than spelled as literals at the check.
|
||||||
|
pub const maximum_resolve_path = 224;
|
||||||
|
pub const maximum_mount_prefix = 64;
|
||||||
|
pub const maximum_mount_rewrite = 32;
|
||||||
|
|
||||||
/// Auxiliary-vector entry types (System V AMD64 process entry). Only what the
|
/// Auxiliary-vector entry types (System V AMD64 process entry). Only what the
|
||||||
/// kernel emits today; a C runtime scans the vector until the null terminator.
|
/// kernel emits today; a C runtime scans the vector until the null terminator.
|
||||||
const auxiliary_vector_null: u64 = 0; // AT_NULL — end of the vector
|
const auxiliary_vector_null: u64 = 0; // AT_NULL — end of the vector
|
||||||
@@ -378,6 +387,12 @@ fn systemIpcSend(state: *architecture.CpuState) void {
|
|||||||
|
|
||||||
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
|
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
|
||||||
/// buffer (up to `maximum` entries), returning the total device count.
|
/// buffer (up to `maximum` entries), returning the total device count.
|
||||||
|
///
|
||||||
|
/// The broker fills a small kernel chunk which `copyToUser` then places in the
|
||||||
|
/// caller's buffer: the kernel never stores through a user pointer, so a bad one
|
||||||
|
/// is -EFAULT instead of a ring-0 page fault. A DeviceDescriptor is a few hundred
|
||||||
|
/// bytes, so the chunk is deliberately tiny — the 16 KiB kernel stack could not
|
||||||
|
/// hold a whole user-requested array of them.
|
||||||
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||||
const maximum = architecture.systemCallArg(state, 1);
|
const maximum = architecture.systemCallArg(state, 1);
|
||||||
@@ -385,8 +400,20 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
|||||||
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||||
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
||||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||||
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
|
||||||
architecture.setSystemCallResult(state, devices_broker.enumerate(out[0..@intCast(cap)]));
|
var chunk: [2]device_abi.DeviceDescriptor = undefined;
|
||||||
|
var copied: u64 = 0;
|
||||||
|
var start: usize = 0;
|
||||||
|
while (copied < cap) {
|
||||||
|
const filled = devices_broker.enumerateFrom(start, &chunk);
|
||||||
|
if (filled == 0) break;
|
||||||
|
start += filled;
|
||||||
|
const take = @min(@as(u64, filled), cap - copied);
|
||||||
|
const bytes = std.mem.sliceAsBytes(chunk[0..@intCast(take)]);
|
||||||
|
if (!user_memory.copyToUser(t.address_space, buffer_ptr + copied * sz, bytes)) return failErr(state, ipc.EFAULT);
|
||||||
|
copied += take;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, devices_broker.deviceCount());
|
||||||
}
|
}
|
||||||
|
|
||||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||||
@@ -969,7 +996,17 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
|||||||
const image = ramdisk_image orelse return fail(state);
|
const image = ramdisk_image orelse return fail(state);
|
||||||
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
||||||
|
|
||||||
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
|
// Both buffers come in through the checked copy layer, once. The lengths are
|
||||||
|
// already bounded above, so the staging arrays are small and fixed — and
|
||||||
|
// because the bytes are now the kernel's own, nothing below can be changed
|
||||||
|
// by another thread of the caller between validation and use.
|
||||||
|
var name_storage: [scheduler.maximum_task_name]u8 = undefined;
|
||||||
|
const name = name_storage[0..@intCast(len)];
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, ptr, name)) return failErr(state, ipc.EFAULT);
|
||||||
|
var argument_storage: [maximum_argument_bytes]u8 = undefined;
|
||||||
|
const arguments = argument_storage[0..@intCast(arguments_len)];
|
||||||
|
if (arguments_len != 0 and !user_memory.copyFromUser(t.address_space, arguments_ptr, arguments)) return failErr(state, ipc.EFAULT);
|
||||||
|
|
||||||
// Exact path first, basename fallback second; either way argv[0] (and hence
|
// Exact path first, basename fallback second; either way argv[0] (and hence
|
||||||
// the task name, and the log ring's attribution) is the stored full path.
|
// the task name, and the log ring's attribution) is the stored full path.
|
||||||
const item = rd.find(name) orelse return fail(state); // no bundled binary by that name
|
const item = rd.find(name) orelse return fail(state); // no bundled binary by that name
|
||||||
@@ -977,8 +1014,7 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
|||||||
argv[0] = item.name;
|
argv[0] = item.name;
|
||||||
var argc: usize = 1;
|
var argc: usize = 1;
|
||||||
if (arguments_len != 0) {
|
if (arguments_len != 0) {
|
||||||
const blob = @as([*]const u8, @ptrFromInt(arguments_ptr))[0..arguments_len];
|
var pieces = std.mem.tokenizeScalar(u8, arguments, 0);
|
||||||
var pieces = std.mem.tokenizeScalar(u8, blob, 0);
|
|
||||||
while (pieces.next()) |piece| {
|
while (pieces.next()) |piece| {
|
||||||
if (argc == maximum_arguments) return fail(state);
|
if (argc == maximum_arguments) return fail(state);
|
||||||
argv[argc] = piece;
|
argv[argc] = piece;
|
||||||
@@ -1129,8 +1165,27 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
|||||||
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||||
const sz = @sizeOf(abi.ProcessDescriptor);
|
const sz = @sizeOf(abi.ProcessDescriptor);
|
||||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||||
const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr);
|
|
||||||
architecture.setSystemCallResult(state, scheduler.enumerate(out[0..@intCast(cap)]));
|
// The scheduler describes a chunk of the table into kernel memory, then
|
||||||
|
// `copyToUser` places it — the kernel never stores through the user pointer.
|
||||||
|
// Once the caller's buffer is full the walk continues with an empty chunk,
|
||||||
|
// because the result is the true live count, not what fitted.
|
||||||
|
var chunk: [8]abi.ProcessDescriptor = undefined;
|
||||||
|
var copied: u64 = 0;
|
||||||
|
var total: u64 = 0;
|
||||||
|
var cursor: usize = 0;
|
||||||
|
while (true) {
|
||||||
|
const room: []abi.ProcessDescriptor = if (copied < cap) chunk[0..@intCast(@min(chunk.len, cap - copied))] else chunk[0..0];
|
||||||
|
const found = scheduler.enumerateFrom(&cursor, room);
|
||||||
|
total += found.live;
|
||||||
|
if (found.filled != 0) {
|
||||||
|
const bytes = std.mem.sliceAsBytes(chunk[0..found.filled]);
|
||||||
|
if (!user_memory.copyToUser(t.address_space, buffer_ptr + copied * sz, bytes)) return failErr(state, ipc.EFAULT);
|
||||||
|
copied += found.filled;
|
||||||
|
}
|
||||||
|
if (found.done) break;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, total);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// process_kill(id) -> 0 / -ESRCH / -EPERM: end the process `id`. Only its
|
/// process_kill(id) -> 0 / -ESRCH / -EPERM: end the process `id`. Only its
|
||||||
@@ -1682,9 +1737,10 @@ fn systemIrqAck(state: *architecture.CpuState) void {
|
|||||||
/// The pointer must lie in the user (low) half, so kernel addresses and
|
/// The pointer must lie in the user (low) half, so kernel addresses and
|
||||||
/// non-canonical values fall outside it and the read below can't be steered at
|
/// non-canonical values fall outside it and the read below can't be steered at
|
||||||
/// kernel data. Length is checked first so the upper-bound add can't overflow.
|
/// kernel data. Length is checked first so the upper-bound add can't overflow.
|
||||||
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
|
/// The message then comes in ONCE through the checked copy layer: an unmapped
|
||||||
/// the user half passes the check and the read #PFs -> on_fault halts — a
|
/// hole in the user half is -EFAULT rather than a kernel fault, and the bytes the
|
||||||
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
|
/// log stamps are the same bytes that were validated (the old code read the user
|
||||||
|
/// buffer twice — once to stage it, once again inside `log.append`).
|
||||||
///
|
///
|
||||||
/// The emit runs under the kernel lock, so a message is atomic on the wire — two
|
/// The emit runs under the kernel lock, so a message is atomic on the wire — two
|
||||||
/// processes writing from different cores can interleave *messages*, never bytes.
|
/// processes writing from different cores can interleave *messages*, never bytes.
|
||||||
@@ -1697,7 +1753,6 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
|
|||||||
const len = architecture.systemCallArg(state, 1);
|
const len = architecture.systemCallArg(state, 1);
|
||||||
const level_raw = architecture.systemCallArg(state, 2);
|
const level_raw = architecture.systemCallArg(state, 2);
|
||||||
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
||||||
const source: [*]const u8 = @ptrFromInt(ptr);
|
|
||||||
// Levels above the enum range clamp to raw — old two-arg callers land
|
// Levels above the enum range clamp to raw — old two-arg callers land
|
||||||
// there naturally (garbage in arg 2 stays harmless).
|
// there naturally (garbage in arg 2 stays harmless).
|
||||||
const level: abi.KlogLevel = if (level_raw <= @intFromEnum(abi.KlogLevel.raw))
|
const level: abi.KlogLevel = if (level_raw <= @intFromEnum(abi.KlogLevel.raw))
|
||||||
@@ -1707,14 +1762,16 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
|
|||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@memcpy(write_buffer[0..len], source[0..len]); // keep the latest message
|
// One copy in, under the lock; `write_buffer` (the latest message, which
|
||||||
|
// the kernel tests assert on) doubles as the staging buffer the log reads.
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, ptr, write_buffer[0..len])) return failErr(state, ipc.EFAULT);
|
||||||
write_len = len;
|
write_len = len;
|
||||||
write_from_user = architecture.fromUser(state);
|
write_from_user = architecture.fromUser(state);
|
||||||
write_count += 1;
|
write_count += 1;
|
||||||
// The kernel stamps the sender's identity — attribution is structural,
|
// The kernel stamps the sender's identity — attribution is structural,
|
||||||
// not a prefix convention the payload could forge (and it is stamped
|
// not a prefix convention the payload could forge (and it is stamped
|
||||||
// per line inside log.append).
|
// per line inside log.append).
|
||||||
log.append(t.id, t.name(), level, source[0..len]);
|
log.append(t.id, t.name(), level, write_buffer[0..len]);
|
||||||
architecture.setSystemCallResult(state, len);
|
architecture.setSystemCallResult(state, len);
|
||||||
} else {
|
} else {
|
||||||
fail(state);
|
fail(state);
|
||||||
@@ -1729,18 +1786,35 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
|
|||||||
/// [KlogRecordHeader][name][message] frames out of the byte stream (abi.zig).
|
/// [KlogRecordHeader][name][message] frames out of the byte stream (abi.zig).
|
||||||
///
|
///
|
||||||
/// The mirror of `debug_write`: the same overflow-safe user-half bounds check,
|
/// The mirror of `debug_write`: the same overflow-safe user-half bounds check,
|
||||||
/// but the copy runs kernel -> user, under the log lock (inside log.readAt) so
|
/// but the copy runs kernel -> user. The ring is drained a chunk at a time into a
|
||||||
/// the stream can't move underneath the copy. A read-only diagnostic.
|
/// kernel staging buffer (each chunk read under the log lock, so the stream can't
|
||||||
|
/// move underneath it) and each chunk is then placed with `copyToUser` — a
|
||||||
|
/// reader may ask for a megabyte, and the kernel stack is 16 KiB. A partial
|
||||||
|
/// result is honest: the reader advances its cursor by what it got. A read-only
|
||||||
|
/// diagnostic.
|
||||||
fn systemKlogRead(state: *architecture.CpuState) void {
|
fn systemKlogRead(state: *architecture.CpuState) void {
|
||||||
const offset = architecture.systemCallArg(state, 0);
|
const offset = architecture.systemCallArg(state, 0);
|
||||||
const ptr = architecture.systemCallArg(state, 1);
|
const ptr = architecture.systemCallArg(state, 1);
|
||||||
const len = architecture.systemCallArg(state, 2);
|
const len = architecture.systemCallArg(state, 2);
|
||||||
|
const t = scheduler.current();
|
||||||
// Confine the whole destination span to the user (low) half. `len <=
|
// Confine the whole destination span to the user (low) half. `len <=
|
||||||
// user_half_end - ptr` bounds the length without an overflowing add.
|
// user_half_end - ptr` bounds the length without an overflowing add.
|
||||||
if (ptr < user_half_end and len <= user_half_end - ptr) {
|
if (ptr < user_half_end and len <= user_half_end - ptr) {
|
||||||
const dest: [*]u8 = @ptrFromInt(ptr);
|
var chunk: [512]u8 = undefined;
|
||||||
const n = log.readAt(offset, dest[0..len]) orelse return fail(state);
|
var done: u64 = 0;
|
||||||
architecture.setSystemCallResult(state, n);
|
while (done < len) {
|
||||||
|
const want = @min(@as(u64, chunk.len), len - done);
|
||||||
|
const n = log.readAt(offset + done, chunk[0..@intCast(want)]) orelse {
|
||||||
|
// The cursor fell behind the ring's tail mid-drain. What was
|
||||||
|
// already placed stands; only a first-chunk miss fails the call.
|
||||||
|
if (done == 0) return fail(state);
|
||||||
|
break;
|
||||||
|
};
|
||||||
|
if (n == 0) break; // caught up
|
||||||
|
if (!user_memory.copyToUser(t.address_space, ptr + done, chunk[0..n])) return failErr(state, ipc.EFAULT);
|
||||||
|
done += n;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, done);
|
||||||
} else {
|
} else {
|
||||||
fail(state);
|
fail(state);
|
||||||
}
|
}
|
||||||
@@ -1753,10 +1827,10 @@ fn systemKlogRead(state: *architecture.CpuState) void {
|
|||||||
fn systemKlogStatus(state: *architecture.CpuState) void {
|
fn systemKlogStatus(state: *architecture.CpuState) void {
|
||||||
const ptr = architecture.systemCallArg(state, 0);
|
const ptr = architecture.systemCallArg(state, 0);
|
||||||
const size = @sizeOf(abi.KlogStatus);
|
const size = @sizeOf(abi.KlogStatus);
|
||||||
|
const t = scheduler.current();
|
||||||
if (ptr < user_half_end and size <= user_half_end - ptr) {
|
if (ptr < user_half_end and size <= user_half_end - ptr) {
|
||||||
var status = log.status();
|
var status = log.status();
|
||||||
const dest: [*]u8 = @ptrFromInt(ptr);
|
if (!user_memory.copyValueToUser(t.address_space, ptr, &status)) return failErr(state, ipc.EFAULT);
|
||||||
@memcpy(dest[0..size], std.mem.asBytes(&status)[0..size]);
|
|
||||||
architecture.setSystemCallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
} else {
|
} else {
|
||||||
fail(state);
|
fail(state);
|
||||||
@@ -1775,10 +1849,12 @@ fn systemFsResolve(state: *architecture.CpuState) void {
|
|||||||
const flags = architecture.systemCallArg(state, 2);
|
const flags = architecture.systemCallArg(state, 2);
|
||||||
const out_ptr = architecture.systemCallArg(state, 3);
|
const out_ptr = architecture.systemCallArg(state, 3);
|
||||||
const out_cap = architecture.systemCallArg(state, 4);
|
const out_cap = architecture.systemCallArg(state, 4);
|
||||||
if (path_len == 0 or path_len > 224 or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
|
if (path_len == 0 or path_len > maximum_resolve_path or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
|
||||||
if (out_cap != 0 and (out_ptr >= user_half_end or out_ptr + out_cap > user_half_end)) return fail(state);
|
if (out_cap != 0 and !user_memory.userRangeOk(out_ptr, @intCast(out_cap))) return fail(state);
|
||||||
const path = @as([*]const u8, @ptrFromInt(path_ptr))[0..path_len];
|
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
|
var path_storage: [maximum_resolve_path]u8 = undefined;
|
||||||
|
const path = path_storage[0..@intCast(path_len)];
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, path_ptr, path)) return failErr(state, ipc.EFAULT);
|
||||||
|
|
||||||
const flags_lock = sync.enter();
|
const flags_lock = sync.enter();
|
||||||
defer sync.leave(flags_lock);
|
defer sync.leave(flags_lock);
|
||||||
@@ -1790,14 +1866,15 @@ fn systemFsResolve(state: *architecture.CpuState) void {
|
|||||||
.backend => |*backend| {
|
.backend => |*backend| {
|
||||||
// The rewritten path goes back in the out buffer behind a u16
|
// The rewritten path goes back in the out buffer behind a u16
|
||||||
// length prefix (a third result register would collide with r8's
|
// length prefix (a third result register would collide with r8's
|
||||||
// argument role in the userspace stub).
|
// argument role in the userspace stub). Both halves are placed with
|
||||||
|
// the checked copy, and *before* the handle is installed, so an
|
||||||
|
// -EFAULT never strands a capability in the caller's table.
|
||||||
if (backend.path_len + 2 > out_cap) return fail(state);
|
if (backend.path_len + 2 > out_cap) return fail(state);
|
||||||
|
const prefix = [2]u8{ @intCast(backend.path_len & 0xFF), @intCast(backend.path_len >> 8) };
|
||||||
|
if (!user_memory.copyToUser(t.address_space, out_ptr, &prefix)) return failErr(state, ipc.EFAULT);
|
||||||
|
if (!user_memory.copyToUser(t.address_space, out_ptr + 2, backend.path[0..backend.path_len])) return failErr(state, ipc.EFAULT);
|
||||||
const handle = ipc.installHandleDeduped(t, backend.endpoint);
|
const handle = ipc.installHandleDeduped(t, backend.endpoint);
|
||||||
if (handle < 0) return fail(state);
|
if (handle < 0) return fail(state);
|
||||||
const destination: [*]u8 = @ptrFromInt(out_ptr);
|
|
||||||
destination[0] = @intCast(backend.path_len & 0xFF);
|
|
||||||
destination[1] = @intCast(backend.path_len >> 8);
|
|
||||||
@memcpy(destination[2..][0..backend.path_len], backend.path[0..backend.path_len]);
|
|
||||||
architecture.setSystemCallResult(state, abi.fs_route_backend);
|
architecture.setSystemCallResult(state, abi.fs_route_backend);
|
||||||
architecture.setSystemCallResult2(state, @intCast(handle));
|
architecture.setSystemCallResult2(state, @intCast(handle));
|
||||||
},
|
},
|
||||||
@@ -1809,6 +1886,12 @@ fn systemFsResolve(state: *architecture.CpuState) void {
|
|||||||
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
|
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
|
||||||
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
|
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
|
||||||
/// of the immutable initrd never take the kernel lock.
|
/// of the immutable initrd never take the kernel lock.
|
||||||
|
///
|
||||||
|
/// Every result reaches the caller through `copyToUser`, never a store through
|
||||||
|
/// the user pointer. `read` stages the file bytes a chunk at a time — a caller
|
||||||
|
/// may ask for the 64 KiB ceiling, which no kernel stack could hold — so a
|
||||||
|
/// mid-way -EFAULT is possible; the call fails and the already-placed prefix is
|
||||||
|
/// meaningless, exactly as a failed read should be.
|
||||||
fn systemFsNode(state: *architecture.CpuState) void {
|
fn systemFsNode(state: *architecture.CpuState) void {
|
||||||
const operation = architecture.systemCallArg(state, 0);
|
const operation = architecture.systemCallArg(state, 0);
|
||||||
const node_token = architecture.systemCallArg(state, 1);
|
const node_token = architecture.systemCallArg(state, 1);
|
||||||
@@ -1817,16 +1900,24 @@ fn systemFsNode(state: *architecture.CpuState) void {
|
|||||||
const buf_len = architecture.systemCallArg(state, 4);
|
const buf_len = architecture.systemCallArg(state, 4);
|
||||||
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
|
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
|
||||||
const capped = @min(buf_len, 64 * 1024); // bound any single copy
|
const capped = @min(buf_len, 64 * 1024); // bound any single copy
|
||||||
const destination: [*]u8 = @ptrFromInt(buf_ptr);
|
const t = scheduler.current();
|
||||||
switch (operation) {
|
switch (operation) {
|
||||||
abi.fs_node_read => {
|
abi.fs_node_read => {
|
||||||
const n = vfs.nodeRead(node_token, offset, destination[0..capped]) orelse return fail(state);
|
var chunk: [512]u8 = undefined;
|
||||||
architecture.setSystemCallResult(state, n);
|
var done: u64 = 0;
|
||||||
|
while (done < capped) {
|
||||||
|
const want = @min(@as(u64, chunk.len), capped - done);
|
||||||
|
const n = vfs.nodeRead(node_token, offset + done, chunk[0..@intCast(want)]) orelse return fail(state);
|
||||||
|
if (n == 0) break; // end of file
|
||||||
|
if (!user_memory.copyToUser(t.address_space, buf_ptr + done, chunk[0..n])) return failErr(state, ipc.EFAULT);
|
||||||
|
done += n;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, done);
|
||||||
},
|
},
|
||||||
abi.fs_node_status => {
|
abi.fs_node_status => {
|
||||||
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
|
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
|
||||||
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
|
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
|
||||||
@memcpy(destination[0..@sizeOf(abi.FileAttributes)], std.mem.asBytes(&attributes));
|
if (!user_memory.copyValueToUser(t.address_space, buf_ptr, &attributes)) return failErr(state, ipc.EFAULT);
|
||||||
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
|
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
|
||||||
},
|
},
|
||||||
abi.fs_node_readdir => {
|
abi.fs_node_readdir => {
|
||||||
@@ -1837,10 +1928,13 @@ fn systemFsNode(state: *architecture.CpuState) void {
|
|||||||
architecture.setSystemCallResult(state, 0); // past the end
|
architecture.setSystemCallResult(state, 0); // past the end
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
var header = result.header;
|
// Header and name are staged contiguously so one entry is one copy.
|
||||||
|
var entry: [@sizeOf(abi.DirectoryEntryHeader) + name_buffer.len]u8 = undefined;
|
||||||
|
const header = result.header;
|
||||||
const total = header_size + @min(result.name_len, capped - header_size);
|
const total = header_size + @min(result.name_len, capped - header_size);
|
||||||
@memcpy(destination[0..header_size], std.mem.asBytes(&header));
|
@memcpy(entry[0..header_size], std.mem.asBytes(&header));
|
||||||
@memcpy(destination[header_size..total], name_buffer[0 .. total - header_size]);
|
@memcpy(entry[header_size..total], name_buffer[0 .. total - header_size]);
|
||||||
|
if (!user_memory.copyToUser(t.address_space, buf_ptr, entry[0..total])) return failErr(state, ipc.EFAULT);
|
||||||
architecture.setSystemCallResult(state, total);
|
architecture.setSystemCallResult(state, total);
|
||||||
},
|
},
|
||||||
else => fail(state),
|
else => fail(state),
|
||||||
@@ -1857,12 +1951,19 @@ fn systemFsMount(state: *architecture.CpuState) void {
|
|||||||
const backend_handle = architecture.systemCallArg(state, 2);
|
const backend_handle = architecture.systemCallArg(state, 2);
|
||||||
const rewrite_ptr = architecture.systemCallArg(state, 3);
|
const rewrite_ptr = architecture.systemCallArg(state, 3);
|
||||||
const rewrite_len = architecture.systemCallArg(state, 4);
|
const rewrite_len = architecture.systemCallArg(state, 4);
|
||||||
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
if (prefix_len == 0 or prefix_len > maximum_mount_prefix or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
||||||
if (rewrite_len > 32) return fail(state);
|
if (rewrite_len > maximum_mount_rewrite) return fail(state);
|
||||||
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
|
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
|
// Both strings come in through the checked copy; `mountBackend` copies them
|
||||||
const rewrite = if (rewrite_len == 0) "" else @as([*]const u8, @ptrFromInt(rewrite_ptr))[0..rewrite_len];
|
// again into the mount table, so these staging buffers only need to outlive
|
||||||
|
// this call.
|
||||||
|
var prefix_storage: [maximum_mount_prefix]u8 = undefined;
|
||||||
|
const prefix = prefix_storage[0..@intCast(prefix_len)];
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
|
||||||
|
var rewrite_storage: [maximum_mount_rewrite]u8 = undefined;
|
||||||
|
const rewrite = rewrite_storage[0..@intCast(rewrite_len)];
|
||||||
|
if (rewrite_len != 0 and !user_memory.copyFromUser(t.address_space, rewrite_ptr, rewrite)) return failErr(state, ipc.EFAULT);
|
||||||
|
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@@ -1879,8 +1980,11 @@ fn systemFsMount(state: *architecture.CpuState) void {
|
|||||||
fn systemFsUnmount(state: *architecture.CpuState) void {
|
fn systemFsUnmount(state: *architecture.CpuState) void {
|
||||||
const prefix_ptr = architecture.systemCallArg(state, 0);
|
const prefix_ptr = architecture.systemCallArg(state, 0);
|
||||||
const prefix_len = architecture.systemCallArg(state, 1);
|
const prefix_len = architecture.systemCallArg(state, 1);
|
||||||
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
if (prefix_len == 0 or prefix_len > maximum_mount_prefix or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
||||||
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
|
const t = scheduler.current();
|
||||||
|
var prefix_storage: [maximum_mount_prefix]u8 = undefined;
|
||||||
|
const prefix = prefix_storage[0..@intCast(prefix_len)];
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
if (!vfs.unmount(prefix)) return fail(state);
|
if (!vfs.unmount(prefix)) return fail(state);
|
||||||
|
|||||||
+48
-11
@@ -1203,18 +1203,56 @@ pub fn destroyTaskLocked(t: *Task) void {
|
|||||||
/// Snapshot the task table into `out` (up to its length), returning the total
|
/// Snapshot the task table into `out` (up to its length), returning the total
|
||||||
/// number of live tasks — the kernel half of `process_enumerate`, mirroring
|
/// number of live tasks — the kernel half of `process_enumerate`, mirroring
|
||||||
/// devices_broker.enumerate. Kernel tasks are included (empty name, supervisor 0):
|
/// devices_broker.enumerate. Kernel tasks are included (empty name, supervisor 0):
|
||||||
/// an honest `ps` shows the idle tasks too. `out` may be user memory: the caller's
|
/// an honest `ps` shows the idle tasks too. `out` is always KERNEL memory: the
|
||||||
/// address space is loaded during its system call, and the same bring-up trust
|
/// system call bounces it out to the caller through the checked copy layer
|
||||||
/// applies as for device_enumerate (an unmapped user page faults the kernel).
|
/// (system/kernel/user-memory.zig), so a bad user pointer fails the call instead
|
||||||
|
/// of faulting ring 0.
|
||||||
pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||||
|
var total: u64 = 0;
|
||||||
|
var cursor: usize = 0;
|
||||||
|
while (true) {
|
||||||
|
// Past the buffer, keep walking with an empty chunk: the total is the
|
||||||
|
// whole live count, however few descriptors the caller had room for.
|
||||||
|
const room = if (total < out.len) out[@intCast(total)..] else out[out.len..];
|
||||||
|
const chunk = enumerateFrom(&cursor, room);
|
||||||
|
total += chunk.live;
|
||||||
|
if (chunk.done) return total;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What one chunk of the task-table walk found.
|
||||||
|
pub const TaskChunk = struct {
|
||||||
|
/// Live tasks passed in this chunk, whether or not they fit in `out` — this
|
||||||
|
/// is what the running total (and hence `process_enumerate`'s result) counts.
|
||||||
|
live: usize,
|
||||||
|
/// How many of those were written into `out` (`@min(live, out.len)`).
|
||||||
|
filled: usize,
|
||||||
|
/// The cursor reached the end of the table: this was the last chunk.
|
||||||
|
done: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One chunk of the task table: starting at slot `cursor` (advanced past
|
||||||
|
/// everything scanned), describe up to `out.len` live tasks into `out` — or, with
|
||||||
|
/// an empty `out`, just count the rest. `cursor == tasks.len` ends the walk.
|
||||||
|
///
|
||||||
|
/// The chunked form exists so `process_enumerate` can stage each chunk in a small
|
||||||
|
/// kernel buffer and copy it out with `user_memory.copyToUser`, rather than
|
||||||
|
/// handing a user pointer to the kernel's own stores. A *slot* cursor, rather
|
||||||
|
/// than a "skip the first N live tasks" count, keeps chunks from duplicating or
|
||||||
|
/// losing an entry when a task exits between them.
|
||||||
|
pub fn enumerateFrom(cursor: *usize, out: []abi.ProcessDescriptor) TaskChunk {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
var total: u64 = 0;
|
var live: usize = 0;
|
||||||
for (&tasks) |*t| {
|
var filled: usize = 0;
|
||||||
|
const limit = if (out.len == 0) tasks.len else out.len; // always makes progress
|
||||||
|
while (cursor.* < tasks.len and live < limit) {
|
||||||
|
const t = &tasks[cursor.*];
|
||||||
|
cursor.* += 1;
|
||||||
if (t.state == .free or t.state == .reaping) continue; // reaping = already exited
|
if (t.state == .free or t.state == .reaping) continue; // reaping = already exited
|
||||||
if (total < out.len) {
|
live += 1;
|
||||||
const d = &out[total];
|
if (filled == out.len) continue;
|
||||||
d.* = .{
|
out[filled] = .{
|
||||||
.id = t.id,
|
.id = t.id,
|
||||||
.supervisor = t.supervisor,
|
.supervisor = t.supervisor,
|
||||||
.leader = t.leader,
|
.leader = t.leader,
|
||||||
@@ -1228,10 +1266,9 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
|||||||
.name_length = t.name_length,
|
.name_length = t.name_length,
|
||||||
.name = t.name_buffer,
|
.name = t.name_buffer,
|
||||||
};
|
};
|
||||||
|
filled += 1;
|
||||||
}
|
}
|
||||||
total += 1;
|
return .{ .live = live, .filled = filled, .done = cursor.* >= tasks.len };
|
||||||
}
|
|
||||||
return total;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether the running task is a user process (has its own address space).
|
/// Whether the running task is a user process (has its own address space).
|
||||||
|
|||||||
+115
-11
@@ -29,6 +29,7 @@ const process = @import("process.zig");
|
|||||||
const initial_ramdisk = @import("initial-ramdisk");
|
const initial_ramdisk = @import("initial-ramdisk");
|
||||||
const kernel_log = @import("log.zig");
|
const kernel_log = @import("log.zig");
|
||||||
const kernel_vfs = @import("vfs.zig");
|
const kernel_vfs = @import("vfs.zig");
|
||||||
|
const user_memory = @import("user-memory.zig");
|
||||||
|
|
||||||
/// Formatted test-marker write. Goes through the kernel log (not straight to
|
/// Formatted test-marker write. Goes through the kernel log (not straight to
|
||||||
/// serial): the log lock is what keeps marker lines from interleaving with
|
/// serial): the log lock is what keeps marker lines from interleaving with
|
||||||
@@ -141,6 +142,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
faultNull();
|
faultNull();
|
||||||
} else if (eql(case, "usermem")) {
|
} else if (eql(case, "usermem")) {
|
||||||
userMemTest();
|
userMemTest();
|
||||||
|
} else if (eql(case, "user-memory")) {
|
||||||
|
userMemoryTest(boot_information);
|
||||||
} else if (eql(case, "user-pf")) {
|
} else if (eql(case, "user-pf")) {
|
||||||
userPfTest();
|
userPfTest();
|
||||||
} else if (eql(case, "fault-recovery")) {
|
} else if (eql(case, "fault-recovery")) {
|
||||||
@@ -1023,6 +1026,103 @@ fn userMemTest() void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The checked copy layer (system/kernel/user-memory.zig), against a scratch
|
||||||
|
/// address space built here rather than a live process — so the refusals can be
|
||||||
|
/// provoked exactly: a kernel-half address, an unmapped user page, and a user
|
||||||
|
/// page mapped read-only. Then the fixture proves the same refusals reach ring 3
|
||||||
|
/// as -errno instead of a kernel fault.
|
||||||
|
fn userMemoryTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: user-memory\n", .{});
|
||||||
|
const base_free = pmm.stats().free_frames;
|
||||||
|
|
||||||
|
const address_space = architecture.createAddressSpace() orelse {
|
||||||
|
check("created a scratch address space", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
check("created a scratch address space", address_space != 0);
|
||||||
|
|
||||||
|
// Three consecutive pages: two writable, the third read-only — so a copy that
|
||||||
|
// straddles into the third proves the write check applies per page, not just
|
||||||
|
// to the first one the walk touches.
|
||||||
|
const writable_pages = 2;
|
||||||
|
const total_pages = 3;
|
||||||
|
const arena = process.heap_arena_base;
|
||||||
|
var frames: [total_pages]u64 = undefined;
|
||||||
|
var mapped: usize = 0;
|
||||||
|
while (mapped < total_pages) : (mapped += 1) {
|
||||||
|
frames[mapped] = pmm.alloc() orelse break;
|
||||||
|
architecture.mapUserPageInto(address_space, arena + mapped * abi.page_size, frames[mapped], mapped < writable_pages, false);
|
||||||
|
}
|
||||||
|
check("mapped two writable and one read-only user page", mapped == total_pages);
|
||||||
|
if (mapped == total_pages) {
|
||||||
|
const read_only = arena + writable_pages * abi.page_size;
|
||||||
|
const unmapped = arena + total_pages * abi.page_size;
|
||||||
|
const kernel_half: u64 = 0xFFFF_8000_0000_0000;
|
||||||
|
|
||||||
|
var out: [16]u8 = undefined;
|
||||||
|
const pattern = [_]u8{ 0xC0, 0xDE, 0xF0, 0x0D, 0xBA, 0xAD, 0xF0, 0x0D };
|
||||||
|
|
||||||
|
// A round trip through the writable page: what copyToUser placed is what
|
||||||
|
// copyFromUser brings back, and the frame really holds it.
|
||||||
|
const wrote = user_memory.copyToUser(address_space, arena + 32, &pattern);
|
||||||
|
const read_back = user_memory.copyFromUser(address_space, arena + 32, out[0..pattern.len]);
|
||||||
|
const frame_view: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frames[0] + 32));
|
||||||
|
check("copyToUser/copyFromUser round trip", wrote and read_back and
|
||||||
|
eql(out[0..pattern.len], &pattern) and eql(frame_view[0..pattern.len], &pattern));
|
||||||
|
|
||||||
|
// Straddling the 4 KiB boundary between the two writable pages.
|
||||||
|
const straddle = arena + abi.page_size - 4;
|
||||||
|
check("a page-straddling round trip", user_memory.copyToUser(address_space, straddle, &pattern) and
|
||||||
|
user_memory.copyFromUser(address_space, straddle, out[0..pattern.len]) and
|
||||||
|
eql(out[0..pattern.len], &pattern));
|
||||||
|
|
||||||
|
// Kernel-half addresses are refused by the range check, before any walk.
|
||||||
|
check("copyToUser refuses a kernel-half address", !user_memory.copyToUser(address_space, kernel_half, &pattern));
|
||||||
|
check("copyFromUser refuses a kernel-half address", !user_memory.copyFromUser(address_space, kernel_half, out[0..pattern.len]));
|
||||||
|
check("a range running off the end of the user half is refused", !user_memory.copyToUser(address_space, user_memory.user_half_end - 4, &pattern));
|
||||||
|
|
||||||
|
// An unmapped-but-in-range page: the latent kernel fault H1 exists to kill.
|
||||||
|
check("copyToUser refuses an unmapped user page", !user_memory.copyToUser(address_space, unmapped, &pattern));
|
||||||
|
check("copyFromUser refuses an unmapped user page", !user_memory.copyFromUser(address_space, unmapped, out[0..pattern.len]));
|
||||||
|
|
||||||
|
// The leaf permission bits: a read-only user page may be read, never written.
|
||||||
|
check("copyToUser refuses a read-only user mapping", !user_memory.copyToUser(address_space, read_only, &pattern));
|
||||||
|
check("copyFromUser accepts a read-only user mapping", user_memory.copyFromUser(address_space, read_only, out[0..pattern.len]));
|
||||||
|
check("a write straddling into a read-only page is refused", !user_memory.copyToUser(address_space, read_only - 4, &pattern));
|
||||||
|
|
||||||
|
// The kernel's own address space is not a user address space.
|
||||||
|
check("copyToUser refuses address space 0", !user_memory.copyToUser(0, arena, &pattern));
|
||||||
|
check("copyFromUser refuses address space 0", !user_memory.copyFromUser(0, arena, out[0..pattern.len]));
|
||||||
|
}
|
||||||
|
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < mapped) : (i += 1) {
|
||||||
|
const va = arena + i * abi.page_size;
|
||||||
|
architecture.unmapUserPageInto(address_space, va);
|
||||||
|
pmm.free(frames[i]);
|
||||||
|
}
|
||||||
|
architecture.destroyAddressSpace(address_space);
|
||||||
|
check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free);
|
||||||
|
|
||||||
|
// Now the ring-3 half: the fixture aims bad pointers at the converted system
|
||||||
|
// calls and must get failures back with the machine still running.
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
check("user-memory-test spawned", spawnNamed(rd, "user-memory-test"));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
// --- synchronous IPC --------------------------------------------------------
|
// --- synchronous IPC --------------------------------------------------------
|
||||||
|
|
||||||
var ipc_endpoint: *ipcsync.Endpoint = undefined;
|
var ipc_endpoint: *ipcsync.Endpoint = undefined;
|
||||||
@@ -1970,7 +2070,7 @@ fn initTest(boot_information: *const BootInformation) void {
|
|||||||
check("init loaded and spawned as a process", spawned);
|
check("init loaded and spawned as a process", spawned);
|
||||||
|
|
||||||
// Wait (real time) until the LAST write is a heartbeat — proving init got
|
// Wait (real time) until the LAST write is a heartbeat — proving init got
|
||||||
// through its boot chatter (heap ok, the /etc/init.csv lookup) and settled
|
// through its boot chatter (heap ok, the /system/configuration/init.csv lookup) and settled
|
||||||
// into its beat-and-sleep loop (~1 s between beats). Waiting on the text
|
// into its beat-and-sleep loop (~1 s between beats). Waiting on the text
|
||||||
// rather than a raw write count: the boot chatter alone satisfies a count,
|
// rather than a raw write count: the boot chatter alone satisfies a count,
|
||||||
// which is exactly the too-early check that used to fail here.
|
// which is exactly the too-early check that used to fail here.
|
||||||
@@ -2220,7 +2320,7 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.write_count = 0;
|
process.write_count = 0;
|
||||||
// The full tree: the storage chain must come up for /mnt/usb to exist —
|
// The full tree: the storage chain must come up for /volumes/usb to exist —
|
||||||
// the fat server (not a router) now owns client file state and its sweep.
|
// the fat server (not a router) now owns client file state and its sweep.
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
|
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
|
||||||
@@ -2606,7 +2706,7 @@ fn usbStorageTest(boot_information: *const BootInformation) void {
|
|||||||
|
|
||||||
/// The FAT mount chain: boot the full tree (init spawns the fat server, which
|
/// The FAT mount chain: boot the full tree (init spawns the fat server, which
|
||||||
/// brings up the USB storage chain, mounts the FAT volume, and mounts itself into
|
/// brings up the USB storage chain, mounts the FAT volume, and mounts itself into
|
||||||
/// the VFS at /mnt/usb), then spawn a fat-test client that lists and reads through
|
/// the VFS at /volumes/usb), then spawn a fat-test client that lists and reads through
|
||||||
/// the mount. The harness attaches a usb-storage device; the expect regex requires
|
/// the mount. The harness attaches a usb-storage device; the expect regex requires
|
||||||
/// the fat mount and the client's success.
|
/// the fat mount and the client's success.
|
||||||
fn fatMountTest(boot_information: *const BootInformation) void {
|
fn fatMountTest(boot_information: *const BootInformation) void {
|
||||||
@@ -2801,11 +2901,13 @@ fn initialRamdiskTest(boot_information: *const BootInformation) void {
|
|||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
// The FHS boot tree ferries data files too (/etc/devices.csv,
|
// The boot tree ferries data files too (/system/configuration/devices.csv,
|
||||||
// /etc/init.csv — served read-only by the kernel VFS, never spawned);
|
// /system/configuration/init.csv — served read-only by the kernel VFS,
|
||||||
// only the /system and /test trees hold programs, so only those count
|
// never spawned); only the /system and /test trees hold programs, and
|
||||||
// toward the spawn-everything sweep.
|
// /system/configuration holds none, so only the rest counts toward the
|
||||||
const is_program = std.mem.startsWith(u8, item.name, "/system/") or
|
// spawn-everything sweep.
|
||||||
|
const is_program = (std.mem.startsWith(u8, item.name, "/system/") and
|
||||||
|
!std.mem.startsWith(u8, item.name, "/system/configuration/")) or
|
||||||
std.mem.startsWith(u8, item.name, "/test/");
|
std.mem.startsWith(u8, item.name, "/test/");
|
||||||
if (!is_program) continue;
|
if (!is_program) continue;
|
||||||
programs += 1;
|
programs += 1;
|
||||||
@@ -3267,21 +3369,23 @@ fn kernelVfsTest(boot_information: *const BootInformation) void {
|
|||||||
check("its first bytes are an ELF magic", n == 4 and header[0] == 0x7f and header[1] == 'E' and header[2] == 'L' and header[3] == 'F');
|
check("its first bytes are an ELF magic", n == 4 and header[0] == 0x7f and header[1] == 'E' and header[2] == 'L' and header[3] == 'F');
|
||||||
}
|
}
|
||||||
|
|
||||||
// Directories resolve and enumerate: /system lists services/drivers.
|
// Directories resolve and enumerate: /system lists services/drivers/
|
||||||
|
// configuration (the CSV data files ride the same initrd tree).
|
||||||
const root_directory = kernel_vfs.resolvePath("/system", false);
|
const root_directory = kernel_vfs.resolvePath("/system", false);
|
||||||
check("/system resolves to a directory node", root_directory == .kernel_node);
|
check("/system resolves to a directory node", root_directory == .kernel_node);
|
||||||
var saw_services = false;
|
var saw_services = false;
|
||||||
var saw_drivers = false;
|
var saw_drivers = false;
|
||||||
|
var saw_configuration = false;
|
||||||
var saw_stray_in_root = false;
|
var saw_stray_in_root = false;
|
||||||
var saw_files_in_services = false;
|
var saw_files_in_services = false;
|
||||||
if (root_directory == .kernel_node) {
|
if (root_directory == .kernel_node) {
|
||||||
var cursor: u64 = 0;
|
var cursor: u64 = 0;
|
||||||
var name: [64]u8 = undefined;
|
var name: [64]u8 = undefined;
|
||||||
while (kernel_vfs.nodeReaddir(root_directory.kernel_node, cursor, &name)) |entry| : (cursor += 1) {
|
while (kernel_vfs.nodeReaddir(root_directory.kernel_node, cursor, &name)) |entry| : (cursor += 1) {
|
||||||
if (eql(name[0..entry.name_len], "services")) saw_services = true else if (eql(name[0..entry.name_len], "drivers")) saw_drivers = true else saw_stray_in_root = true;
|
if (eql(name[0..entry.name_len], "services")) saw_services = true else if (eql(name[0..entry.name_len], "drivers")) saw_drivers = true else if (eql(name[0..entry.name_len], "configuration")) saw_configuration = true else saw_stray_in_root = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
check("readdir /system yields services and drivers", saw_services and saw_drivers);
|
check("readdir /system yields services, drivers, configuration", saw_services and saw_drivers and saw_configuration);
|
||||||
check("readdir /system yields nothing else (no /test leakage)", !saw_stray_in_root);
|
check("readdir /system yields nothing else (no /test leakage)", !saw_stray_in_root);
|
||||||
const services = kernel_vfs.resolvePath("/system/services", false);
|
const services = kernel_vfs.resolvePath("/system/services", false);
|
||||||
if (services == .kernel_node) {
|
if (services == .kernel_node) {
|
||||||
|
|||||||
@@ -0,0 +1,117 @@
|
|||||||
|
//! The single trusted door between ring 0 and a process's memory.
|
||||||
|
//!
|
||||||
|
//! Kernel code never dereferences a user virtual address. It walks that address
|
||||||
|
//! space's page tables through the physmap — kernel mappings throughout — and
|
||||||
|
//! moves the bytes there. Three properties fall out of that one decision:
|
||||||
|
//!
|
||||||
|
//! - **A bad pointer fails the system call.** danos has no fault-recovering
|
||||||
|
//! copy-in, so a raw dereference of an unmapped-but-in-range user page would
|
||||||
|
//! halt the machine. Here it is a `false` return and an `-EFAULT`.
|
||||||
|
//! - **The copy is a single fetch.** A struct pulled in once cannot be changed
|
||||||
|
//! underneath the checks that follow it — no TOCTOU against a hostile pointer.
|
||||||
|
//! - **It is SMAP-proof by construction.** No ring-0 access to a user-mapped
|
||||||
|
//! page ever happens, so the CR4.SMAP bit needs no `stac` window anywhere
|
||||||
|
//! (docs/os-development/smep-smap.md). There is no `stac` in this tree, and a
|
||||||
|
//! change that adds one is wrong by definition.
|
||||||
|
//!
|
||||||
|
//! The walk enforces the permissions ring 3 itself would face: a read needs the
|
||||||
|
//! leaf user-accessible (U/S at every level), a write needs it writable too
|
||||||
|
//! (R/W at every level). So a syscall argument cannot steer the kernel at a
|
||||||
|
//! kernel-only mapping, nor make it write a process's own read-only text — the
|
||||||
|
//! properties that matter once shared or copy-on-write mappings exist, and the
|
||||||
|
//! reason the "presence only" caveat that used to sit at the top of
|
||||||
|
//! ipc-synchronous.zig is gone.
|
||||||
|
//!
|
||||||
|
//! Scope: this module knows only about *user* address spaces. The kernel side of
|
||||||
|
//! a copy (an IPC reply staged in kernel memory, a bounce buffer) is trusted and
|
||||||
|
//! translated without permission checks — see `resolve`.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const architecture = @import("architecture");
|
||||||
|
|
||||||
|
const page_size = abi.page_size;
|
||||||
|
|
||||||
|
/// End of the user (low) canonical half. Every user buffer must lie below it, so
|
||||||
|
/// kernel addresses and non-canonical values are refused by the range check
|
||||||
|
/// alone, before any table is read.
|
||||||
|
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||||
|
|
||||||
|
/// Whether `[virtual, virtual + len)` lies wholly inside the user half. The
|
||||||
|
/// length is compared against the remaining span rather than added to the base,
|
||||||
|
/// so a huge `len` cannot wrap the check.
|
||||||
|
pub fn userRangeOk(virtual: u64, len: usize) bool {
|
||||||
|
if (virtual >= user_half_end) return false;
|
||||||
|
return len <= user_half_end - virtual;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve one address for a copy. `address_space == 0` means the kernel's own
|
||||||
|
/// tables — trusted, translated as-is. A real address space is a process's, and
|
||||||
|
/// the walk demands what ring 3 would need: user-accessible, plus writable when
|
||||||
|
/// this side of the copy is the destination.
|
||||||
|
///
|
||||||
|
/// A user frame must also be reachable *through the physmap*, because that is how
|
||||||
|
/// the copy loops touch it. The physmap covers RAM only: `paging.init` skips every
|
||||||
|
/// `.mmio` region, while `mmio_map` hands a driver its device's BAR as an ordinary
|
||||||
|
/// user-accessible mapping. Such a page satisfies the permission walk and would
|
||||||
|
/// then fault ring 0 on the physmap alias — the very #PF this layer exists to make
|
||||||
|
/// impossible — so coverage is confirmed before the address is returned, and an
|
||||||
|
/// uncovered frame is refused like any other bad buffer. (Confirming coverage,
|
||||||
|
/// rather than testing the `device_grant` bit, is what keeps physmap-backed RAM
|
||||||
|
/// that merely carries that bit — a scanout surface — usable as a buffer.)
|
||||||
|
pub fn resolve(address_space: u64, virtual: u64, for_write: bool) ?u64 {
|
||||||
|
if (address_space == 0) return architecture.translate(architecture.kernelPageTable(), virtual);
|
||||||
|
const physical = architecture.translateUser(address_space, virtual, for_write) orelse return null;
|
||||||
|
if (architecture.translate(architecture.kernelPageTable(), boot_handoff.physicalToVirtual(physical)) == null) return null;
|
||||||
|
return physical;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the
|
||||||
|
/// kernel buffer `destination`. False — never a #PF — if the range escapes the
|
||||||
|
/// user half, or any source page is unmapped or not readable from ring 3.
|
||||||
|
/// Handles page-straddling buffers.
|
||||||
|
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||||
|
if (user_as == 0) return false; // not a user address space
|
||||||
|
if (!userRangeOk(user_va, destination.len)) return false;
|
||||||
|
var off: usize = 0;
|
||||||
|
while (off < destination.len) {
|
||||||
|
const physical = resolve(user_as, user_va + off, false) orelse return false;
|
||||||
|
const left = page_size - ((user_va + off) & (page_size - 1));
|
||||||
|
const n = @min(left, destination.len - off);
|
||||||
|
const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||||
|
@memcpy(destination[off..][0..n], source[0..n]);
|
||||||
|
off += n;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The write direction: copy the kernel buffer `source` out to `user_va` in
|
||||||
|
/// address space `user_as`. False — never a #PF, never a partial promise — if the
|
||||||
|
/// range escapes the user half, or any destination page is unmapped, kernel-only,
|
||||||
|
/// or read-only for ring 3. (A refusal mid-way may already have written earlier
|
||||||
|
/// pages; the caller fails the whole system call, so the buffer's contents are
|
||||||
|
/// meaningless either way.)
|
||||||
|
///
|
||||||
|
/// The mirror of `copyFromUser`, and the only way kernel data reaches a user
|
||||||
|
/// buffer outside the IPC path's `copyAcross`.
|
||||||
|
pub fn copyToUser(user_as: u64, user_va: u64, source: []const u8) bool {
|
||||||
|
if (user_as == 0) return false; // not a user address space
|
||||||
|
if (!userRangeOk(user_va, source.len)) return false;
|
||||||
|
var off: usize = 0;
|
||||||
|
while (off < source.len) {
|
||||||
|
const physical = resolve(user_as, user_va + off, true) orelse return false;
|
||||||
|
const left = page_size - ((user_va + off) & (page_size - 1));
|
||||||
|
const n = @min(left, source.len - off);
|
||||||
|
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||||
|
@memcpy(destination[0..n], source[off..][0..n]);
|
||||||
|
off += n;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `copyToUser` for any fixed-layout value — the shape most write-direction
|
||||||
|
/// system calls want (a `KlogStatus`, a `FileAttributes`).
|
||||||
|
pub fn copyValueToUser(user_as: u64, user_va: u64, value: anytype) bool {
|
||||||
|
return copyToUser(user_as, user_va, std.mem.asBytes(value));
|
||||||
|
}
|
||||||
+38
-15
@@ -7,7 +7,8 @@
|
|||||||
//! mount (the initrd trees at /system and /test, the scratch ram nodes) resolves to a
|
//! mount (the initrd trees at /system and /test, the scratch ram nodes) resolves to a
|
||||||
//! stateless node TOKEN served directly by `fs_node` (read/status/readdir
|
//! stateless node TOKEN served directly by `fs_node` (read/status/readdir
|
||||||
//! with copy-out). A path under a USERSPACE mount (the fat server at
|
//! with copy-out). A path under a USERSPACE mount (the fat server at
|
||||||
//! /mnt/usb and /var) resolves to the backend's ENDPOINT: the kernel
|
//! /volumes/usb, /system/configuration, and /system/logs) resolves to the
|
||||||
|
//! backend's ENDPOINT: the kernel
|
||||||
//! installs a (deduplicated) handle in the caller's table, rewrites the
|
//! installs a (deduplicated) handle in the caller's table, rewrites the
|
||||||
//! path mount-relative, and the caller speaks the unchanged vfs-protocol
|
//! path mount-relative, and the caller speaks the unchanged vfs-protocol
|
||||||
//! to the backend over the ordinary ipc_call rendezvous. The kernel never
|
//! to the backend over the ordinary ipc_call rendezvous. The kernel never
|
||||||
@@ -21,9 +22,10 @@
|
|||||||
//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the
|
//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the
|
||||||
//! backend endpoint handle is the capability, exactly the trust of the old
|
//! backend endpoint handle is the capability, exactly the trust of the old
|
||||||
//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount
|
//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount
|
||||||
//! into the backend's namespace ("/var" -> fat's "/var" subtree while the same
|
//! into the backend's namespace ("/system/logs" -> the boot volume's
|
||||||
//! backend also serves "/mnt/usb" from its root), so FHS paths stay decoupled
|
//! identically-named subtree while the same backend also serves "/volumes/usb"
|
||||||
//! from which volume happens to carry them.
|
//! from its root), so hierarchy paths stay decoupled from which volume happens
|
||||||
|
//! to carry them.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
@@ -99,7 +101,7 @@ var directory_count: usize = 0;
|
|||||||
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by
|
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by
|
||||||
/// a path separator — return the path relative to the mount ("/" for an exact
|
/// a path separator — return the path relative to the mount ("/" for an exact
|
||||||
/// match, otherwise the tail beginning with '/'). Null when not under the
|
/// match, otherwise the tail beginning with '/'). Null when not under the
|
||||||
/// mount, so "/mnt/usb" never captures "/mnt/usbextra".
|
/// mount, so "/volumes/usb" never captures "/volumes/usbextra".
|
||||||
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
||||||
if (path.len < mount_prefix.len) return null;
|
if (path.len < mount_prefix.len) return null;
|
||||||
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
||||||
@@ -182,11 +184,16 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
|
|||||||
|
|
||||||
// --- resolve -----------------------------------------------------------------
|
// --- resolve -----------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Longest rewritten mount-relative path a backend resolution can carry — the
|
||||||
|
/// size `fs_resolve`'s caller has to have room for, so it is named rather than
|
||||||
|
/// spelled out at the one place that builds it.
|
||||||
|
pub const maximum_backend_path = maximum_rewrite + maximum_prefix + 160;
|
||||||
|
|
||||||
pub const Resolved = union(enum) {
|
pub const Resolved = union(enum) {
|
||||||
/// Kernel-served: a permanent node token.
|
/// Kernel-served: a permanent node token.
|
||||||
kernel_node: u64,
|
kernel_node: u64,
|
||||||
/// Backend-served: the endpoint plus the rewritten mount-relative path.
|
/// Backend-served: the endpoint plus the rewritten mount-relative path.
|
||||||
backend: struct { endpoint: *ipc.Endpoint, path: [maximum_rewrite + maximum_prefix + 160]u8, path_len: usize },
|
backend: struct { endpoint: *ipc.Endpoint, path: [maximum_backend_path]u8, path_len: usize },
|
||||||
not_found: void,
|
not_found: void,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -326,14 +333,30 @@ pub fn nodeReaddir(node_token: u64, cursor: u64, name_out: []u8) ?struct { heade
|
|||||||
|
|
||||||
// --- mount/unmount (syscall bodies; caller resolved the handle) --------------
|
// --- mount/unmount (syscall bodies; caller resolved the handle) --------------
|
||||||
|
|
||||||
|
/// The writable subtrees a backend may mount beneath an initrd tree — exactly
|
||||||
|
/// these two, nothing else. Longest-prefix resolution then routes them to the
|
||||||
|
/// volume while every other /system and /test path stays initrd-served, so no
|
||||||
|
/// bundled binary can ever be shadowed.
|
||||||
|
const initrd_carve_outs = [_][]const u8{ "/system/configuration", "/system/logs" };
|
||||||
|
|
||||||
|
fn isInitrdCarveOut(prefix: []const u8) bool {
|
||||||
|
for (initrd_carve_outs) |allowed| {
|
||||||
|
if (std.mem.eql(u8, prefix, allowed)) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
||||||
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
||||||
/// shadowing or replacing the initrd trees (/system, /test).
|
/// shadowing or replacing the initrd trees (/system, /test) — except the two
|
||||||
|
/// carve-outs in `initrd_carve_outs`, the writable configuration/log subtrees.
|
||||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
||||||
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
||||||
if (rewrite.len > maximum_rewrite) return false;
|
if (rewrite.len > maximum_rewrite) return false;
|
||||||
for (&mounts) |*m| { // the initrd trees are not shadowable
|
for (&mounts) |*m| { // the initrd trees are not shadowable (carve-outs aside)
|
||||||
if (m.used and m.kind == .kernel_initrd and underMount(prefix, m.prefixSlice()) != null) return false;
|
if (m.used and m.kind == .kernel_initrd and underMount(prefix, m.prefixSlice()) != null) {
|
||||||
|
if (!isInitrdCarveOut(prefix)) return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
installMount(prefix, .backend, backend, rewrite);
|
installMount(prefix, .backend, backend, rewrite);
|
||||||
return true;
|
return true;
|
||||||
@@ -353,12 +376,12 @@ pub fn unmount(prefix: []const u8) bool {
|
|||||||
// --- tests (host) ------------------------------------------------------------
|
// --- tests (host) ------------------------------------------------------------
|
||||||
|
|
||||||
test "underMount matches only at path boundaries" {
|
test "underMount matches only at path boundaries" {
|
||||||
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
|
try std.testing.expectEqualStrings("/", underMount("/volumes/usb", "/volumes/usb").?);
|
||||||
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
|
try std.testing.expectEqualStrings("/system/kernel", underMount("/volumes/usb/system/kernel", "/volumes/usb").?);
|
||||||
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null);
|
try std.testing.expect(underMount("/volumes/usbextra", "/volumes/usb") == null);
|
||||||
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null);
|
try std.testing.expect(underMount("/volumes", "/volumes/usb") == null);
|
||||||
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
|
try std.testing.expect(underMount("/other", "/volumes/usb") == null);
|
||||||
try std.testing.expect(underMount("greeting", "/mnt/usb") == null);
|
try std.testing.expect(underMount("greeting", "/volumes/usb") == null);
|
||||||
}
|
}
|
||||||
|
|
||||||
test "parentOf walks toward the root" {
|
test "parentOf walks toward the root" {
|
||||||
|
|||||||
@@ -215,7 +215,7 @@ fn onInit(endpoint: ipc.Handle) bool {
|
|||||||
const entry = registered[i];
|
const entry = registered[i];
|
||||||
const hid = entry.hid[0..entry.hid_len];
|
const hid = entry.hid[0..entry.hid_len];
|
||||||
// The devices.csv columns (bus=acpi, hid) then the human-readable name — a
|
// The devices.csv columns (bus=acpi, hid) then the human-readable name — a
|
||||||
// would-be /etc/devices.csv row read straight off the boot log.
|
// would-be /system/configuration/devices.csv row read straight off the boot log.
|
||||||
const desc = acpi_ids.description(hid);
|
const desc = acpi_ids.description(hid);
|
||||||
if (desc.len != 0)
|
if (desc.len != 0)
|
||||||
std.log.info("device {d} bus=acpi hid={s} — {s} ({d} resources)", .{ entry.device_id, hid, desc, entry.resource_count })
|
std.log.info("device {d} bus=acpi hid={s} — {s} ({d} resources)", .{ entry.device_id, hid, desc, entry.resource_count })
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ const registry = @import("device-registry");
|
|||||||
const fs = @import("file-system");
|
const fs = @import("file-system");
|
||||||
|
|
||||||
// --- the device registry ------------------------------------------------------
|
// --- the device registry ------------------------------------------------------
|
||||||
// Driver matching is data-driven and authoritative: /etc/devices.csv (parsed by
|
// Driver matching is data-driven and authoritative: /system/configuration/devices.csv (parsed by
|
||||||
// the device-registry module) names, per bus, which driver binds a reported
|
// the device-registry module) names, per bus, which driver binds a reported
|
||||||
// device, the most-specific match winning. There is no compiled-in fallback — a
|
// device, the most-specific match winning. There is no compiled-in fallback — a
|
||||||
// device no row matches goes unbound and is logged. This retired the hand-kept
|
// device no row matches goes unbound and is logged. This retired the hand-kept
|
||||||
@@ -41,12 +41,12 @@ var registry_source: [8192]u8 = undefined;
|
|||||||
var registry_rules: [64]registry.Rule = undefined;
|
var registry_rules: [64]registry.Rule = undefined;
|
||||||
var registry_count: usize = 0;
|
var registry_count: usize = 0;
|
||||||
|
|
||||||
/// Read and parse /etc/devices.csv once at boot. The file lives in the initial
|
/// Read and parse /system/configuration/devices.csv once at boot. The file lives in the initial
|
||||||
/// ramdisk, which the kernel serves directly — no filesystem service need be up
|
/// ramdisk, which the kernel serves directly — no filesystem service need be up
|
||||||
/// (fat is spawned after the manager), so this is a plain fs.open + read.
|
/// (fat is spawned after the manager), so this is a plain fs.open + read.
|
||||||
fn loadRegistry() void {
|
fn loadRegistry() void {
|
||||||
var file = fs.open("/etc/devices.csv", .{}) orelse {
|
var file = fs.open("/system/configuration/devices.csv", .{}) orelse {
|
||||||
_ = logging.write("/system/services/device-manager: /etc/devices.csv missing — nothing will match\n");
|
_ = logging.write("/system/services/device-manager: /system/configuration/devices.csv missing — nothing will match\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
defer file.close();
|
defer file.close();
|
||||||
@@ -58,9 +58,9 @@ fn loadRegistry() void {
|
|||||||
}
|
}
|
||||||
const result = registry.parse(registry_source[0..used], ®istry_rules);
|
const result = registry.parse(registry_source[0..used], ®istry_rules);
|
||||||
registry_count = result.count;
|
registry_count = result.count;
|
||||||
if (result.malformed != 0) std.log.info("/etc/devices.csv: {d} malformed line(s) skipped", .{result.malformed});
|
if (result.malformed != 0) std.log.info("/system/configuration/devices.csv: {d} malformed line(s) skipped", .{result.malformed});
|
||||||
if (result.truncated) _ = logging.write("/system/services/device-manager: /etc/devices.csv has more rules than the table holds\n");
|
if (result.truncated) _ = logging.write("/system/services/device-manager: /system/configuration/devices.csv has more rules than the table holds\n");
|
||||||
std.log.info("/etc/devices.csv: {d} rule(s) loaded", .{registry_count});
|
std.log.info("/system/configuration/devices.csv: {d} rule(s) loaded", .{registry_count});
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build a registry Identity from a bus driver's report: the bus it named, the
|
/// Build a registry Identity from a bus driver's report: the bus it named, the
|
||||||
@@ -443,7 +443,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
|||||||
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||||
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||||
if (status == 0) publishEvent(message[0..device_manager_protocol.child_added_size]);
|
if (status == 0) publishEvent(message[0..device_manager_protocol.child_added_size]);
|
||||||
// Matching from reports (M19.3), now data-driven via the /etc/devices.csv
|
// Matching from reports (M19.3), now data-driven via the /system/configuration/devices.csv
|
||||||
// registry: a registered child gets the most-specific driver its identity
|
// registry: a registered child gets the most-specific driver its identity
|
||||||
// matches, once — re-reports after a bus restart dedupe on the registered
|
// matches, once — re-reports after a bus restart dedupe on the registered
|
||||||
// id, exactly like the registrations do.
|
// id, exactly like the registrations do.
|
||||||
@@ -451,7 +451,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
|||||||
const id = identityFromReport(report);
|
const id = identityFromReport(report);
|
||||||
if (registry.matchDriver(registry_rules[0..registry_count], id)) |match| {
|
if (registry.matchDriver(registry_rules[0..registry_count], id)) |match| {
|
||||||
if (match.ambiguous)
|
if (match.ambiguous)
|
||||||
std.log.info("/etc/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
|
std.log.info("/system/configuration/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
|
||||||
if (id.bus == .acpi) {
|
if (id.bus == .acpi) {
|
||||||
// An hid-matched driver (ps2-bus) is a singleton that finds its
|
// An hid-matched driver (ps2-bus) is a singleton that finds its
|
||||||
// own devices once spawned — spawn it once, no device assignment.
|
// own devices once spawned — spawn it once, no device assignment.
|
||||||
|
|||||||
@@ -73,7 +73,7 @@ const entries_per_sector = sector_size / @sizeOf(on_disk.DirectoryEntry); // 16
|
|||||||
// A small write-through cache of single-sector (metadata) accesses: FAT sectors,
|
// A small write-through cache of single-sector (metadata) accesses: FAT sectors,
|
||||||
// directory sectors, and directory-entry writebacks. Its payoff is repeated scans
|
// directory sectors, and directory-entry writebacks. Its payoff is repeated scans
|
||||||
// — resolving many paths under the same directory (a logging burst opening dozens
|
// — resolving many paths under the same directory (a logging burst opening dozens
|
||||||
// of files under /var/log/<stamp>/) re-reads the same directory and FAT sectors,
|
// of files under /system/logs/<stamp>/) re-reads the same directory and FAT sectors,
|
||||||
// which now come from RAM instead of a USB round trip each. Bulk file data (the
|
// which now come from RAM instead of a USB round trip each. Bulk file data (the
|
||||||
// multi-sector run path) bypasses the cache — it is large and not re-read — and a
|
// multi-sector run path) bypasses the cache — it is large and not re-read — and a
|
||||||
// run write invalidates any overlapping cached sector to stay coherent.
|
// run write invalidates any overlapping cached sector to stay coherent.
|
||||||
|
|||||||
+17
-10
@@ -1,8 +1,8 @@
|
|||||||
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
||||||
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
||||||
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
||||||
//! the VFS at /mnt/usb. From then on the VFS forwards every open/read/write/
|
//! the VFS at /volumes/usb. From then on the VFS forwards every open/read/write/
|
||||||
//! status/readdir/close under /mnt/usb to this server, which serves the same
|
//! status/readdir/close under /volumes/usb to this server, which serves the same
|
||||||
//! vfs-protocol as a backend — turning block reads into file reads.
|
//! vfs-protocol as a backend — turning block reads into file reads.
|
||||||
//!
|
//!
|
||||||
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
||||||
@@ -22,7 +22,7 @@ const engine = @import("engine.zig");
|
|||||||
const on_disk = @import("on-disk.zig");
|
const on_disk = @import("on-disk.zig");
|
||||||
const vfs_protocol = @import("vfs-protocol");
|
const vfs_protocol = @import("vfs-protocol");
|
||||||
|
|
||||||
const mount_point = "/mnt/usb";
|
const mount_point = "/volumes/usb";
|
||||||
|
|
||||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||||
// buffer the driver reads/writes by physical address.
|
// buffer the driver reads/writes by physical address.
|
||||||
@@ -144,18 +144,25 @@ fn tryBringUp() void {
|
|||||||
};
|
};
|
||||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||||
|
|
||||||
// Mount ourselves into the kernel VFS at /mnt/usb — and serve /var from the
|
// Mount ourselves into the kernel VFS at /volumes/usb — and serve
|
||||||
// volume's /var subtree, so FHS paths (the logger's /var/log) stay decoupled
|
// /system/configuration and /system/logs from the volume's identically-named
|
||||||
// from which volume carries them.
|
// subtrees (the boot volume is hierarchy-shaped, so rewrite == prefix), so
|
||||||
|
// hierarchy paths (the logger's /system/logs) stay decoupled from which
|
||||||
|
// volume carries them.
|
||||||
if (file_system.mount(mount_point, endpointForMount())) {
|
if (file_system.mount(mount_point, endpointForMount())) {
|
||||||
std.log.info("mounted {s}", .{mount_point});
|
std.log.info("mounted {s}", .{mount_point});
|
||||||
} else {
|
} else {
|
||||||
_ = logging.write("/system/services/fat: could not mount /mnt/usb\n");
|
_ = logging.write("/system/services/fat: could not mount /volumes/usb\n");
|
||||||
}
|
}
|
||||||
if (file_system.mountRewritten("/var", endpointForMount(), "/var")) {
|
if (file_system.mountRewritten("/system/configuration", endpointForMount(), "/system/configuration")) {
|
||||||
std.log.info("mounted /var", .{});
|
std.log.info("mounted /system/configuration", .{});
|
||||||
} else {
|
} else {
|
||||||
_ = logging.write("/system/services/fat: could not mount /var\n");
|
_ = logging.write("/system/services/fat: could not mount /system/configuration\n");
|
||||||
|
}
|
||||||
|
if (file_system.mountRewritten("/system/logs", endpointForMount(), "/system/logs")) {
|
||||||
|
std.log.info("mounted /system/logs", .{});
|
||||||
|
} else {
|
||||||
|
_ = logging.write("/system/services/fat: could not mount /system/logs\n");
|
||||||
}
|
}
|
||||||
mounted = true;
|
mounted = true;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -29,17 +29,17 @@ const fs = @import("file-system");
|
|||||||
const csv = @import("csv");
|
const csv = @import("csv");
|
||||||
|
|
||||||
/// The system services init brings up at boot are init's policy, not the kernel's —
|
/// The system services init brings up at boot are init's policy, not the kernel's —
|
||||||
/// and that policy is now data: `/etc/init.csv` (see `loadServices`), read at
|
/// and that policy is now data: `/system/configuration/init.csv` (see `loadServices`), read at
|
||||||
/// startup instead of a hardcoded list. Drivers are absent on purpose: the device
|
/// startup instead of a hardcoded list. Drivers are absent on purpose: the device
|
||||||
/// manager owns those.
|
/// manager owns those.
|
||||||
///
|
///
|
||||||
/// The most services `/etc/init.csv` can list, and the most argv entries (beyond the
|
/// The most services `/system/configuration/init.csv` can list, and the most argv entries (beyond the
|
||||||
/// path) each may carry. Fixed caps because init parses the list into static storage —
|
/// path) each may carry. Fixed caps because init parses the list into static storage —
|
||||||
/// the freestanding, no-allocator counterpart to the device manager's registry table.
|
/// the freestanding, no-allocator counterpart to the device manager's registry table.
|
||||||
const max_services = 16;
|
const max_services = 16;
|
||||||
const max_service_args = 4;
|
const max_service_args = 4;
|
||||||
|
|
||||||
/// One service init starts, parsed from a row of `/etc/init.csv`: its binary path
|
/// One service init starts, parsed from a row of `/system/configuration/init.csv`: its binary path
|
||||||
/// and argv, both slices into `init_csv` (held for the life of the process).
|
/// and argv, both slices into `init_csv` (held for the life of the process).
|
||||||
const Service = struct {
|
const Service = struct {
|
||||||
path: []const u8 = "",
|
path: []const u8 = "",
|
||||||
@@ -50,7 +50,7 @@ const Service = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// The `/etc/init.csv` bytes, held because the parsed services slice into them.
|
/// The `/system/configuration/init.csv` bytes, held because the parsed services slice into them.
|
||||||
var init_csv: [4096]u8 = undefined;
|
var init_csv: [4096]u8 = undefined;
|
||||||
var services: [max_services]Service = .{Service{}} ** max_services;
|
var services: [max_services]Service = .{Service{}} ** max_services;
|
||||||
var service_count: usize = 0;
|
var service_count: usize = 0;
|
||||||
@@ -65,16 +65,16 @@ var restart_counts: [max_services]u32 = .{0} ** max_services;
|
|||||||
var shutting_down = false;
|
var shutting_down = false;
|
||||||
var supervision_endpoint: ipc.Handle = 0;
|
var supervision_endpoint: ipc.Handle = 0;
|
||||||
|
|
||||||
/// Parse `/etc/init.csv` into `services`, in file order (startup order; shutdown is
|
/// Parse `/system/configuration/init.csv` into `services`, in file order (startup order; shutdown is
|
||||||
/// the reverse). Each row is a binary path followed by its argv, comma-separated;
|
/// the reverse). Each row is a binary path followed by its argv, comma-separated;
|
||||||
/// `#` comments and blank lines are ignored. The file lives in the initial ramdisk,
|
/// `#` comments and blank lines are ignored. The file lives in the initial ramdisk,
|
||||||
/// which the kernel serves directly, so init — PID 1, running before any filesystem
|
/// which the kernel serves directly, so init — PID 1, running before any filesystem
|
||||||
/// service — reads it with a plain fs.open, the same mechanism the device manager
|
/// service — reads it with a plain fs.open, the same mechanism the device manager
|
||||||
/// uses for /etc/devices.csv. A missing file means no services (the no-ramdisk
|
/// uses for /system/configuration/devices.csv. A missing file means no services (the no-ramdisk
|
||||||
/// isolation test): loud, but not fatal.
|
/// isolation test): loud, but not fatal.
|
||||||
fn loadServices() void {
|
fn loadServices() void {
|
||||||
var file = fs.open("/etc/init.csv", .{}) orelse {
|
var file = fs.open("/system/configuration/init.csv", .{}) orelse {
|
||||||
_ = logging.write("/system/services/init: /etc/init.csv missing — no services started\n");
|
_ = logging.write("/system/services/init: /system/configuration/init.csv missing — no services started\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
defer file.close();
|
defer file.close();
|
||||||
@@ -89,7 +89,7 @@ fn loadServices() void {
|
|||||||
const body = csv.stripComment(line);
|
const body = csv.stripComment(line);
|
||||||
if (body.len == 0) continue;
|
if (body.len == 0) continue;
|
||||||
if (service_count >= services.len) {
|
if (service_count >= services.len) {
|
||||||
_ = logging.write("/system/services/init: /etc/init.csv has more services than the table holds\n");
|
_ = logging.write("/system/services/init: /system/configuration/init.csv has more services than the table holds\n");
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
var it = csv.fields(body);
|
var it = csv.fields(body);
|
||||||
@@ -137,7 +137,7 @@ pub fn main() void {
|
|||||||
|
|
||||||
// Load the service list, then bring each up supervised so init can stop them
|
// Load the service list, then bring each up supervised so init can stop them
|
||||||
// cleanly. Best-effort and silent: each service announces its own readiness,
|
// cleanly. Best-effort and silent: each service announces its own readiness,
|
||||||
// and with no /etc/init.csv (an isolation test) the loop starts nothing.
|
// and with no /system/configuration/init.csv (an isolation test) the loop starts nothing.
|
||||||
loadServices();
|
loadServices();
|
||||||
for (services[0..service_count], 0..) |*service, i| {
|
for (services[0..service_count], 0..) |*service, i| {
|
||||||
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
|
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
//! demultiplexes it into **one file per process** on the flash volume:
|
//! demultiplexes it into **one file per process** on the flash volume:
|
||||||
//!
|
//!
|
||||||
//! <base>/<boot-stamp>/<binary-path>.log
|
//! <base>/<boot-stamp>/<binary-path>.log
|
||||||
//! e.g. /mnt/usb/var/log/2026-07-21T101530Z/system/services/fat.log
|
//! e.g. /volumes/usb/system/logs/2026-07-21T101530Z/system/services/fat.log
|
||||||
//!
|
//!
|
||||||
//! The boot stamp is the wall-clock time of boot (from klog_status), so one
|
//! The boot stamp is the wall-clock time of boot (from klog_status), so one
|
||||||
//! boot session is one self-contained directory; the kernel's own records go to
|
//! boot session is one self-contained directory; the kernel's own records go to
|
||||||
@@ -37,11 +37,11 @@ const time = @import("time");
|
|||||||
const logging = @import("logging");
|
const logging = @import("logging");
|
||||||
|
|
||||||
|
|
||||||
/// Where log trees live: the FHS path. The kernel VFS routes /var to whatever
|
/// Where log trees live: the hierarchy path. The kernel VFS routes /system/logs
|
||||||
/// volume the fat server mounted there (today: the /var subtree of the USB
|
/// to whatever volume the fat server mounted there (today: the /system/logs
|
||||||
/// flash volume) — swapping the persistent medium later touches fat's two
|
/// subtree of the USB flash volume) — swapping the persistent medium later
|
||||||
/// mount calls, never this constant.
|
/// touches fat's mount calls, never this constant.
|
||||||
const base = "/var/log";
|
const base = "/system/logs";
|
||||||
|
|
||||||
/// Drain cadence and the quiet period after which files are closed (flushed).
|
/// Drain cadence and the quiet period after which files are closed (flushed).
|
||||||
const tick_ms = 250;
|
const tick_ms = 250;
|
||||||
@@ -128,7 +128,7 @@ fn onTerminate() void {
|
|||||||
|
|
||||||
fn tick() void {
|
fn tick() void {
|
||||||
if (!storage_ready) {
|
if (!storage_ready) {
|
||||||
// makePath doubles as the readiness probe: while /var is unmounted the
|
// makePath doubles as the readiness probe: while /system/logs is unmounted the
|
||||||
// resolve fails fast (no storage round trip) and the ring buffers; the
|
// resolve fails fast (no storage round trip) and the ring buffers; the
|
||||||
// first success creates the whole per-boot tree.
|
// first success creates the whole per-boot tree.
|
||||||
if (!fs.makePath(boot_directory[0..boot_directory_len])) return;
|
if (!fs.makePath(boot_directory[0..boot_directory_len])) return;
|
||||||
|
|||||||
+16
-6
@@ -172,7 +172,7 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||||
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
||||||
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
||||||
@@ -208,7 +208,7 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||||
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
||||||
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
||||||
@@ -343,6 +343,16 @@ CASES = [
|
|||||||
{"name": "usermem",
|
{"name": "usermem",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# The checked copy layer (system/kernel/user-memory.zig): kernel-side unit
|
||||||
|
# checks for the bound, presence, and leaf U/S + writable refusals, then a
|
||||||
|
# fixture aiming unmapped-but-in-range pointers at klog_read/klog_status/
|
||||||
|
# process_enumerate/device_enumerate/fs_resolve/debug_write. Each must come
|
||||||
|
# back as a wrapped -errno with the machine still running — before H1 every
|
||||||
|
# one of them dereferenced the bad page in ring 0 and halted it.
|
||||||
|
{"name": "user-memory",
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"(?s)(?=.*DANOS-TEST-RESULT: PASS)(?=.*user-memory-test: ok)",
|
||||||
|
"fail": r"user-memory-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
|
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
|
||||||
# 0x5 (present|user) at the user IP. ([\s\S] spans lines; `.` doesn't.)
|
# 0x5 (present|user) at the user IP. ([\s\S] spans lines; `.` doesn't.)
|
||||||
{"name": "user-pf",
|
{"name": "user-pf",
|
||||||
@@ -623,13 +633,13 @@ CASES = [
|
|||||||
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
|
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
|
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
|
||||||
# FAT32 image) into the VFS at /mnt/usb. A fat-test client then lists and reads
|
# FAT32 image) into the VFS at /volumes/usb. A fat-test client then lists and reads
|
||||||
# through the mount — proof of the whole stack: block device -> FAT parse ->
|
# through the mount — proof of the whole stack: block device -> FAT parse ->
|
||||||
# VFS routing -> file read.
|
# VFS routing -> file read.
|
||||||
{"name": "fat-mount",
|
{"name": "fat-mount",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
"expect": r"fat: mounted /mnt/usb[\s\S]*fat-test: ok",
|
"expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
||||||
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
||||||
@@ -714,7 +724,7 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
"qmp_after": {"delay": 8, "command": "system_powerdown"},
|
"qmp_after": {"delay": 8, "command": "system_powerdown"},
|
||||||
"expect": r"logger: logging to /var/log/\d{4}-\d{2}-\d{2}T\d{6}Z[\s\S]*"
|
"expect": r"logger: logging to /system/logs/\d{4}-\d{2}-\d{2}T\d{6}Z[\s\S]*"
|
||||||
r"init: shutting down[\s\S]*"
|
r"init: shutting down[\s\S]*"
|
||||||
r"logger: flushed through sequence \d+[\s\S]*"
|
r"logger: flushed through sequence \d+[\s\S]*"
|
||||||
r"power: entering S5",
|
r"power: entering S5",
|
||||||
@@ -935,7 +945,7 @@ def run_case(arch, case):
|
|||||||
# The bootable FAT32 USB image the build produced (tools/make-fat-image.py),
|
# The bootable FAT32 USB image the build produced (tools/make-fat-image.py),
|
||||||
# presented to the guest as a usb-storage device (see qemu_args).
|
# presented to the guest as a usb-storage device (see qemu_args).
|
||||||
# Boot a per-run COPY of the image: the guest MUTATES its boot volume (the
|
# Boot a per-run COPY of the image: the guest MUTATES its boot volume (the
|
||||||
# fat tests create/delete files; the logger writes /var/log), and QEMU is
|
# fat tests create/delete files; the logger writes /system/logs), and QEMU is
|
||||||
# hard-killed after a match — booting the build artifact in place let one
|
# hard-killed after a match — booting the build artifact in place let one
|
||||||
# run's leftovers fail the next (a stale TESTDIR trips the mkdir-duplicate
|
# run's leftovers fail the next (a stale TESTDIR trips the mkdir-duplicate
|
||||||
# refusal) and dirtied the build cache's own output.
|
# refusal) and dirtied the build cache's own output.
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! test/system/services/fat-test — a client that proves the FAT mount end to end:
|
//! test/system/services/fat-test — a client that proves the FAT mount end to end:
|
||||||
//! it waits for the fat server to mount the USB volume at /mnt/usb, lists the
|
//! it waits for the fat server to mount the USB volume at /volumes/usb, lists the
|
||||||
//! root directory through the VFS (which routes /mnt/usb to the fat backend), and
|
//! root directory through the VFS (which routes /volumes/usb to the fat backend), and
|
||||||
//! reads a known file off it. Shipped in the initial_ramdisk; the `fat-mount`
|
//! reads a known file off it. Shipped in the initial_ramdisk; the `fat-mount`
|
||||||
//! kernel test spawns it alongside init.
|
//! kernel test spawns it alongside init.
|
||||||
|
|
||||||
@@ -18,16 +18,16 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
|||||||
pub fn main(init: process.Init) void {
|
pub fn main(init: process.Init) void {
|
||||||
_ = init;
|
_ = init;
|
||||||
|
|
||||||
// Wait for /mnt/usb to be mounted — the fat server races us at boot (it must
|
// Wait for /volumes/usb to be mounted — the fat server races us at boot (it must
|
||||||
// bring up the whole USB storage chain first).
|
// bring up the whole USB storage chain first).
|
||||||
var opened: ?fs.Directory = null;
|
var opened: ?fs.Directory = null;
|
||||||
var tries: u32 = 0;
|
var tries: u32 = 0;
|
||||||
while (opened == null and tries < 1400) : (tries += 1) {
|
while (opened == null and tries < 1400) : (tries += 1) {
|
||||||
opened = fs.openDirectory("/mnt/usb");
|
opened = fs.openDirectory("/volumes/usb");
|
||||||
if (opened == null) time.sleepMillis(50);
|
if (opened == null) time.sleepMillis(50);
|
||||||
}
|
}
|
||||||
var dir = opened orelse {
|
var dir = opened orelse {
|
||||||
_ = logging.write("fat-test: /mnt/usb never became available\n");
|
_ = logging.write("fat-test: /volumes/usb never became available\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -43,56 +43,56 @@ pub fn main(init: process.Init) void {
|
|||||||
|
|
||||||
// Read a known file off the boot volume through the mount (best effort): the
|
// Read a known file off the boot volume through the mount (best effort): the
|
||||||
// kernel image is an ELF, so its first bytes are the ELF magic.
|
// kernel image is an ELF, so its first bytes are the ELF magic.
|
||||||
if (fs.open("/mnt/usb/system/kernel", .{})) |opened_file| {
|
if (fs.open("/volumes/usb/system/kernel", .{})) |opened_file| {
|
||||||
var file = opened_file;
|
var file = opened_file;
|
||||||
var magic: [4]u8 = undefined;
|
var magic: [4]u8 = undefined;
|
||||||
const n = file.read(&magic) orelse 0;
|
const n = file.read(&magic) orelse 0;
|
||||||
file.close();
|
file.close();
|
||||||
if (n == 4 and magic[0] == 0x7F and magic[1] == 'E' and magic[2] == 'L' and magic[3] == 'F') {
|
if (n == 4 and magic[0] == 0x7F and magic[1] == 'E' and magic[2] == 'L' and magic[3] == 'F') {
|
||||||
_ = logging.write("fat-test: read /mnt/usb/system/kernel ELF magic ok\n");
|
_ = logging.write("fat-test: read /volumes/usb/system/kernel ELF magic ok\n");
|
||||||
} else {
|
} else {
|
||||||
writeLine("fat-test: /mnt/usb/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
writeLine("fat-test: /volumes/usb/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Exercise directory + file mutation through the mount: mkdir, create a file
|
// Exercise directory + file mutation through the mount: mkdir, create a file
|
||||||
// inside it, read it back, then remove it — proof mkdir/unlink reach the engine.
|
// inside it, read it back, then remove it — proof mkdir/unlink reach the engine.
|
||||||
if (fs.makeDirectory("/mnt/usb/TESTDIR")) {
|
if (fs.makeDirectory("/volumes/usb/TESTDIR")) {
|
||||||
var wrote = false;
|
var wrote = false;
|
||||||
if (fs.open("/mnt/usb/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
if (fs.open("/volumes/usb/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||||
var f = created;
|
var f = created;
|
||||||
wrote = (f.writeAll("mutation-ok") orelse 0) == "mutation-ok".len;
|
wrote = (f.writeAll("mutation-ok") orelse 0) == "mutation-ok".len;
|
||||||
f.close();
|
f.close();
|
||||||
}
|
}
|
||||||
// The created file carries a real modification time (stamped from the RTC).
|
// The created file carries a real modification time (stamped from the RTC).
|
||||||
var mtime_ok = false;
|
var mtime_ok = false;
|
||||||
if (fs.attributes("/mnt/usb/TESTDIR/HELLO.TXT")) |attrs| {
|
if (fs.attributes("/volumes/usb/TESTDIR/HELLO.TXT")) |attrs| {
|
||||||
writeLine("fat-test: mtime {d}\n", .{attrs.mtime});
|
writeLine("fat-test: mtime {d}\n", .{attrs.mtime});
|
||||||
mtime_ok = attrs.mtime > 1_577_836_800; // after 2020-01-01
|
mtime_ok = attrs.mtime > 1_577_836_800; // after 2020-01-01
|
||||||
}
|
}
|
||||||
if (mtime_ok) _ = logging.write("fat-test: mtime ok\n");
|
if (mtime_ok) _ = logging.write("fat-test: mtime ok\n");
|
||||||
|
|
||||||
// Rename it, then read from the new name and confirm the old name is gone.
|
// Rename it, then read from the new name and confirm the old name is gone.
|
||||||
const renamed = fs.rename("/mnt/usb/TESTDIR/HELLO.TXT", "/mnt/usb/TESTDIR/RENAMED.TXT");
|
const renamed = fs.rename("/volumes/usb/TESTDIR/HELLO.TXT", "/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||||
const old_gone = !fs.exists("/mnt/usb/TESTDIR/HELLO.TXT");
|
const old_gone = !fs.exists("/volumes/usb/TESTDIR/HELLO.TXT");
|
||||||
if (renamed and old_gone) _ = logging.write("fat-test: rename ok\n");
|
if (renamed and old_gone) _ = logging.write("fat-test: rename ok\n");
|
||||||
var readback = false;
|
var readback = false;
|
||||||
if (fs.open("/mnt/usb/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
if (fs.open("/volumes/usb/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||||
var f = reopened;
|
var f = reopened;
|
||||||
var buf: [16]u8 = undefined;
|
var buf: [16]u8 = undefined;
|
||||||
const got = f.read(&buf) orelse 0;
|
const got = f.read(&buf) orelse 0;
|
||||||
f.close();
|
f.close();
|
||||||
readback = std.mem.eql(u8, buf[0..got], "mutation-ok");
|
readback = std.mem.eql(u8, buf[0..got], "mutation-ok");
|
||||||
}
|
}
|
||||||
const removed = fs.remove("/mnt/usb/TESTDIR/RENAMED.TXT");
|
const removed = fs.remove("/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||||
const gone = !fs.exists("/mnt/usb/TESTDIR/RENAMED.TXT");
|
const gone = !fs.exists("/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||||
if (wrote and mtime_ok and renamed and old_gone and readback and removed and gone) {
|
if (wrote and mtime_ok and renamed and old_gone and readback and removed and gone) {
|
||||||
_ = logging.write("fat-test: mutations ok\n");
|
_ = logging.write("fat-test: mutations ok\n");
|
||||||
} else {
|
} else {
|
||||||
writeLine("fat-test: mutations FAILED (wrote={} mtime={} renamed={} oldgone={} read={} removed={} gone={})\n", .{ wrote, mtime_ok, renamed, old_gone, readback, removed, gone });
|
writeLine("fat-test: mutations FAILED (wrote={} mtime={} renamed={} oldgone={} read={} removed={} gone={})\n", .{ wrote, mtime_ok, renamed, old_gone, readback, removed, gone });
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
_ = logging.write("fat-test: mkdir /mnt/usb/TESTDIR failed\n");
|
_ = logging.write("fat-test: mkdir /volumes/usb/TESTDIR failed\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
if (count > 0) {
|
if (count > 0) {
|
||||||
|
|||||||
@@ -0,0 +1,15 @@
|
|||||||
|
//! The user-memory-test fixture as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "user-memory-test",
|
||||||
|
.root_source_file = b.path("user-memory-test.zig"),
|
||||||
|
.imports = &.{ "abi", "system-call", "logging", "driver" },
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
}
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
.{
|
||||||
|
.name = .user_memory_test,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0xde57dc97fd81a402, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../../library/kernel" },
|
||||||
|
.device = .{ .path = "../../../../library/device" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -0,0 +1,111 @@
|
|||||||
|
//! user-memory-test — QEMU fixture for the kernel's checked copy layer
|
||||||
|
//! (system/kernel/user-memory.zig).
|
||||||
|
//!
|
||||||
|
//! Every check comes in a pair: the same system call is made once with a sound
|
||||||
|
//! buffer and once with a pointer that is *inside* the user half but mapped in no
|
||||||
|
//! process. The sound call must succeed and the unsound one must return a wrapped
|
||||||
|
//! -errno — proving the refusal is attributable to the pointer and not to the
|
||||||
|
//! call being impossible. Before H1 the unsound half of each pair dereferenced an
|
||||||
|
//! unmapped page in ring 0, which halts the machine; the fixture reaching its
|
||||||
|
//! final marker at all is the substance of the test.
|
||||||
|
//!
|
||||||
|
//! Prints `user-memory-test: <check> ok` per pair and `user-memory-test: ok` at
|
||||||
|
//! the end, which the harness asserts on (test/qemu_test.py).
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call");
|
||||||
|
const logging = @import("logging");
|
||||||
|
const device = @import("driver");
|
||||||
|
|
||||||
|
/// A user-half address that is mapped in no process: below the code image
|
||||||
|
/// (0x0000_7000_0000_0000) and far from every arena the kernel hands out. It
|
||||||
|
/// passes each system call's bounds check and then fails the page walk — the
|
||||||
|
/// exact shape the checked copy exists for.
|
||||||
|
const unmapped: usize = 0x0000_6000_0000_0000;
|
||||||
|
|
||||||
|
/// The kernel returns failures as a small negative errno in the result register.
|
||||||
|
fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
var failures: u32 = 0;
|
||||||
|
|
||||||
|
fn check(comptime name: []const u8, ok: bool) void {
|
||||||
|
if (ok) {
|
||||||
|
_ = logging.write("user-memory-test: " ++ name ++ " ok\n");
|
||||||
|
} else {
|
||||||
|
failures += 1;
|
||||||
|
_ = logging.write("user-memory-test: FAIL " ++ name ++ "\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// klog_read writes the ring's bytes out to the caller. Read from the ring's own
|
||||||
|
/// tail so the offset is certainly valid and the copy is what decides the call.
|
||||||
|
fn klogRead() void {
|
||||||
|
var status: abi.KlogStatus = undefined;
|
||||||
|
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) {
|
||||||
|
check("klog_status (sound out buffer accepted)", false);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
var buffer: [64]u8 = undefined;
|
||||||
|
const good = sc.systemCall3(.klog_read, status.tail, @intFromPtr(&buffer), buffer.len);
|
||||||
|
const bad = sc.systemCall3(.klog_read, status.tail, unmapped, buffer.len);
|
||||||
|
check("klog_read (bad out buffer refused)", !failed(good) and failed(bad));
|
||||||
|
|
||||||
|
// The status struct itself is a write-direction copy of its own.
|
||||||
|
const bad_status = sc.systemCall1(.klog_status, unmapped);
|
||||||
|
check("klog_status (bad out buffer refused)", failed(bad_status));
|
||||||
|
}
|
||||||
|
|
||||||
|
fn processEnumerate() void {
|
||||||
|
var table: [4]abi.ProcessDescriptor = undefined;
|
||||||
|
const good = sc.systemCall2(.process_enumerate, @intFromPtr(&table), table.len);
|
||||||
|
const bad = sc.systemCall2(.process_enumerate, unmapped, table.len);
|
||||||
|
check("process_enumerate (bad out buffer refused)", good != 0 and !failed(good) and failed(bad));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Descriptors are hundreds of bytes each — keep them off the stack. Four spans
|
||||||
|
/// more than one kernel chunk, so the chunked copy-out is genuinely exercised.
|
||||||
|
var descriptors: [4]device.DeviceDescriptor = undefined;
|
||||||
|
|
||||||
|
fn deviceEnumerate() void {
|
||||||
|
const good = sc.systemCall2(.device_enumerate, @intFromPtr(&descriptors), descriptors.len);
|
||||||
|
const bad = sc.systemCall2(.device_enumerate, unmapped, descriptors.len);
|
||||||
|
check("device_enumerate (bad out buffer refused)", !failed(good) and failed(bad));
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fsResolve() void {
|
||||||
|
var out: [256]u8 = undefined;
|
||||||
|
const path = "/system/services/init";
|
||||||
|
const good = sc.systemCall5(.fs_resolve, @intFromPtr(path.ptr), path.len, 0, @intFromPtr(&out), out.len);
|
||||||
|
const bad = sc.systemCall5(.fs_resolve, unmapped, path.len, 0, @intFromPtr(&out), out.len);
|
||||||
|
check("fs_resolve (bad path buffer refused)", !failed(good) and failed(bad));
|
||||||
|
|
||||||
|
// The out-buffer capacity is unbounded ring-3 input: the range check must
|
||||||
|
// not add it to the base, or the sum wraps the user-half bound and traps
|
||||||
|
// the kernel's own overflow check. Surviving this call is the assertion.
|
||||||
|
const wrapping = sc.systemCall5(.fs_resolve, @intFromPtr(path.ptr), path.len, 0, @intFromPtr(&out), ~@as(usize, 0));
|
||||||
|
check("fs_resolve (wrapping out capacity refused)", failed(wrapping));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// debug_write reads the caller's message; a bad pointer must not take the
|
||||||
|
/// kernel down on the way to the log ring.
|
||||||
|
fn debugWrite() void {
|
||||||
|
const bad = sc.systemCall3(.debug_write, unmapped, 16, @intFromEnum(abi.KlogLevel.info));
|
||||||
|
check("debug_write (bad message buffer refused)", failed(bad));
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main() void {
|
||||||
|
klogRead();
|
||||||
|
processEnumerate();
|
||||||
|
deviceEnumerate();
|
||||||
|
fsResolve();
|
||||||
|
debugWrite();
|
||||||
|
// Reaching here at all means seven bad user pointers failed their calls
|
||||||
|
// instead of faulting ring 0.
|
||||||
|
if (failures == 0) {
|
||||||
|
_ = logging.write("user-memory-test: ok\n");
|
||||||
|
} else {
|
||||||
|
_ = logging.write("user-memory-test: FAILED\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -77,7 +77,7 @@ fn park() void {
|
|||||||
var parked: ?fs.File = null;
|
var parked: ?fs.File = null;
|
||||||
var tries: u32 = 0;
|
var tries: u32 = 0;
|
||||||
while (parked == null and tries < 1000) : (tries += 1) {
|
while (parked == null and tries < 1000) : (tries += 1) {
|
||||||
parked = fs.open("/mnt/usb/parked", .{ .create = true });
|
parked = fs.open("/volumes/usb/parked", .{ .create = true });
|
||||||
if (parked == null) time.sleepMillis(20);
|
if (parked == null) time.sleepMillis(20);
|
||||||
}
|
}
|
||||||
if (parked == null) {
|
if (parked == null) {
|
||||||
|
|||||||
Reference in New Issue
Block a user