From 62eb2a748ac1383ef63f5bc06ed4c519ac424a82 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Mon, 10 Aug 2026 03:02:58 +0100 Subject: [PATCH] exfat: the engine read path (S4 step 2) mount + resolve + list + read over a BlockDevice, host-tested against a RAM-backed image the tests build with formatExfat. mount reads the VBR, loads the geometry, and scans the root for the Allocation Bitmap (0x81) and Up-case Table (0x82); the up-case prefix is decompressed (0xFFFF identity runs) into a bounded table so names fold correctly. A directory is read as consecutive 32-byte entries via readChain, so a File/Stream/Name SET that straddles a sector or cluster boundary assembles cleanly; each set's checksum is validated before it counts as a file. A stream's no_fat_chain flag picks contiguous-arithmetic vs FAT-follow cluster walking. reads honor valid_data_length (allocated- but-unwritten tail reads as zero) and clamp data_length to the vfs u32 offset surface. Tests cover mount + up-case fold, listing (skipping the metadata entries), case-insensitive resolve, and reads across a cluster boundary on both a contiguous and a FAT-fragmented file. Wired via engine.zig (which imports on-disk.zig) into the host-test aggregate. 12/12, bounds green. Write path is step 3. --- build.zig | 7 +- system/services/exfat/engine.zig | 730 +++++++++++++++++++++++++++++++ 2 files changed, 734 insertions(+), 3 deletions(-) create mode 100644 system/services/exfat/engine.zig diff --git a/build.zig b/build.zig index 19680c1..0c50026 100644 --- a/build.zig +++ b/build.zig @@ -422,9 +422,10 @@ pub fn build(b: *std.Build) void { "system/boot-handoff.zig", "system/abi.zig", "system/initial-ramdisk.zig", // v2 path-named entries: find/basename/magic - // The exFAT engine's pure on-disk layer (S4), std-only, before its package - // exists — moves into the `exfat` package's own test step once that lands. - "system/services/exfat/on-disk.zig", + // The exFAT engine (S4), std-only, before its package exists — the + // engine imports on-disk.zig, so this one entry runs both files' tests. + // Moves into the `exfat` package's own test step once that lands (step 4). + "system/services/exfat/engine.zig", }) |root| { const mod_tests = b.addTest(.{ .root_module = b.createModule(.{ diff --git a/system/services/exfat/engine.zig b/system/services/exfat/engine.zig new file mode 100644 index 0000000..da0ec56 --- /dev/null +++ b/system/services/exfat/engine.zig @@ -0,0 +1,730 @@ +//! The exFAT filesystem engine: mount a block device, walk the allocation +//! structures and directory entry sets, and (this step) resolve, list, and read. +//! Pure logic over a `BlockDevice` interface — no IPC — host-testable against a +//! RAM-backed image (the tests at the bottom build one with `formatExfat`). The +//! exfat.zig server wraps a real `.block` device and serves this over the VFS +//! protocol through the shared filesystem harness, exactly as fat.zig does. +//! +//! Three things differ from FAT and shape this file. A file is a directory-entry +//! SET — a File entry (0x85), a Stream Extension (0xC0), then File Name entries +//! (0xC1) — read as consecutive 32-byte entries that may cross sector and cluster +//! boundaries. A stream's `no_fat_chain` flag says its clusters are contiguous +//! (walk by arithmetic) or fragmented (follow the 32-bit FAT). And names are +//! matched case-folded through the volume's own on-disk up-case table, a bounded +//! prefix of which is loaded at mount. Allocation authority (the bitmap) is the +//! write path's concern (step 3); reads never touch it. +//! +//! Everything works in 512-byte sectors; a cluster is N sectors. + +const std = @import("std"); +const on_disk = @import("on-disk.zig"); + +const sector_size = 512; +const name_units_per_entry = on_disk.name_units_per_entry; // 15 + +/// bound: UTF-16 units of a file name (the longest name a listing/resolve handles) +/// decided-by: external +/// protects: the on-stack name buffers and the Listing name_buffer +/// at-limit: truncate - the exFAT format caps a name at 255 units, so a longer +/// name is impossible on a valid volume; a corrupt over-long name is cut and the +/// set's checksum mismatch (checked on read) flags it +/// observed-by: a set-checksum rejection in scanDirectory +const name_maximum = 255; + +/// bound: code points whose case-fold the engine loads from the up-case table +/// decided-by: ours +/// protects: the in-memory `upcase` table +/// at-limit: truncate - code points at or above this fold to themselves, so two +/// names differing only in case ABOVE this point compare as distinct (danos +/// names are ASCII, far below it) +/// observed-by: a case-only-different high-plane name resolving as not-found +const upcase_fold_limit = 256; + +/// bound: 32-byte entries one directory scan will read before giving up +/// decided-by: ours +/// protects: scanDirectory / mount against a directory with no end marker or a +/// cyclic cluster chain (a corrupt medium) +/// at-limit: truncate - entries beyond are not listed or resolved; a real +/// directory is far smaller, so hitting this means corruption +/// observed-by: an on-disk file absent from a listing under a huge directory +const directory_entry_scan_maximum = 65536; + +/// The most sectors one multi-sector transfer moves (the DMA bounce the exfat +/// server sizes to). The engine's read path works a sector at a time, so this only +/// bounds the server's buffer; kept for parity with the fat engine. +/// bound: sectors in one coalesced device transfer +/// decided-by: ours +/// protects: the exfat server's DMA bounce buffer (max_transfer_sectors * 512) +/// at-limit: truncate - a longer run is split into several transfers, no data lost +/// observed-by: more device commands than the ideal, never a wrong byte +pub const max_transfer_sectors = 8; + +const block_cache_lines = 16; + +/// A block device the engine reads and writes in fixed-size blocks (identical to +/// the fat engine's — the shared harness is generic over whichever engine wraps a +/// real `.block` driver or, in tests, a RAM buffer). +pub const BlockDevice = struct { + context: *anyopaque, + block_size: u32, + block_count: u64, + readBlocksFn: *const fn (context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool, + writeBlocksFn: *const fn (context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool, + + pub fn readBlocks(self: BlockDevice, lba: u64, count: u32, buffer: []u8) bool { + return self.readBlocksFn(self.context, lba, count, buffer); + } + pub fn writeBlocks(self: BlockDevice, lba: u64, count: u32, buffer: []const u8) bool { + return self.writeBlocksFn(self.context, lba, count, buffer); + } + pub fn readBlock(self: BlockDevice, lba: u64, buffer: []u8) bool { + return self.readBlocksFn(self.context, lba, 1, buffer); + } + pub fn writeBlock(self: BlockDevice, lba: u64, buffer: []const u8) bool { + return self.writeBlocksFn(self.context, lba, 1, buffer); + } +}; + +/// A resolved filesystem object and where its directory-entry set lives, so writes +/// (step 3) can rewrite the Stream entry's sizes/first-cluster and recompute the +/// set checksum. +pub const Node = struct { + first_cluster: u32, + size: u32, // data_length, clamped to the vfs u32 offset surface + is_directory: bool, + no_fat_chain: bool = false, + valid_data_length: u32 = 0, // bytes actually written; [valid, size) read as zero + mtime: u64 = 0, + // The set's home: its parent directory's chain and the File entry's linear + // 32-byte-entry index within it, plus how many secondary entries follow. + parent_first_cluster: u32 = 0, + parent_no_fat_chain: bool = false, + entry_index: u64 = 0, + secondary_count: u8 = 0, + has_entry: bool = false, +}; + +pub const Listing = struct { + name_buffer: [name_maximum]u8 = undefined, + name_len: usize = 0, + is_directory: bool = false, + size: u32 = 0, + mtime: u64 = 0, +}; + +const CacheLine = struct { + lba: u64 = 0, + valid: bool = false, + data: [sector_size]u8 = undefined, +}; + +pub const FileSystem = struct { + device: BlockDevice, + geometry: on_disk.Geometry, + base_lba: u64 = 0, // the volume manager confines the channel volume-relative + sector: [sector_size]u8 = undefined, + cache: [block_cache_lines]CacheLine = [_]CacheLine{.{}} ** block_cache_lines, + cache_cursor: u32 = 0, + // The case-fold table, a bounded prefix loaded at mount (unit -> uppercase). + upcase: [upcase_fold_limit]u16 = undefined, + upcase_len: usize = 0, + // The allocation bitmap's location, for the write path (step 3). + bitmap_first_cluster: u32 = 0, + bitmap_length: u64 = 0, // bytes + // Wall-clock (Unix epoch seconds) the server sets before a mutating op. + current_time_epoch: u64 = 0, + + // --- single-sector write-through cache ---------------------------------- + + fn cacheFind(self: *FileSystem, lba: u64) ?*CacheLine { + for (&self.cache) |*line| if (line.valid and line.lba == lba) return line; + return null; + } + fn cacheInstall(self: *FileSystem, lba: u64, data: []const u8) void { + const line = self.cacheFind(lba) orelse blk: { + const slot = &self.cache[self.cache_cursor]; + self.cache_cursor = (self.cache_cursor + 1) % block_cache_lines; + slot.valid = true; + slot.lba = lba; + break :blk slot; + }; + @memcpy(&line.data, data[0..sector_size]); + } + fn cacheInvalidateRange(self: *FileSystem, lba: u64, count: u32) void { + for (&self.cache) |*line| { + if (line.valid and line.lba >= lba and line.lba < lba + count) line.valid = false; + } + } + + fn blockRead(self: *FileSystem, lba: u64, buffer: []u8) bool { + if (self.cacheFind(lba)) |line| { + @memcpy(buffer[0..sector_size], &line.data); + return true; + } + if (!self.device.readBlock(self.base_lba + lba, buffer)) return false; + self.cacheInstall(lba, buffer); + return true; + } + fn blockWrite(self: *FileSystem, lba: u64, buffer: []const u8) bool { + if (!self.device.writeBlock(self.base_lba + lba, buffer)) return false; + self.cacheInstall(lba, buffer); + return true; + } + + // --- cluster <-> sector, and the FAT ------------------------------------ + + fn clusterBytes(self: *const FileSystem) u32 { + return self.geometry.sectors_per_cluster * sector_size; + } + + fn clusterSector(self: *const FileSystem, cluster: u32, sector_in_cluster: u32) u64 { + return @as(u64, self.geometry.cluster_heap_offset_sectors) + + @as(u64, cluster - 2) * self.geometry.sectors_per_cluster + sector_in_cluster; + } + + fn validCluster(self: *const FileSystem, cluster: u32) bool { + return cluster >= on_disk.first_data_cluster and cluster < self.geometry.cluster_count + on_disk.first_data_cluster; + } + + /// Read the 32-bit FAT entry for `cluster` (exFAT's FAT is only consulted for a + /// fragmented chain — a stream with `no_fat_chain` clear, or the metadata). + fn readFatEntry(self: *FileSystem, cluster: u32) u32 { + const byte = @as(u64, self.geometry.fat_offset_sectors) * sector_size + @as(u64, cluster) * 4; + const lba = byte / sector_size; + const within: usize = @intCast(byte % sector_size); + if (!self.blockRead(lba, &self.sector)) return on_disk.end_of_chain; + return std.mem.readInt(u32, self.sector[within..][0..4], .little); + } + + fn isEndOfChain(_: *const FileSystem, value: u32) bool { + return value >= 0xFFFFFFF8; // EOC (0xFFFFFFFF) or bad (0xFFFFFFF7) and up + } + + /// The `index`-th cluster of a chain: contiguous arithmetic when `no_fat_chain`, + /// else a FAT walk. Null past the end or off a corrupt/out-of-range link. + fn clusterOfChain(self: *FileSystem, first: u32, no_fat_chain: bool, index: u32) ?u32 { + if (!self.validCluster(first)) return null; + if (no_fat_chain) { + const cluster = first + index; + return if (self.validCluster(cluster)) cluster else null; + } + var cluster = first; + var steps = index; + while (steps > 0) : (steps -= 1) { + const next = self.readFatEntry(cluster); + if (self.isEndOfChain(next) or !self.validCluster(next)) return null; + cluster = next; + } + return cluster; + } + + /// Read up to `out.len` bytes of a cluster chain starting at `byte_offset`, + /// a sector at a time. Returns bytes produced (short at the chain's end). + fn readChain(self: *FileSystem, first: u32, no_fat_chain: bool, byte_offset: u64, out: []u8) usize { + const cluster_bytes = self.clusterBytes(); + var produced: usize = 0; + var position = byte_offset; + while (produced < out.len) { + const cluster = self.clusterOfChain(first, no_fat_chain, @intCast(position / cluster_bytes)) orelse break; + const in_cluster: u32 = @intCast(position % cluster_bytes); + const lba = self.clusterSector(cluster, in_cluster / sector_size); + const in_sector = in_cluster % sector_size; + if (!self.blockRead(lba, &self.sector)) break; + const n = @min(out.len - produced, sector_size - in_sector); + @memcpy(out[produced .. produced + n], self.sector[in_sector .. in_sector + n]); + produced += n; + position += n; + } + return produced; + } + + /// Read the `index`-th 32-byte directory entry (an entry never straddles a + /// sector: 512/32 divides evenly and entries are 32-aligned). + fn entryAt(self: *FileSystem, dir_first: u32, dir_no_fat_chain: bool, index: u64, out: *[on_disk.entry_bytes]u8) bool { + return self.readChain(dir_first, dir_no_fat_chain, index * on_disk.entry_bytes, out) == on_disk.entry_bytes; + } + + // --- the up-case table -------------------------------------------------- + + fn fold(self: *const FileSystem, unit: u16) u16 { + return if (unit < self.upcase_len) self.upcase[unit] else unit; + } + + /// Load a bounded prefix of the on-disk up-case table (unit -> uppercase), + /// decompressing 0xFFFF identity runs. Any prefix entry the table does not + /// reach folds to itself. + fn loadUpcase(self: *FileSystem, first_cluster: u32, data_length: u64) void { + var raw: [upcase_fold_limit * 2]u8 = undefined; + const want: usize = @intCast(@min(data_length, @as(u64, raw.len))); + const got = self.readChain(first_cluster, false, 0, raw[0..want]); + var out_index: usize = 0; + var in_index: usize = 0; + while (out_index < upcase_fold_limit and in_index + 2 <= got) { + const value = std.mem.readInt(u16, raw[in_index..][0..2], .little); + in_index += 2; + if (value == 0xFFFF and in_index + 2 <= got) { + var run = std.mem.readInt(u16, raw[in_index..][0..2], .little); + in_index += 2; + while (run > 0 and out_index < upcase_fold_limit) : (run -= 1) { + self.upcase[out_index] = @intCast(out_index); + out_index += 1; + } + } else { + self.upcase[out_index] = value; + out_index += 1; + } + } + while (out_index < upcase_fold_limit) : (out_index += 1) self.upcase[out_index] = @intCast(out_index); + self.upcase_len = upcase_fold_limit; + } + + // --- mount -------------------------------------------------------------- + + /// Mount an exFAT volume: read the VBR, load the geometry, then scan the root + /// directory for the Allocation Bitmap (0x81) and Up-case Table (0x82). Null + /// unless it is a 512-byte-sector exFAT volume. + pub fn mount(device: BlockDevice) ?FileSystem { + var boot: [sector_size]u8 = undefined; + if (!device.readBlock(0, &boot)) return null; + const geometry = on_disk.geometryOf(&boot) orelse return null; + if (geometry.bytes_per_sector != sector_size) return null; // danos handles 512-byte sectors + + var fs = FileSystem{ .device = device, .geometry = geometry }; + // Identity fold until the table loads, so a volume with no up-case entry + // still matches ASCII case-insensitively is NOT assumed — we only fold what + // the table gives; unloaded means case-sensitive. loadUpcase fills it. + fs.upcase_len = 0; + + var index: u64 = 0; + var found_upcase = false; + while (index < directory_entry_scan_maximum) : (index += 1) { + var raw: [on_disk.entry_bytes]u8 = undefined; + if (!fs.entryAt(geometry.first_cluster_of_root, false, index, &raw)) break; + switch (raw[0]) { + on_disk.entry_type_end_of_directory => break, + on_disk.entry_type_allocation_bitmap => { + const entry = std.mem.bytesToValue(on_disk.AllocationBitmapEntry, &raw); + if (fs.bitmap_first_cluster == 0) { // the first (active, flags bit0=0) bitmap + fs.bitmap_first_cluster = entry.first_cluster; + fs.bitmap_length = entry.data_length; + } + }, + on_disk.entry_type_upcase_table => { + if (!found_upcase) { + const entry = std.mem.bytesToValue(on_disk.UpcaseTableEntry, &raw); + fs.loadUpcase(entry.first_cluster, entry.data_length); + found_upcase = true; + } + }, + else => {}, + } + } + return fs; + } + + pub fn rootNode(self: *const FileSystem) Node { + return .{ + .first_cluster = self.geometry.first_cluster_of_root, + .size = 0, + .is_directory = true, + .no_fat_chain = false, // the root's chain is followed via the FAT + .has_entry = false, + }; + } + + // --- directory scan (entry-set assembly) -------------------------------- + + fn nameMatches(self: *const FileSystem, display: []const u8, query: []const u8) bool { + if (display.len != query.len) return false; + for (display, query) |a, b| { + if (self.fold(a) != self.fold(b)) return false; + } + return true; + } + + /// Assemble each File-entry SET in `dir` and hand it to `visit`. Stops when + /// `visit` returns true, the directory ends, or the scan cap is hit. + fn scanDirectory( + self: *FileSystem, + dir: Node, + context: anytype, + comptime visit: fn (@TypeOf(context), node: Node, name: []const u8) bool, + ) void { + var index: u64 = 0; + while (index < directory_entry_scan_maximum) { + var raw: [on_disk.entry_bytes]u8 = undefined; + if (!self.entryAt(dir.first_cluster, dir.no_fat_chain, index, &raw)) return; + if (raw[0] == on_disk.entry_type_end_of_directory) return; + if (raw[0] != on_disk.entry_type_file) { + index += 1; + continue; + } + const file = std.mem.bytesToValue(on_disk.FileEntry, &raw); + + var stream_raw: [on_disk.entry_bytes]u8 = undefined; + if (!self.entryAt(dir.first_cluster, dir.no_fat_chain, index + 1, &stream_raw)) return; + if (stream_raw[0] != on_disk.entry_type_stream_extension) { + index += 1; + continue; // a File entry without its Stream — malformed, skip + } + const stream = std.mem.bytesToValue(on_disk.StreamExtensionEntry, &stream_raw); + + // Reconstruct the name from the File Name entries, and validate the + // whole set's checksum as we go (over File + Stream + Name bytes). + var checksum_bytes: [on_disk.entry_bytes * (1 + 255)]u8 = undefined; + @memcpy(checksum_bytes[0..on_disk.entry_bytes], &raw); + @memcpy(checksum_bytes[on_disk.entry_bytes .. on_disk.entry_bytes * 2], &stream_raw); + var set_bytes: usize = on_disk.entry_bytes * 2; + + var name_buf: [name_maximum]u8 = undefined; + var name_len: usize = 0; + const name_entries = (@as(usize, stream.name_length) + name_units_per_entry - 1) / name_units_per_entry; + var e: usize = 0; + var malformed = false; + while (e < name_entries) : (e += 1) { + var fn_raw: [on_disk.entry_bytes]u8 = undefined; + if (!self.entryAt(dir.first_cluster, dir.no_fat_chain, index + 2 + e, &fn_raw)) return; + if (fn_raw[0] != on_disk.entry_type_file_name) { + malformed = true; + break; + } + @memcpy(checksum_bytes[set_bytes .. set_bytes + on_disk.entry_bytes], &fn_raw); + set_bytes += on_disk.entry_bytes; + const name_entry = std.mem.bytesToValue(on_disk.FileNameEntry, &fn_raw); + for (name_entry.file_name) |unit| { + if (name_len >= stream.name_length) break; + if (name_len < name_buf.len) name_buf[name_len] = if (unit < 0x80) @truncate(unit) else '?'; + name_len += 1; + } + } + const advance = @as(u64, file.secondary_count) + 1; + if (malformed or on_disk.setChecksum(checksum_bytes[0..set_bytes]) != file.set_checksum) { + index += advance; + continue; // a set that does not check out is not a file + } + + const node = Node{ + .first_cluster = stream.first_cluster, + .size = clampU32(stream.data_length), + .is_directory = file.file_attributes & on_disk.attribute_directory != 0, + .no_fat_chain = stream.general_secondary_flags & on_disk.secondary_flag_no_fat_chain != 0, + .valid_data_length = clampU32(stream.valid_data_length), + .mtime = on_disk.timestampToEpoch(file.last_modified_timestamp), + .parent_first_cluster = dir.first_cluster, + .parent_no_fat_chain = dir.no_fat_chain, + .entry_index = index, + .secondary_count = file.secondary_count, + .has_entry = true, + }; + if (visit(context, node, name_buf[0..@min(name_len, name_buf.len)])) return; + index += advance; + } + } + + const FindResult = struct { found: bool = false, node: Node = undefined }; + const FindContext = struct { fs: *FileSystem, query: []const u8, result: *FindResult }; + + fn findVisit(context: *FindContext, node: Node, name: []const u8) bool { + if (!context.fs.nameMatches(name, context.query)) return false; + context.result.* = .{ .found = true, .node = node }; + return true; + } + + fn findChild(self: *FileSystem, dir: Node, name: []const u8) ?Node { + var result = FindResult{}; + var context = FindContext{ .fs = self, .query = name, .result = &result }; + self.scanDirectory(dir, &context, findVisit); + return if (result.found) result.node else null; + } + + /// Resolve an absolute or "/"-relative path to a node. "/" is the root. + pub fn resolve(self: *FileSystem, path: []const u8) ?Node { + var node = self.rootNode(); + var it = std.mem.tokenizeScalar(u8, path, '/'); + while (it.next()) |component| { + if (component.len == 0) continue; + if (!node.is_directory) return null; + node = self.findChild(node, component) orelse return null; + } + return node; + } + + const ListContext = struct { target: u32, index: u32 = 0, out: *Listing, done: bool = false }; + + fn listVisit(context: *ListContext, node: Node, name: []const u8) bool { + if (context.index == context.target) { + const n = @min(name.len, context.out.name_buffer.len); + @memcpy(context.out.name_buffer[0..n], name[0..n]); + context.out.name_len = n; + context.out.is_directory = node.is_directory; + context.out.size = node.size; + context.out.mtime = node.mtime; + context.done = true; + return true; + } + context.index += 1; + return false; + } + + /// The `cursor`th entry of a directory (for readdir): name, kind, size, mtime. + pub fn listEntry(self: *FileSystem, dir: Node, cursor: u32) ?Listing { + var listing = Listing{}; + var context = ListContext{ .target = cursor, .out = &listing }; + self.scanDirectory(dir, &context, listVisit); + return if (context.done) listing else null; + } + + // --- read --------------------------------------------------------------- + + /// Read up to `buffer.len` bytes of `node` from `offset`. Bytes at or past the + /// stream's valid-data-length read as zero even though clusters are allocated. + pub fn readFile(self: *FileSystem, node: Node, offset: u32, buffer: []u8) usize { + if (offset >= node.size or !self.validCluster(node.first_cluster)) return 0; + const want = @min(buffer.len, node.size - offset); + const got = self.readChain(node.first_cluster, node.no_fat_chain, offset, buffer[0..want]); + // Zero the region beyond valid_data_length (allocated but never written). + if (offset + got > node.valid_data_length) { + const zero_from: usize = if (offset >= node.valid_data_length) 0 else node.valid_data_length - offset; + if (zero_from < got) @memset(buffer[zero_from..got], 0); + } + return got; + } +}; + +fn clampU32(value: u64) u32 { + return @intCast(@min(value, @as(u64, std.math.maxInt(u32)))); +} + +// --- tests: a RAM-backed exFAT image ----------------------------------------- + +const RamDisk = struct { + bytes: []u8, + fn readBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool { + const self: *RamDisk = @ptrCast(@alignCast(context)); + const len = @as(usize, count) * sector_size; + const start = lba * sector_size; + if (start + len > self.bytes.len or buffer.len < len) return false; + @memcpy(buffer[0..len], self.bytes[start .. start + len]); + return true; + } + fn writeBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool { + const self: *RamDisk = @ptrCast(@alignCast(context)); + const len = @as(usize, count) * sector_size; + const start = lba * sector_size; + if (start + len > self.bytes.len or buffer.len < len) return false; + @memcpy(self.bytes[start .. start + len], buffer[0..len]); + return true; + } + fn device(self: *RamDisk) BlockDevice { + return .{ + .context = self, + .block_size = sector_size, + .block_count = self.bytes.len / sector_size, + .readBlocksFn = readBlocks, + .writeBlocksFn = writeBlocks, + }; + } +}; + +// A minimal exFAT layout for the tests: 512-byte clusters (spc=1), FAT at sector +// 8, cluster heap at sector 9. Cluster 2 = allocation bitmap, 3 = up-case table, +// 4 = root. Two files: HELLO (contiguous, clusters 5-6) and SPLIT (fragmented, +// clusters 7,9 linked through the FAT), each 1024 bytes (two clusters, so a read +// crosses a cluster boundary). +const test_clusters = 64; +const test_fat_sector = 8; +const test_heap_sector = 9; +const test_read_bytes = 1024; // the two-cluster test files are 1024 bytes + +fn testCluster(cluster: u32) usize { + return (test_heap_sector + (cluster - 2)) * sector_size; +} + +fn writeFatEntry(bytes: []u8, cluster: u32, value: u32) void { + std.mem.writeInt(u32, bytes[test_fat_sector * sector_size + cluster * 4 ..][0..4], value, .little); +} + +fn asciiUpper(c: u8) u16 { + return if (c >= 'a' and c <= 'z') c - 'a' + 'A' else c; +} + +// Lay a File+Stream+Name set into the root at `entry_index`, naming `name` and +// pointing at `first_cluster`/`length`, contiguous or FAT-linked. +fn writeFileSet(bytes: []u8, entry_index: usize, name: []const u8, first_cluster: u32, length: u64, no_fat_chain: bool) void { + const root = testCluster(4); + var set: [on_disk.entry_bytes * 3]u8 = [_]u8{0} ** (on_disk.entry_bytes * 3); + // File entry (0x85). + set[0] = on_disk.entry_type_file; + set[1] = 2; // stream + one name entry (names <= 15 units) + std.mem.writeInt(u16, set[4..6], on_disk.attribute_archive, .little); + // Stream entry (0xC0). + const s = on_disk.entry_bytes; + set[s + 0] = on_disk.entry_type_stream_extension; + set[s + 1] = on_disk.secondary_flag_allocation_possible | (if (no_fat_chain) on_disk.secondary_flag_no_fat_chain else 0); + set[s + 3] = @intCast(name.len); // name_length + var upname: [name_maximum]u16 = undefined; + for (name, 0..) |c, i| upname[i] = asciiUpper(c); + std.mem.writeInt(u16, set[s + 4 ..][0..2], on_disk.nameHash(upname[0..name.len]), .little); + std.mem.writeInt(u64, set[s + 8 ..][0..8], length, .little); // valid_data_length + std.mem.writeInt(u32, set[s + 20 ..][0..4], first_cluster, .little); + std.mem.writeInt(u64, set[s + 24 ..][0..8], length, .little); // data_length + // File Name entry (0xC1). + const f = on_disk.entry_bytes * 2; + set[f + 0] = on_disk.entry_type_file_name; + for (name, 0..) |c, i| std.mem.writeInt(u16, set[f + 2 + i * 2 ..][0..2], c, .little); + // Set checksum (over all three entries, skipping its own two bytes). + std.mem.writeInt(u16, set[2..4], on_disk.setChecksum(&set), .little); + @memcpy(bytes[root + entry_index * on_disk.entry_bytes ..][0 .. on_disk.entry_bytes * 3], &set); +} + +fn formatExfat(bytes: []u8) void { + @memset(bytes, 0); + // VBR (sector 0). + @memcpy(bytes[3..11], "EXFAT "); + std.mem.writeInt(u64, bytes[72..80], bytes.len / sector_size, .little); // volume_length + std.mem.writeInt(u32, bytes[80..84], test_fat_sector, .little); // fat_offset + std.mem.writeInt(u32, bytes[84..88], 1, .little); // fat_length + std.mem.writeInt(u32, bytes[88..92], test_heap_sector, .little); // cluster_heap_offset + std.mem.writeInt(u32, bytes[92..96], test_clusters, .little); // cluster_count + std.mem.writeInt(u32, bytes[96..100], 4, .little); // first_cluster_of_root + std.mem.writeInt(u32, bytes[100..104], 0x1234ABCD, .little); // volume_serial_number + bytes[108] = 9; // bytes_per_sector_shift = 512 + bytes[109] = 0; // sectors_per_cluster_shift = 1 + bytes[110] = 1; // number_of_fats + bytes[510] = 0x55; + bytes[511] = 0xAA; + + // FAT: reserved entries + the chains that must be walkable. + writeFatEntry(bytes, 0, 0xFFFFFFF8); + writeFatEntry(bytes, 1, 0xFFFFFFFF); + for ([_]u32{ 2, 3, 4, 5, 6 }) |c| writeFatEntry(bytes, c, 0xFFFFFFFF); // metadata + contiguous file (single-cluster chains / EOC) + writeFatEntry(bytes, 7, 9); // SPLIT: cluster 7 -> 9 + writeFatEntry(bytes, 9, 0xFFFFFFFF); // SPLIT ends + + // Allocation bitmap (cluster 2): clusters 2,3,4,5,6,7,9 in use. + const bitmap = testCluster(2); + for ([_]u32{ 2, 3, 4, 5, 6, 7, 9 }) |c| { + const bit = c - 2; + bytes[bitmap + bit / 8] |= @as(u8, 1) << @intCast(bit % 8); + } + + // Up-case table (cluster 3): 256 explicit units, a-z -> A-Z. + const upcase = testCluster(3); + var i: u32 = 0; + while (i < upcase_fold_limit) : (i += 1) { + std.mem.writeInt(u16, bytes[upcase + i * 2 ..][0..2], asciiUpper(@intCast(i)), .little); + } + const table_checksum = on_disk.upcaseChecksum(bytes[upcase .. upcase + upcase_fold_limit * 2]); + + // Root directory (cluster 4): bitmap entry, up-case entry, then the two files. + const root = testCluster(4); + var bmp = std.mem.zeroes(on_disk.AllocationBitmapEntry); + bmp.entry_type = on_disk.entry_type_allocation_bitmap; + bmp.first_cluster = 2; + bmp.data_length = (test_clusters + 7) / 8; + @memcpy(bytes[root..][0..on_disk.entry_bytes], std.mem.asBytes(&bmp)); + var uct = std.mem.zeroes(on_disk.UpcaseTableEntry); + uct.entry_type = on_disk.entry_type_upcase_table; + uct.first_cluster = 3; + uct.data_length = upcase_fold_limit * 2; + uct.table_checksum = table_checksum; + @memcpy(bytes[root + on_disk.entry_bytes ..][0..on_disk.entry_bytes], std.mem.asBytes(&uct)); + + writeFileSet(bytes, 2, "HELLO", 5, 1024, true); // contiguous (clusters 5-6) + writeFileSet(bytes, 5, "SPLIT", 7, 1024, false); // fragmented (clusters 7,9) + + // File contents: a distinct byte pattern per file, spanning both clusters. + var b: usize = 0; + while (b < 1024) : (b += 1) { + bytes[testCluster(5) + b] = @truncate(b); // HELLO (5,6 contiguous) + } + b = 0; + while (b < 512) : (b += 1) bytes[testCluster(7) + b] = @truncate(b +% 100); // SPLIT first cluster + b = 0; + while (b < 512) : (b += 1) bytes[testCluster(9) + b] = @truncate((b + 512) +% 100); // SPLIT second cluster +} + +test "mount an exFAT image and read its geometry + up-case table" { + const allocator = std.testing.allocator; + const bytes = try allocator.alloc(u8, 128 * sector_size); + defer allocator.free(bytes); + formatExfat(bytes); + + var disk = RamDisk{ .bytes = bytes }; + var fs = FileSystem.mount(disk.device()) orelse return error.ShouldMount; + try std.testing.expectEqual(@as(u32, 512), fs.geometry.bytes_per_sector); + try std.testing.expectEqual(@as(u32, 4), fs.geometry.first_cluster_of_root); + try std.testing.expectEqual(@as(u32, 2), fs.bitmap_first_cluster); + // The up-case table loaded and folds ASCII. + try std.testing.expectEqual(@as(u16, 'A'), fs.fold('a')); + try std.testing.expectEqual(@as(u16, 'Z'), fs.fold('z')); + try std.testing.expectEqual(@as(u16, '5'), fs.fold('5')); +} + +test "list the root directory (skipping the bitmap/up-case entries)" { + const allocator = std.testing.allocator; + const bytes = try allocator.alloc(u8, 128 * sector_size); + defer allocator.free(bytes); + formatExfat(bytes); + var disk = RamDisk{ .bytes = bytes }; + var fs = FileSystem.mount(disk.device()).?; + + const first = fs.listEntry(fs.rootNode(), 0).?; + try std.testing.expectEqualStrings("HELLO", first.name_buffer[0..first.name_len]); + try std.testing.expectEqual(@as(u32, 1024), first.size); + const second = fs.listEntry(fs.rootNode(), 1).?; + try std.testing.expectEqualStrings("SPLIT", second.name_buffer[0..second.name_len]); + try std.testing.expect(fs.listEntry(fs.rootNode(), 2) == null); +} + +test "resolve is case-insensitive through the up-case table" { + const allocator = std.testing.allocator; + const bytes = try allocator.alloc(u8, 128 * sector_size); + defer allocator.free(bytes); + formatExfat(bytes); + var disk = RamDisk{ .bytes = bytes }; + var fs = FileSystem.mount(disk.device()).?; + + try std.testing.expect(fs.resolve("/hello") != null); // lower-case query matches HELLO + try std.testing.expect(fs.resolve("/Hello") != null); + try std.testing.expect(fs.resolve("/NOPE") == null); + const node = fs.resolve("/HELLO").?; + try std.testing.expectEqual(@as(u32, 5), node.first_cluster); + try std.testing.expect(node.no_fat_chain); +} + +test "read a contiguous file across its cluster boundary" { + const allocator = std.testing.allocator; + const bytes = try allocator.alloc(u8, 128 * sector_size); + defer allocator.free(bytes); + formatExfat(bytes); + var disk = RamDisk{ .bytes = bytes }; + var fs = FileSystem.mount(disk.device()).?; + + const node = fs.resolve("/HELLO").?; + var readback: [test_read_bytes]u8 = undefined; + const got = fs.readFile(node, 0, &readback); + try std.testing.expectEqual(@as(usize, 1024), got); + for (readback, 0..) |byte, i| try std.testing.expectEqual(@as(u8, @truncate(i)), byte); +} + +test "read a fragmented file follows the FAT across non-contiguous clusters" { + const allocator = std.testing.allocator; + const bytes = try allocator.alloc(u8, 128 * sector_size); + defer allocator.free(bytes); + formatExfat(bytes); + var disk = RamDisk{ .bytes = bytes }; + var fs = FileSystem.mount(disk.device()).?; + + const node = fs.resolve("/SPLIT").?; + try std.testing.expect(!node.no_fat_chain); + var readback: [test_read_bytes]u8 = undefined; + const got = fs.readFile(node, 0, &readback); + try std.testing.expectEqual(@as(usize, 1024), got); + // Cluster 7 held b+100; cluster 9 (the FAT-linked second) held (b+512)+100. + for (readback, 0..) |byte, i| try std.testing.expectEqual(@as(u8, @truncate(i +% 100)), byte); + // A mid-file read that starts in the second cluster still lands right. + const mid = fs.readFile(node, 600, readback[0..10]); + try std.testing.expectEqual(@as(usize, 10), mid); + for (readback[0..10], 600..) |byte, i| try std.testing.expectEqual(@as(u8, @truncate(i +% 100)), byte); +}