fat: a small write-through block cache for metadata sectors (B5b)

Directory scans and FAT-chain walks re-read the same sectors constantly:
resolving many paths under one directory (a logging burst opening dozens
of files under /var/log/<stamp>/) re-read that directory's sectors and the
FAT from the device every time. Add a 16-line write-through cache under the
single-sector blockRead/blockWrite path, keyed by filesystem-relative LBA
with round-robin eviction, so repeated metadata reads come from RAM instead
of a USB round trip each.

Disciplines that keep it safe:
- Write-through: every write reaches the device immediately and refreshes
  the cache, so it never holds dirty-only data — the crash-safe
  data->FAT->directory write order and the flush-on-close are unchanged.
- Bulk file data (the B5a multi-sector run path) bypasses the cache and a
  run write invalidates any overlapping cached sector, so metadata that
  shares a range can never go stale.
- Cluster-zeroing writes uncached (write-once bulk that would only evict
  live metadata), invalidating any stale copy.

A new engine test proves a repeated resolve does zero device reads and that
a write is coherent both in-cache and against a cold-mounted filesystem
(it really reached the device).

Full QEMU suite green (the one intermittent `logger` miss is the documented
pre-existing AP ring-3 fault: 16/16 logger reruns pass on this change and
every other storage case is green; the fat engine is single-threaded and
the cache is bounded global state, so it cannot itself fault intermittently).
This commit is contained in:
Daniel Samson
2026-07-22 01:08:34 +01:00
parent 6e261ccda1
commit 7e4071a065
+108 -6
View File
@@ -70,6 +70,20 @@ pub const Node = struct {
const sector_size = 512;
const entries_per_sector = sector_size / @sizeOf(on_disk.DirectoryEntry); // 16
// A small write-through cache of single-sector (metadata) accesses: FAT sectors,
// directory sectors, and directory-entry writebacks. Its payoff is repeated scans
// — resolving many paths under the same directory (a logging burst opening dozens
// of files under /var/log/<stamp>/) re-reads the same directory and FAT sectors,
// which now come from RAM instead of a USB round trip each. Bulk file data (the
// multi-sector run path) bypasses the cache — it is large and not re-read — and a
// run write invalidates any overlapping cached sector to stay coherent.
const block_cache_lines = 16;
const CacheLine = struct {
lba: u64 = 0,
valid: bool = false,
data: [sector_size]u8 = undefined,
};
pub const FileSystem = struct {
device: BlockDevice,
geometry: on_disk.Geometry,
@@ -95,23 +109,75 @@ pub const FileSystem = struct {
// scan serve consecutive entries from one device read; every write through
// the sector keeps the cache coherent (writeFatBytes updates it in place).
fat_sector_lba: u64 = 0,
// Write-through single-sector cache (see CacheLine). Keyed by filesystem-
// relative LBA; round-robin eviction.
cache: [block_cache_lines]CacheLine = [_]CacheLine{.{}} ** block_cache_lines,
cache_cursor: u32 = 0,
// Every filesystem-relative sector access adds the partition base.
// --- single-sector cache ------------------------------------------------
fn cacheFind(self: *FileSystem, lba: u64) ?*CacheLine {
for (&self.cache) |*line| {
if (line.valid and line.lba == lba) return line;
}
return null;
}
// Install `data` (one sector) for `lba`: refresh an existing line or evict the
// next round-robin slot. Called on every device read (populate) and write
// (write-through), so a hit always mirrors the device.
fn cacheInstall(self: *FileSystem, lba: u64, data: []const u8) void {
const line = self.cacheFind(lba) orelse blk: {
const slot = &self.cache[self.cache_cursor];
self.cache_cursor = (self.cache_cursor + 1) % block_cache_lines;
slot.valid = true;
slot.lba = lba;
break :blk slot;
};
@memcpy(&line.data, data[0..sector_size]);
}
fn cacheInvalidateRange(self: *FileSystem, lba: u64, count: u32) void {
for (&self.cache) |*line| {
if (line.valid and line.lba >= lba and line.lba < lba + count) line.valid = false;
}
}
// Every filesystem-relative sector access adds the partition base. Single-sector
// reads/writes go through the cache; the write path is write-through.
fn blockRead(self: *FileSystem, lba: u64, buffer: []u8) bool {
return self.device.readBlock(self.base_lba + lba, buffer);
if (self.cacheFind(lba)) |line| {
@memcpy(buffer[0..sector_size], &line.data);
return true;
}
if (!self.device.readBlock(self.base_lba + lba, buffer)) return false;
self.cacheInstall(lba, buffer);
return true;
}
fn blockWrite(self: *FileSystem, lba: u64, buffer: []const u8) bool {
return self.device.writeBlock(self.base_lba + lba, buffer);
if (!self.device.writeBlock(self.base_lba + lba, buffer)) return false;
self.cacheInstall(lba, buffer);
return true;
}
// Write one sector without caching it: for bulk cluster-zeroing, whose sectors
// are write-once and would only evict useful metadata. Still invalidates any
// stale cached copy so a later read sees the zeros.
fn blockWriteUncached(self: *FileSystem, lba: u64, buffer: []const u8) bool {
if (!self.device.writeBlock(self.base_lba + lba, buffer)) return false;
self.cacheInvalidateRange(lba, 1);
return true;
}
// Move `count` contiguous full sectors in one device command. `buffer` must be
// exactly count*sector_size and sector-aligned in meaning (whole sectors only);
// callers use these for the aligned middle of a file transfer, falling back to
// the single-sector read-modify-write path for partial head/tail sectors.
// the single-sector read-modify-write path for partial head/tail sectors. Bulk
// data bypasses the cache; a run write invalidates any overlapping cached
// sector so metadata that happens to share the range never goes stale.
fn blockReadRun(self: *FileSystem, lba: u64, count: u32, buffer: []u8) bool {
return self.device.readBlocks(self.base_lba + lba, count, buffer);
}
fn blockWriteRun(self: *FileSystem, lba: u64, count: u32, buffer: []const u8) bool {
return self.device.writeBlocks(self.base_lba + lba, count, buffer);
if (!self.device.writeBlocks(self.base_lba + lba, count, buffer)) return false;
self.cacheInvalidateRange(lba, count);
return true;
}
/// Mount the filesystem on `device`: either a bare FAT with its boot sector at
@@ -344,7 +410,9 @@ pub const FileSystem = struct {
var zero = [_]u8{0} ** sector_size;
var s: u32 = 0;
while (s < self.geometry.sectors_per_cluster) : (s += 1) {
_ = self.blockWrite(self.clusterSector(cluster, s), &zero);
// Uncached: a freshly-zeroed cluster is write-once bulk; caching its
// sectors would only evict live metadata.
_ = self.blockWriteUncached(self.clusterSector(cluster, s), &zero);
}
}
@@ -1233,6 +1301,40 @@ test "create, write, read back a file through the engine" {
try std.testing.expect(fs.listEntry(fs.rootNode(), 1) == null);
}
test "the block cache serves repeated metadata reads and stays write-through coherent" {
const allocator = std.testing.allocator;
const bytes = try allocator.alloc(u8, 5000 * sector_size);
defer allocator.free(bytes);
formatFat16(bytes);
var disk = RamDisk{ .bytes = bytes };
var fs = FileSystem.mount(disk.device()).?;
var node = fs.createFile(fs.rootNode(), "CACHE.TXT").?;
var small = [_]u8{0x5A} ** 64;
_ = fs.writeFile(&node, 0, &small);
// First resolve warms the cache; the second re-reads the same directory (and
// FAT) sectors, which must now come entirely from RAM — zero device reads.
_ = fs.resolve("/CACHE.TXT").?;
disk.reads = 0;
const again = fs.resolve("/CACHE.TXT").?;
try std.testing.expectEqual(@as(usize, 0), disk.reads);
try std.testing.expectEqual(@as(u32, small.len), again.size);
// Write-through coherence: a mid-file overwrite reaches the device (a fresh
// mount, cold cache, reads the new bytes back) and the cache reflects it too.
var patch = [_]u8{0xC3} ** 16;
_ = fs.writeFile(&node, 16, &patch);
var cached_back: [16]u8 = undefined;
_ = fs.readFile(fs.resolve("/CACHE.TXT").?, 16, &cached_back);
try std.testing.expectEqualSlices(u8, &patch, &cached_back);
var cold = FileSystem.mount(disk.device()).?; // cold cache, reads from device
var device_back: [16]u8 = undefined;
_ = cold.readFile(cold.resolve("/CACHE.TXT").?, 16, &device_back);
try std.testing.expectEqualSlices(u8, &patch, &device_back);
}
test "multi-sector transfers coalesce and round-trip identically" {
const allocator = std.testing.allocator;
const bytes = try allocator.alloc(u8, 8000 * sector_size);