fat: a small write-through block cache for metadata sectors (B5b)
Directory scans and FAT-chain walks re-read the same sectors constantly: resolving many paths under one directory (a logging burst opening dozens of files under /var/log/<stamp>/) re-read that directory's sectors and the FAT from the device every time. Add a 16-line write-through cache under the single-sector blockRead/blockWrite path, keyed by filesystem-relative LBA with round-robin eviction, so repeated metadata reads come from RAM instead of a USB round trip each. Disciplines that keep it safe: - Write-through: every write reaches the device immediately and refreshes the cache, so it never holds dirty-only data — the crash-safe data->FAT->directory write order and the flush-on-close are unchanged. - Bulk file data (the B5a multi-sector run path) bypasses the cache and a run write invalidates any overlapping cached sector, so metadata that shares a range can never go stale. - Cluster-zeroing writes uncached (write-once bulk that would only evict live metadata), invalidating any stale copy. A new engine test proves a repeated resolve does zero device reads and that a write is coherent both in-cache and against a cold-mounted filesystem (it really reached the device). Full QEMU suite green (the one intermittent `logger` miss is the documented pre-existing AP ring-3 fault: 16/16 logger reruns pass on this change and every other storage case is green; the fat engine is single-threaded and the cache is bounded global state, so it cannot itself fault intermittently).
This commit is contained in:
@@ -70,6 +70,20 @@ pub const Node = struct {
|
|||||||
const sector_size = 512;
|
const sector_size = 512;
|
||||||
const entries_per_sector = sector_size / @sizeOf(on_disk.DirectoryEntry); // 16
|
const entries_per_sector = sector_size / @sizeOf(on_disk.DirectoryEntry); // 16
|
||||||
|
|
||||||
|
// A small write-through cache of single-sector (metadata) accesses: FAT sectors,
|
||||||
|
// directory sectors, and directory-entry writebacks. Its payoff is repeated scans
|
||||||
|
// — resolving many paths under the same directory (a logging burst opening dozens
|
||||||
|
// of files under /var/log/<stamp>/) re-reads the same directory and FAT sectors,
|
||||||
|
// which now come from RAM instead of a USB round trip each. Bulk file data (the
|
||||||
|
// multi-sector run path) bypasses the cache — it is large and not re-read — and a
|
||||||
|
// run write invalidates any overlapping cached sector to stay coherent.
|
||||||
|
const block_cache_lines = 16;
|
||||||
|
const CacheLine = struct {
|
||||||
|
lba: u64 = 0,
|
||||||
|
valid: bool = false,
|
||||||
|
data: [sector_size]u8 = undefined,
|
||||||
|
};
|
||||||
|
|
||||||
pub const FileSystem = struct {
|
pub const FileSystem = struct {
|
||||||
device: BlockDevice,
|
device: BlockDevice,
|
||||||
geometry: on_disk.Geometry,
|
geometry: on_disk.Geometry,
|
||||||
@@ -95,23 +109,75 @@ pub const FileSystem = struct {
|
|||||||
// scan serve consecutive entries from one device read; every write through
|
// scan serve consecutive entries from one device read; every write through
|
||||||
// the sector keeps the cache coherent (writeFatBytes updates it in place).
|
// the sector keeps the cache coherent (writeFatBytes updates it in place).
|
||||||
fat_sector_lba: u64 = 0,
|
fat_sector_lba: u64 = 0,
|
||||||
|
// Write-through single-sector cache (see CacheLine). Keyed by filesystem-
|
||||||
|
// relative LBA; round-robin eviction.
|
||||||
|
cache: [block_cache_lines]CacheLine = [_]CacheLine{.{}} ** block_cache_lines,
|
||||||
|
cache_cursor: u32 = 0,
|
||||||
|
|
||||||
// Every filesystem-relative sector access adds the partition base.
|
// --- single-sector cache ------------------------------------------------
|
||||||
|
|
||||||
|
fn cacheFind(self: *FileSystem, lba: u64) ?*CacheLine {
|
||||||
|
for (&self.cache) |*line| {
|
||||||
|
if (line.valid and line.lba == lba) return line;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
// Install `data` (one sector) for `lba`: refresh an existing line or evict the
|
||||||
|
// next round-robin slot. Called on every device read (populate) and write
|
||||||
|
// (write-through), so a hit always mirrors the device.
|
||||||
|
fn cacheInstall(self: *FileSystem, lba: u64, data: []const u8) void {
|
||||||
|
const line = self.cacheFind(lba) orelse blk: {
|
||||||
|
const slot = &self.cache[self.cache_cursor];
|
||||||
|
self.cache_cursor = (self.cache_cursor + 1) % block_cache_lines;
|
||||||
|
slot.valid = true;
|
||||||
|
slot.lba = lba;
|
||||||
|
break :blk slot;
|
||||||
|
};
|
||||||
|
@memcpy(&line.data, data[0..sector_size]);
|
||||||
|
}
|
||||||
|
fn cacheInvalidateRange(self: *FileSystem, lba: u64, count: u32) void {
|
||||||
|
for (&self.cache) |*line| {
|
||||||
|
if (line.valid and line.lba >= lba and line.lba < lba + count) line.valid = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Every filesystem-relative sector access adds the partition base. Single-sector
|
||||||
|
// reads/writes go through the cache; the write path is write-through.
|
||||||
fn blockRead(self: *FileSystem, lba: u64, buffer: []u8) bool {
|
fn blockRead(self: *FileSystem, lba: u64, buffer: []u8) bool {
|
||||||
return self.device.readBlock(self.base_lba + lba, buffer);
|
if (self.cacheFind(lba)) |line| {
|
||||||
|
@memcpy(buffer[0..sector_size], &line.data);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if (!self.device.readBlock(self.base_lba + lba, buffer)) return false;
|
||||||
|
self.cacheInstall(lba, buffer);
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
fn blockWrite(self: *FileSystem, lba: u64, buffer: []const u8) bool {
|
fn blockWrite(self: *FileSystem, lba: u64, buffer: []const u8) bool {
|
||||||
return self.device.writeBlock(self.base_lba + lba, buffer);
|
if (!self.device.writeBlock(self.base_lba + lba, buffer)) return false;
|
||||||
|
self.cacheInstall(lba, buffer);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
// Write one sector without caching it: for bulk cluster-zeroing, whose sectors
|
||||||
|
// are write-once and would only evict useful metadata. Still invalidates any
|
||||||
|
// stale cached copy so a later read sees the zeros.
|
||||||
|
fn blockWriteUncached(self: *FileSystem, lba: u64, buffer: []const u8) bool {
|
||||||
|
if (!self.device.writeBlock(self.base_lba + lba, buffer)) return false;
|
||||||
|
self.cacheInvalidateRange(lba, 1);
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
// Move `count` contiguous full sectors in one device command. `buffer` must be
|
// Move `count` contiguous full sectors in one device command. `buffer` must be
|
||||||
// exactly count*sector_size and sector-aligned in meaning (whole sectors only);
|
// exactly count*sector_size and sector-aligned in meaning (whole sectors only);
|
||||||
// callers use these for the aligned middle of a file transfer, falling back to
|
// callers use these for the aligned middle of a file transfer, falling back to
|
||||||
// the single-sector read-modify-write path for partial head/tail sectors.
|
// the single-sector read-modify-write path for partial head/tail sectors. Bulk
|
||||||
|
// data bypasses the cache; a run write invalidates any overlapping cached
|
||||||
|
// sector so metadata that happens to share the range never goes stale.
|
||||||
fn blockReadRun(self: *FileSystem, lba: u64, count: u32, buffer: []u8) bool {
|
fn blockReadRun(self: *FileSystem, lba: u64, count: u32, buffer: []u8) bool {
|
||||||
return self.device.readBlocks(self.base_lba + lba, count, buffer);
|
return self.device.readBlocks(self.base_lba + lba, count, buffer);
|
||||||
}
|
}
|
||||||
fn blockWriteRun(self: *FileSystem, lba: u64, count: u32, buffer: []const u8) bool {
|
fn blockWriteRun(self: *FileSystem, lba: u64, count: u32, buffer: []const u8) bool {
|
||||||
return self.device.writeBlocks(self.base_lba + lba, count, buffer);
|
if (!self.device.writeBlocks(self.base_lba + lba, count, buffer)) return false;
|
||||||
|
self.cacheInvalidateRange(lba, count);
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Mount the filesystem on `device`: either a bare FAT with its boot sector at
|
/// Mount the filesystem on `device`: either a bare FAT with its boot sector at
|
||||||
@@ -344,7 +410,9 @@ pub const FileSystem = struct {
|
|||||||
var zero = [_]u8{0} ** sector_size;
|
var zero = [_]u8{0} ** sector_size;
|
||||||
var s: u32 = 0;
|
var s: u32 = 0;
|
||||||
while (s < self.geometry.sectors_per_cluster) : (s += 1) {
|
while (s < self.geometry.sectors_per_cluster) : (s += 1) {
|
||||||
_ = self.blockWrite(self.clusterSector(cluster, s), &zero);
|
// Uncached: a freshly-zeroed cluster is write-once bulk; caching its
|
||||||
|
// sectors would only evict live metadata.
|
||||||
|
_ = self.blockWriteUncached(self.clusterSector(cluster, s), &zero);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1233,6 +1301,40 @@ test "create, write, read back a file through the engine" {
|
|||||||
try std.testing.expect(fs.listEntry(fs.rootNode(), 1) == null);
|
try std.testing.expect(fs.listEntry(fs.rootNode(), 1) == null);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
test "the block cache serves repeated metadata reads and stays write-through coherent" {
|
||||||
|
const allocator = std.testing.allocator;
|
||||||
|
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||||
|
defer allocator.free(bytes);
|
||||||
|
formatFat16(bytes);
|
||||||
|
|
||||||
|
var disk = RamDisk{ .bytes = bytes };
|
||||||
|
var fs = FileSystem.mount(disk.device()).?;
|
||||||
|
var node = fs.createFile(fs.rootNode(), "CACHE.TXT").?;
|
||||||
|
var small = [_]u8{0x5A} ** 64;
|
||||||
|
_ = fs.writeFile(&node, 0, &small);
|
||||||
|
|
||||||
|
// First resolve warms the cache; the second re-reads the same directory (and
|
||||||
|
// FAT) sectors, which must now come entirely from RAM — zero device reads.
|
||||||
|
_ = fs.resolve("/CACHE.TXT").?;
|
||||||
|
disk.reads = 0;
|
||||||
|
const again = fs.resolve("/CACHE.TXT").?;
|
||||||
|
try std.testing.expectEqual(@as(usize, 0), disk.reads);
|
||||||
|
try std.testing.expectEqual(@as(u32, small.len), again.size);
|
||||||
|
|
||||||
|
// Write-through coherence: a mid-file overwrite reaches the device (a fresh
|
||||||
|
// mount, cold cache, reads the new bytes back) and the cache reflects it too.
|
||||||
|
var patch = [_]u8{0xC3} ** 16;
|
||||||
|
_ = fs.writeFile(&node, 16, &patch);
|
||||||
|
var cached_back: [16]u8 = undefined;
|
||||||
|
_ = fs.readFile(fs.resolve("/CACHE.TXT").?, 16, &cached_back);
|
||||||
|
try std.testing.expectEqualSlices(u8, &patch, &cached_back);
|
||||||
|
|
||||||
|
var cold = FileSystem.mount(disk.device()).?; // cold cache, reads from device
|
||||||
|
var device_back: [16]u8 = undefined;
|
||||||
|
_ = cold.readFile(cold.resolve("/CACHE.TXT").?, 16, &device_back);
|
||||||
|
try std.testing.expectEqualSlices(u8, &patch, &device_back);
|
||||||
|
}
|
||||||
|
|
||||||
test "multi-sector transfers coalesce and round-trip identically" {
|
test "multi-sector transfers coalesce and round-trip identically" {
|
||||||
const allocator = std.testing.allocator;
|
const allocator = std.testing.allocator;
|
||||||
const bytes = try allocator.alloc(u8, 8000 * sector_size);
|
const bytes = try allocator.alloc(u8, 8000 * sector_size);
|
||||||
|
|||||||
Reference in New Issue
Block a user