671 lines
27 KiB
Zig
671 lines
27 KiB
Zig
//! Read-only reader for the Compound File Binary format (MS-CFB), also
|
|
//! known as "OLE2 structured storage". It is a FAT-style filesystem
|
|
//! inside one file, and it is the container a legacy `.xls` workbook
|
|
//! lives in: the spreadsheet itself is the `Workbook` stream inside it.
|
|
//!
|
|
//! Only what reading a stream needs is implemented:
|
|
//!
|
|
//! - header validation (signature, byte order, sector sizes for
|
|
//! major versions 3 and 4)
|
|
//! - the DIFAT -> FAT sector-allocation table, including DIFAT
|
|
//! sectors beyond the 109 entries the header holds
|
|
//! - the directory, scanned linearly (the red-black tree is ignored;
|
|
//! a name lookup over a few dozen entries does not need it)
|
|
//! - regular streams (FAT chains) and small streams (mini FAT chains
|
|
//! inside the root entry's mini stream)
|
|
//!
|
|
//! Every chain walk is bounded by the size of the table it walks, so a
|
|
//! cyclic chain is reported as `CorruptFile` instead of looping, and a
|
|
//! declared size larger than the file is rejected before allocating.
|
|
//!
|
|
//! Spec: [MS-CFB] Compound File Binary File Format.
|
|
|
|
const std = @import("std");
|
|
|
|
pub const Error = error{
|
|
/// The bytes do not start with the compound-file signature.
|
|
NotCompoundFile,
|
|
/// The structure is internally inconsistent: bad header fields,
|
|
/// a chain that loops or points outside its table, an entry larger
|
|
/// than the file can hold.
|
|
CorruptFile,
|
|
/// The file ends before a sector it references. Usually an
|
|
/// interrupted download.
|
|
Truncated,
|
|
OutOfMemory,
|
|
};
|
|
|
|
/// First eight bytes of every compound file.
|
|
pub const signature = [8]u8{ 0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1 };
|
|
|
|
/// Special sector numbers (MS-CFB 2.1). Any value above `max_reg_sect`
|
|
/// is not a real sector.
|
|
const max_reg_sect: u32 = 0xFFFFFFFA;
|
|
const end_of_chain: u32 = 0xFFFFFFFE;
|
|
const free_sect: u32 = 0xFFFFFFFF;
|
|
|
|
const header_size = 512;
|
|
const dir_entry_size = 128;
|
|
const mini_sector_size = 64;
|
|
/// DIFAT entries stored in the header itself.
|
|
const header_difat_count = 109;
|
|
|
|
/// True when `bytes` starts with the compound-file signature. Cheap
|
|
/// enough for content sniffing; does not validate anything else.
|
|
pub fn isCompoundFile(bytes: []const u8) bool {
|
|
return bytes.len >= signature.len and std.mem.eql(u8, bytes[0..signature.len], &signature);
|
|
}
|
|
|
|
pub const EntryType = enum(u8) {
|
|
unknown = 0,
|
|
storage = 1,
|
|
stream = 2,
|
|
root = 5,
|
|
_,
|
|
};
|
|
|
|
/// One directory entry. Only the fields a reader needs.
|
|
pub const Entry = struct {
|
|
/// UTF-16 code units of the name, without the terminating NUL.
|
|
name: [31]u16,
|
|
name_len: u8,
|
|
kind: EntryType,
|
|
start_sector: u32,
|
|
size: u64,
|
|
|
|
/// Case-insensitive comparison against an ASCII name. CFB names
|
|
/// compare case-insensitively (MS-CFB 2.6.4), and every stream name
|
|
/// a reader looks up by literal is ASCII.
|
|
pub fn nameEql(self: Entry, ascii: []const u8) bool {
|
|
if (ascii.len != self.name_len) return false;
|
|
for (self.name[0..self.name_len], ascii) |unit, c| {
|
|
if (unit > 0x7F) return false;
|
|
if (std.ascii.toLower(@intCast(unit)) != std.ascii.toLower(c)) return false;
|
|
}
|
|
return true;
|
|
}
|
|
};
|
|
|
|
pub const File = struct {
|
|
allocator: std.mem.Allocator,
|
|
bytes: []const u8,
|
|
sector_shift: u4,
|
|
mini_cutoff: u32,
|
|
fat: []u32,
|
|
mini_fat: []u32,
|
|
entries: []Entry,
|
|
/// Contents of the root entry's stream, which holds every small
|
|
/// stream. Empty when the file has no small streams.
|
|
mini_stream: []u8,
|
|
|
|
/// Parse the header, allocation tables and directory. `bytes` is
|
|
/// borrowed and must outlive the `File`.
|
|
pub fn init(allocator: std.mem.Allocator, bytes: []const u8) Error!File {
|
|
if (!isCompoundFile(bytes)) return error.NotCompoundFile;
|
|
if (bytes.len < header_size) return error.Truncated;
|
|
|
|
if (readU16(bytes, 28) != 0xFFFE) return error.CorruptFile; // byte order mark
|
|
const major = readU16(bytes, 26);
|
|
const sector_shift: u4 = switch (major) {
|
|
3 => 9,
|
|
4 => 12,
|
|
else => return error.CorruptFile,
|
|
};
|
|
if (readU16(bytes, 30) != sector_shift) return error.CorruptFile;
|
|
if (readU16(bytes, 32) != 6) return error.CorruptFile; // 64-byte mini sectors
|
|
const sector_size = @as(usize, 1) << sector_shift;
|
|
// Version 4 pads the header out to a full 4096-byte sector.
|
|
if (bytes.len < sector_size) return error.Truncated;
|
|
|
|
const num_fat = readU32(bytes, 44);
|
|
const first_dir = readU32(bytes, 48);
|
|
const mini_cutoff = readU32(bytes, 56);
|
|
const first_mini_fat = readU32(bytes, 60);
|
|
const first_difat = readU32(bytes, 68);
|
|
|
|
// Upper bound on sectors this file can address. Every table
|
|
// size below is checked against it before allocating, so a
|
|
// hostile header cannot request a huge allocation.
|
|
const total_sectors = (bytes.len - sector_size + sector_size - 1) >> sector_shift;
|
|
if (num_fat > total_sectors) return error.CorruptFile;
|
|
|
|
var self: File = .{
|
|
.allocator = allocator,
|
|
.bytes = bytes,
|
|
.sector_shift = sector_shift,
|
|
.mini_cutoff = mini_cutoff,
|
|
.fat = &.{},
|
|
.mini_fat = &.{},
|
|
.entries = &.{},
|
|
.mini_stream = &.{},
|
|
};
|
|
errdefer self.deinit();
|
|
|
|
self.fat = try self.readFat(num_fat, first_difat, total_sectors);
|
|
self.entries = try self.readDirectory(first_dir);
|
|
if (self.entries.len == 0 or self.entries[0].kind != .root) return error.CorruptFile;
|
|
self.mini_fat = try self.readMiniFat(first_mini_fat);
|
|
|
|
const root = self.entries[0];
|
|
if (root.size > 0) {
|
|
self.mini_stream = try self.readRegular(allocator, root.start_sector, root.size);
|
|
}
|
|
return self;
|
|
}
|
|
|
|
pub fn deinit(self: *File) void {
|
|
self.allocator.free(self.fat);
|
|
self.allocator.free(self.mini_fat);
|
|
self.allocator.free(self.entries);
|
|
self.allocator.free(self.mini_stream);
|
|
}
|
|
|
|
/// First stream entry whose name matches `ascii` case-insensitively.
|
|
pub fn find(self: File, ascii: []const u8) ?Entry {
|
|
for (self.entries) |e| {
|
|
if (e.kind == .stream and e.nameEql(ascii)) return e;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/// Read a stream's full contents. Caller owns the returned bytes.
|
|
pub fn readStream(self: File, allocator: std.mem.Allocator, entry: Entry) Error![]u8 {
|
|
if (entry.size < self.mini_cutoff) return self.readMini(allocator, entry.start_sector, entry.size);
|
|
return self.readRegular(allocator, entry.start_sector, entry.size);
|
|
}
|
|
|
|
fn sectorSize(self: File) usize {
|
|
return @as(usize, 1) << self.sector_shift;
|
|
}
|
|
|
|
/// Bytes of sector `index`. The final sector of a file may be short
|
|
/// (some writers do not pad it), so the slice can be shorter than a
|
|
/// sector; callers that need a whole sector use `fullSector`.
|
|
fn sector(self: File, index: u32) Error![]const u8 {
|
|
if (index > max_reg_sect) return error.CorruptFile;
|
|
const start = (@as(usize, index) + 1) << self.sector_shift;
|
|
if (start >= self.bytes.len) return error.Truncated;
|
|
const end = @min(start + self.sectorSize(), self.bytes.len);
|
|
return self.bytes[start..end];
|
|
}
|
|
|
|
fn fullSector(self: File, index: u32) Error![]const u8 {
|
|
const s = try self.sector(index);
|
|
if (s.len != self.sectorSize()) return error.Truncated;
|
|
return s;
|
|
}
|
|
|
|
/// Collect the FAT sector numbers (header DIFAT, then the DIFAT
|
|
/// sector chain) and concatenate those sectors into one table.
|
|
fn readFat(self: File, num_fat: u32, first_difat: u32, total_sectors: usize) Error![]u32 {
|
|
const sector_size = self.sectorSize();
|
|
const per_sector = sector_size / 4;
|
|
|
|
const fat_sectors = try self.allocator.alloc(u32, num_fat);
|
|
defer self.allocator.free(fat_sectors);
|
|
|
|
const in_header = @min(num_fat, header_difat_count);
|
|
for (0..in_header) |i| fat_sectors[i] = readU32(self.bytes, 76 + 4 * i);
|
|
|
|
var filled: usize = in_header;
|
|
var difat = first_difat;
|
|
var steps: usize = 0;
|
|
while (filled < num_fat) {
|
|
steps += 1;
|
|
if (difat > max_reg_sect or steps > total_sectors) return error.CorruptFile;
|
|
const s = try self.fullSector(difat);
|
|
// The last entry of a DIFAT sector links to the next one.
|
|
const take = @min(num_fat - filled, per_sector - 1);
|
|
for (0..take) |i| fat_sectors[filled + i] = readU32(s, 4 * i);
|
|
filled += take;
|
|
difat = readU32(s, sector_size - 4);
|
|
}
|
|
|
|
const fat = try self.allocator.alloc(u32, @as(usize, num_fat) * per_sector);
|
|
errdefer self.allocator.free(fat);
|
|
for (fat_sectors, 0..) |fs, n| {
|
|
const s = try self.fullSector(fs);
|
|
for (0..per_sector) |i| fat[n * per_sector + i] = readU32(s, 4 * i);
|
|
}
|
|
return fat;
|
|
}
|
|
|
|
fn readDirectory(self: File, first_dir: u32) Error![]Entry {
|
|
var entries: std.ArrayList(Entry) = .empty;
|
|
errdefer entries.deinit(self.allocator);
|
|
|
|
var it: ChainIterator = .{ .table = self.fat, .next_index = first_dir };
|
|
while (try it.next()) |index| {
|
|
const s = try self.fullSector(index);
|
|
var off: usize = 0;
|
|
while (off + dir_entry_size <= s.len) : (off += dir_entry_size) {
|
|
try entries.append(self.allocator, try parseEntry(s[off..][0..dir_entry_size], self.sector_shift == 9));
|
|
}
|
|
}
|
|
return entries.toOwnedSlice(self.allocator);
|
|
}
|
|
|
|
fn readMiniFat(self: File, first_mini_fat: u32) Error![]u32 {
|
|
var table: std.ArrayList(u32) = .empty;
|
|
errdefer table.deinit(self.allocator);
|
|
|
|
if (first_mini_fat == end_of_chain or first_mini_fat == free_sect) return table.toOwnedSlice(self.allocator);
|
|
var it: ChainIterator = .{ .table = self.fat, .next_index = first_mini_fat };
|
|
while (try it.next()) |index| {
|
|
const s = try self.fullSector(index);
|
|
var off: usize = 0;
|
|
while (off + 4 <= s.len) : (off += 4) try table.append(self.allocator, readU32(s, off));
|
|
}
|
|
return table.toOwnedSlice(self.allocator);
|
|
}
|
|
|
|
/// Read `size` bytes following the FAT chain from `start`.
|
|
fn readRegular(self: File, allocator: std.mem.Allocator, start: u32, size: u64) Error![]u8 {
|
|
// Beyond what the FAT can address is impossible; beyond the end
|
|
// of the bytes we have means the file was cut short.
|
|
if (size > @as(u64, self.fat.len) << self.sector_shift) return error.CorruptFile;
|
|
if (size > self.bytes.len) return error.Truncated;
|
|
const out = try allocator.alloc(u8, @intCast(size));
|
|
errdefer allocator.free(out);
|
|
|
|
var filled: usize = 0;
|
|
var it: ChainIterator = .{ .table = self.fat, .next_index = start };
|
|
while (filled < out.len) {
|
|
const index = (try it.next()) orelse return error.CorruptFile; // chain shorter than size
|
|
const s = try self.sector(index);
|
|
const n = @min(s.len, out.len - filled);
|
|
// A short final sector is only acceptable when it holds the
|
|
// rest of the stream.
|
|
if (n < self.sectorSize() and filled + n < out.len) return error.Truncated;
|
|
@memcpy(out[filled..][0..n], s[0..n]);
|
|
filled += n;
|
|
}
|
|
try it.finish();
|
|
return out;
|
|
}
|
|
|
|
/// Read `size` bytes following the mini FAT chain from `start`
|
|
/// inside the mini stream.
|
|
fn readMini(self: File, allocator: std.mem.Allocator, start: u32, size: u64) Error![]u8 {
|
|
if (size > self.mini_stream.len) return error.CorruptFile;
|
|
const out = try allocator.alloc(u8, @intCast(size));
|
|
errdefer allocator.free(out);
|
|
|
|
var filled: usize = 0;
|
|
var it: ChainIterator = .{ .table = self.mini_fat, .next_index = start };
|
|
while (filled < out.len) {
|
|
const index = (try it.next()) orelse return error.CorruptFile;
|
|
const off = @as(usize, index) * mini_sector_size;
|
|
if (off >= self.mini_stream.len) return error.CorruptFile;
|
|
const n = @min(mini_sector_size, out.len - filled, self.mini_stream.len - off);
|
|
// The mini stream ran out before the stream did.
|
|
if (n < mini_sector_size and filled + n < out.len) return error.CorruptFile;
|
|
@memcpy(out[filled..][0..n], self.mini_stream[off..][0..n]);
|
|
filled += n;
|
|
}
|
|
try it.finish();
|
|
return out;
|
|
}
|
|
};
|
|
|
|
/// Walks a sector chain through a FAT or mini FAT. Each step must land
|
|
/// inside the table, and a chain can visit at most `table.len` sectors,
|
|
/// so loops and dangling links are reported instead of followed.
|
|
const ChainIterator = struct {
|
|
table: []const u32,
|
|
next_index: u32,
|
|
steps: usize = 0,
|
|
|
|
fn next(it: *ChainIterator) Error!?u32 {
|
|
if (it.next_index == end_of_chain) return null;
|
|
if (it.next_index >= it.table.len) return error.CorruptFile;
|
|
it.steps += 1;
|
|
if (it.steps > it.table.len) return error.CorruptFile;
|
|
const current = it.next_index;
|
|
it.next_index = it.table[current];
|
|
return current;
|
|
}
|
|
|
|
/// Walk whatever is left of the chain. A reader stops once it has
|
|
/// a stream's declared size, so a loop late in the chain would
|
|
/// otherwise go unnoticed and its repeated sectors would be
|
|
/// returned as data. Extra sectors that do end are tolerated.
|
|
fn finish(it: *ChainIterator) Error!void {
|
|
while (try it.next()) |_| {}
|
|
}
|
|
};
|
|
|
|
fn parseEntry(raw: *const [dir_entry_size]u8, is_v3: bool) Error!Entry {
|
|
// Name length is in bytes and includes the UTF-16 NUL terminator.
|
|
const name_bytes = readU16(raw, 64);
|
|
if (name_bytes > 64 or name_bytes % 2 != 0) return error.CorruptFile;
|
|
const units: u8 = if (name_bytes == 0) 0 else @intCast(name_bytes / 2 - 1);
|
|
|
|
var e: Entry = .{
|
|
// SAFETY: the first `units` code units are written by the loop
|
|
// below, and nothing reads past `name_len`.
|
|
.name = undefined,
|
|
.name_len = units,
|
|
.kind = @enumFromInt(raw[66]),
|
|
.start_sector = readU32(raw, 116),
|
|
.size = std.mem.readInt(u64, raw[120..128], .little),
|
|
};
|
|
// Version 3 files only define the low 32 bits of the size; writers
|
|
// are allowed to leave junk in the high half (MS-CFB 2.6.3).
|
|
if (is_v3) e.size &= 0xFFFFFFFF;
|
|
for (0..units) |i| e.name[i] = readU16(raw, 2 * i);
|
|
return e;
|
|
}
|
|
|
|
fn readU16(bytes: []const u8, off: usize) u16 {
|
|
return std.mem.readInt(u16, bytes[off..][0..2], .little);
|
|
}
|
|
|
|
fn readU32(bytes: []const u8, off: usize) u32 {
|
|
return std.mem.readInt(u32, bytes[off..][0..4], .little);
|
|
}
|
|
|
|
// ---- Tests ----
|
|
|
|
const testing = std.testing;
|
|
const test_writer = @import("test_writer.zig");
|
|
|
|
test "isCompoundFile" {
|
|
try testing.expect(isCompoundFile(&signature));
|
|
try testing.expect(!isCompoundFile(signature[0..7]));
|
|
try testing.expect(!isCompoundFile("PK\x03\x04 a zip file, i.e. xlsx"));
|
|
}
|
|
|
|
test "reads a regular stream through the FAT" {
|
|
const allocator = testing.allocator;
|
|
const big = try allocator.alloc(u8, 10_000);
|
|
defer allocator.free(big);
|
|
for (big, 0..) |*b, i| b.* = @truncate(i *% 7);
|
|
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
|
defer allocator.free(bytes);
|
|
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
const entry = file.find("Workbook").?;
|
|
const got = try file.readStream(allocator, entry);
|
|
defer allocator.free(got);
|
|
try testing.expectEqualSlices(u8, big, got);
|
|
}
|
|
|
|
test "reads small streams through the mini FAT" {
|
|
const allocator = testing.allocator;
|
|
const bytes = try test_writer.buildCfb(allocator, &.{
|
|
.{ .name = "First", .data = "a small stream, well under the 4096-byte cutoff" },
|
|
.{ .name = "Second", .data = "x" ** 130 }, // spans three mini sectors
|
|
}, .{});
|
|
defer allocator.free(bytes);
|
|
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
|
|
const first = try file.readStream(allocator, file.find("First").?);
|
|
defer allocator.free(first);
|
|
try testing.expectEqualStrings("a small stream, well under the 4096-byte cutoff", first);
|
|
|
|
const second = try file.readStream(allocator, file.find("Second").?);
|
|
defer allocator.free(second);
|
|
try testing.expectEqualStrings("x" ** 130, second);
|
|
}
|
|
|
|
test "version 4 files use 4096-byte sectors" {
|
|
const allocator = testing.allocator;
|
|
const big = try allocator.alloc(u8, 9_000);
|
|
defer allocator.free(big);
|
|
@memset(big, 0x5A);
|
|
|
|
const bytes = try test_writer.buildCfb(allocator, &.{
|
|
.{ .name = "Workbook", .data = big },
|
|
.{ .name = "Tiny", .data = "tiny" },
|
|
}, .{ .major_version = 4 });
|
|
defer allocator.free(bytes);
|
|
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
const got = try file.readStream(allocator, file.find("Workbook").?);
|
|
defer allocator.free(got);
|
|
try testing.expectEqualSlices(u8, big, got);
|
|
const tiny = try file.readStream(allocator, file.find("tiny").?);
|
|
defer allocator.free(tiny);
|
|
try testing.expectEqualStrings("tiny", tiny);
|
|
}
|
|
|
|
test "FAT sectors beyond the header's 109 are found through DIFAT sectors" {
|
|
// 109 FAT sectors address 109 * 128 sectors (~7 MiB). A stream
|
|
// larger than that forces the writer to spill into a DIFAT sector.
|
|
const allocator = testing.allocator;
|
|
const big = try allocator.alloc(u8, 8 * 1024 * 1024);
|
|
defer allocator.free(big);
|
|
for (big, 0..) |*b, i| b.* = @truncate(i >> 9);
|
|
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
|
defer allocator.free(bytes);
|
|
try testing.expect(readU32(bytes, 72) > 0); // the writer really did use DIFAT sectors
|
|
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
const got = try file.readStream(allocator, file.find("Workbook").?);
|
|
defer allocator.free(got);
|
|
try testing.expectEqualSlices(u8, big, got);
|
|
}
|
|
|
|
test "stream lookup is case-insensitive and type-aware" {
|
|
const allocator = testing.allocator;
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = "data" }}, .{});
|
|
defer allocator.free(bytes);
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
|
|
try testing.expect(file.find("WORKBOOK") != null);
|
|
try testing.expect(file.find("workbook") != null);
|
|
try testing.expect(file.find("Workboo") == null);
|
|
// The root entry is not a stream, so its name never matches.
|
|
try testing.expect(file.find("Root Entry") == null);
|
|
}
|
|
|
|
test "Entry.nameEql rejects non-ASCII code units" {
|
|
var e: Entry = .{ .name = @splat(0), .name_len = 1, .kind = .stream, .start_sector = 0, .size = 0 };
|
|
e.name[0] = 0x00E9; // e-acute
|
|
try testing.expect(!e.nameEql("e"));
|
|
}
|
|
|
|
test "rejects a non-compound file" {
|
|
try testing.expectError(error.NotCompoundFile, File.init(testing.allocator, "not an ole file at all"));
|
|
}
|
|
|
|
test "rejects a header shorter than 512 bytes" {
|
|
try testing.expectError(error.Truncated, File.init(testing.allocator, &signature));
|
|
}
|
|
|
|
test "rejects bad header fields" {
|
|
const allocator = testing.allocator;
|
|
const good = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
|
|
defer allocator.free(good);
|
|
|
|
const Patch = struct { off: usize, value: u16 };
|
|
const patches = [_]Patch{
|
|
.{ .off = 28, .value = 0xFEFF }, // byte order
|
|
.{ .off = 26, .value = 5 }, // major version
|
|
.{ .off = 30, .value = 12 }, // sector shift does not match version 3
|
|
.{ .off = 32, .value = 7 }, // mini sector shift
|
|
};
|
|
for (patches) |p| {
|
|
const bad = try allocator.dupe(u8, good);
|
|
defer allocator.free(bad);
|
|
std.mem.writeInt(u16, bad[p.off..][0..2], p.value, .little);
|
|
try testing.expectError(error.CorruptFile, File.init(allocator, bad));
|
|
}
|
|
}
|
|
|
|
test "rejects a FAT sector count the file cannot hold" {
|
|
const allocator = testing.allocator;
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
|
|
defer allocator.free(bytes);
|
|
std.mem.writeInt(u32, bytes[44..48], 0xFFFFFF, .little);
|
|
try testing.expectError(error.CorruptFile, File.init(allocator, bytes));
|
|
}
|
|
|
|
test "rejects a version 4 file cut off inside its header sector" {
|
|
const allocator = testing.allocator;
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{ .major_version = 4 });
|
|
defer allocator.free(bytes);
|
|
try testing.expectError(error.Truncated, File.init(allocator, bytes[0..1000]));
|
|
}
|
|
|
|
test "a FAT sector listed past the end of the file is truncation" {
|
|
const allocator = testing.allocator;
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
|
|
defer allocator.free(bytes);
|
|
std.mem.writeInt(u32, bytes[76..80], 5000, .little); // header DIFAT[0]
|
|
try testing.expectError(error.Truncated, File.init(allocator, bytes));
|
|
}
|
|
|
|
test "a cyclic FAT chain is reported, not followed" {
|
|
const allocator = testing.allocator;
|
|
const big = try allocator.alloc(u8, 5000);
|
|
defer allocator.free(big);
|
|
@memset(big, 1);
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
|
defer allocator.free(bytes);
|
|
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
const entry = file.find("Workbook").?;
|
|
// Point the stream's first sector back at itself.
|
|
file.fat[entry.start_sector] = entry.start_sector;
|
|
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
|
}
|
|
|
|
test "a chain shorter than the declared size is corrupt" {
|
|
const allocator = testing.allocator;
|
|
const big = try allocator.alloc(u8, 5000);
|
|
defer allocator.free(big);
|
|
@memset(big, 1);
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
|
defer allocator.free(bytes);
|
|
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
const entry = file.find("Workbook").?;
|
|
file.fat[entry.start_sector] = end_of_chain;
|
|
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
|
|
|
var mini = entry;
|
|
mini.size = 10; // below the cutoff, so read through the (empty) mini stream
|
|
try testing.expectError(error.CorruptFile, file.readStream(allocator, mini));
|
|
}
|
|
|
|
test "a stream larger than the file is rejected before allocating" {
|
|
const allocator = testing.allocator;
|
|
const big = try allocator.alloc(u8, 5000);
|
|
defer allocator.free(big);
|
|
@memset(big, 1);
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
|
defer allocator.free(bytes);
|
|
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
var entry = file.find("Workbook").?;
|
|
entry.size = 1 << 40;
|
|
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
|
}
|
|
|
|
test "a mini chain pointing outside the mini stream is corrupt" {
|
|
const allocator = testing.allocator;
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "x" ** 100 }}, .{});
|
|
defer allocator.free(bytes);
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
const entry = file.find("S").?;
|
|
// The mini FAT has a full sector of entries (128) but the mini
|
|
// stream only holds two mini sectors, so entry 100 is in the table
|
|
// yet outside the stream.
|
|
file.mini_fat[entry.start_sector] = 100;
|
|
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
|
}
|
|
|
|
test "a mini stream that ends mid-chain is corrupt" {
|
|
const allocator = testing.allocator;
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "x" ** 100 }}, .{});
|
|
defer allocator.free(bytes);
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
var entry = file.find("S").?;
|
|
// Start the chain at the second mini sector and pretend the mini
|
|
// stream ends 36 bytes into it: the first hop yields a short read
|
|
// that cannot be the end of a 100-byte stream.
|
|
entry.start_sector = 1;
|
|
const full_len = file.mini_stream.len;
|
|
file.mini_stream.len = 100;
|
|
defer file.mini_stream.len = full_len;
|
|
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
|
}
|
|
|
|
test "a file cut short mid-stream reports Truncated" {
|
|
const allocator = testing.allocator;
|
|
const big = try allocator.alloc(u8, 20_000);
|
|
defer allocator.free(big);
|
|
@memset(big, 3);
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
|
defer allocator.free(bytes);
|
|
|
|
// The writer places the stream last, so dropping the tail cuts it.
|
|
var file = try File.init(allocator, bytes);
|
|
defer file.deinit();
|
|
const cut_at = bytes.len - 4096;
|
|
file.bytes = bytes[0..cut_at];
|
|
try testing.expectError(error.Truncated, file.readStream(allocator, file.find("Workbook").?));
|
|
|
|
// A cut that lands inside a sector leaves a short sector that is not
|
|
// the stream's last: also Truncated.
|
|
file.bytes = bytes[0 .. cut_at + 100];
|
|
try testing.expectError(error.Truncated, file.readStream(allocator, file.find("Workbook").?));
|
|
}
|
|
|
|
test "an unpadded final sector is accepted when it ends the stream" {
|
|
const allocator = testing.allocator;
|
|
const big = try allocator.alloc(u8, 5000);
|
|
defer allocator.free(big);
|
|
@memset(big, 9);
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
|
defer allocator.free(bytes);
|
|
|
|
// 5000 bytes = 9 full sectors + 392 bytes; drop the padding.
|
|
const unpadded = bytes[0 .. bytes.len - (512 - 392)];
|
|
var file = try File.init(allocator, unpadded);
|
|
defer file.deinit();
|
|
const got = try file.readStream(allocator, file.find("Workbook").?);
|
|
defer allocator.free(got);
|
|
try testing.expectEqualSlices(u8, big, got);
|
|
}
|
|
|
|
test "a directory entry with an impossible name length is corrupt" {
|
|
var raw: [dir_entry_size]u8 = @splat(0);
|
|
std.mem.writeInt(u16, raw[64..66], 66, .little);
|
|
try testing.expectError(error.CorruptFile, parseEntry(&raw, true));
|
|
std.mem.writeInt(u16, raw[64..66], 3, .little);
|
|
try testing.expectError(error.CorruptFile, parseEntry(&raw, true));
|
|
}
|
|
|
|
test "version 3 entries ignore the high half of the size" {
|
|
var raw: [dir_entry_size]u8 = @splat(0);
|
|
std.mem.writeInt(u64, raw[120..128], 0xDEADBEEF_00000010, .little);
|
|
try testing.expectEqual(@as(u64, 0x10), (try parseEntry(&raw, true)).size);
|
|
try testing.expectEqual(@as(u64, 0xDEADBEEF_00000010), (try parseEntry(&raw, false)).size);
|
|
}
|
|
|
|
test "the root entry must come first" {
|
|
const allocator = testing.allocator;
|
|
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
|
|
defer allocator.free(bytes);
|
|
// Root is the first entry of the first directory sector, which the
|
|
// writer places right after the FAT. Retype it as a stream.
|
|
const dir_off = (@as(usize, readU32(bytes, 48)) + 1) * 512;
|
|
bytes[dir_off + 66] = @intFromEnum(EntryType.stream);
|
|
try testing.expectError(error.CorruptFile, File.init(allocator, bytes));
|
|
}
|