From 4f48adaa183dba50d3a573c2b80129f6d9bf1fb3 Mon Sep 17 00:00:00 2001 From: Emil Lerch Date: Sat, 3 Oct 2026 12:30:20 -0700 Subject: [PATCH] initial vibe coded commit --- .gitignore | 5 + .mise.toml | 7 + .pre-commit-config.yaml | 51 +++ LICENSE | 21 + README.md | 93 ++++ build.zig | 47 +++ build.zig.zon | 14 + build/Coverage.zig | 239 +++++++++++ build/bcov.css | 46 ++ build/download_kcov.zig | 85 ++++ src/biff.zig | 910 ++++++++++++++++++++++++++++++++++++++++ src/cfb.zig | 671 +++++++++++++++++++++++++++++ src/root.zig | 183 ++++++++ src/test_writer.zig | 508 ++++++++++++++++++++++ 14 files changed, 2880 insertions(+) create mode 100644 .gitignore create mode 100644 .mise.toml create mode 100644 .pre-commit-config.yaml create mode 100644 LICENSE create mode 100644 README.md create mode 100644 build.zig create mode 100644 build.zig.zon create mode 100644 build/Coverage.zig create mode 100644 build/bcov.css create mode 100644 build/download_kcov.zig create mode 100644 src/biff.zig create mode 100644 src/cfb.zig create mode 100644 src/root.zig create mode 100644 src/test_writer.zig diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..93ffe27 --- /dev/null +++ b/.gitignore @@ -0,0 +1,5 @@ +.zig-cache/ +zig-out/ +zig-pkg/ +coverage/ +.tmp/ diff --git a/.mise.toml b/.mise.toml new file mode 100644 index 0000000..bd31894 --- /dev/null +++ b/.mise.toml @@ -0,0 +1,7 @@ +[tools] +zig = "0.16.0" +zls = "0.16.0" +"github:j178/prek" = "0.4.1" + +[tools."github:DonIsaac/zlint"] +version = "0.9.0" diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml new file mode 100644 index 0000000..8a2c766 --- /dev/null +++ b/.pre-commit-config.yaml @@ -0,0 +1,51 @@ +# See https://pre-commit.com for more information +# See https://pre-commit.com/hooks.html for more hooks +repos: + - repo: https://github.com/pre-commit/pre-commit-hooks + rev: v6.0.0 + hooks: + - id: trailing-whitespace + - id: end-of-file-fixer + - id: check-yaml + - id: check-added-large-files + - repo: local + hooks: + - id: forbid-ai-punctuation + name: Forbid smart punctuation (en/figure dash, minus, ellipsis, arrows, smart quotes) + language: pygrep + entry: '(–|‒|―|−|…|→|⇐|⇒|⇔|“|”|‘|’)' + files: '\.(zig|zon|md|txt|toml|ya?ml)$' + exclude: '^\.pre-commit-config\.yaml$' + - id: forbid-prose-em-dash + name: Forbid prose em-dash (use ASCII hyphen) + language: pygrep + entry: ' — ' + files: '\.(zig|zon|md|txt|toml|ya?ml)$' + exclude: '^\.pre-commit-config\.yaml$' + - repo: https://github.com/batmac/pre-commit-zig + rev: v0.3.0 + hooks: + - id: zig-fmt + - repo: local + hooks: + - id: zlint + name: Run zlint + # zlint accepts file paths only via stdin (-S); positional + # args are interpreted as directory names and silently + # produce no output. + entry: bash -c 'printf "%s\n" "$@" | zlint --deny-warnings --fix -S' -- + language: system + types: [zig] + - repo: https://github.com/batmac/pre-commit-zig + rev: v0.3.0 + hooks: + - id: zig-build + - repo: local + hooks: + - id: test + name: Run zig build coverage + entry: zig + args: ["build", "coverage", "-Dcoverage-threshold=99"] + language: system + types: [file] + pass_filenames: false diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..2252f67 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Emil Lerch + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md new file mode 100644 index 0000000..7000bf2 --- /dev/null +++ b/README.md @@ -0,0 +1,93 @@ +# biff8 + +A read-only Zig reader for legacy binary Excel workbooks: `.xls` files from +Excel 97 through 2003 (BIFF8), which some sites still export as their only +"spreadsheet" download. + +It unwraps the Compound File Binary ("OLE2") container, decodes the BIFF8 +record stream, and gives you each worksheet's cell values. That is all it +does. + +## Usage + +```zig +const biff8 = @import("biff8"); + +var wb = try biff8.Workbook.parse(allocator, bytes); +defer wb.deinit(); + +const sheet = wb.sheet("Positions") orelse return error.NoSuchSheet; +for (sheet.rows, 0..) |row, r| { + for (row, 0..) |cell, c| switch (cell) { + .text => |t| std.debug.print("{d},{d}: {s}\n", .{ r, c, t }), + .number => |n| std.debug.print("{d},{d}: {d}\n", .{ r, c, n }), + .boolean, .error_code, .empty => {}, + }; +} +``` + +`sheet.cell(row, col)` does bounds-safe random access and returns `.empty` +outside the populated area. `Sheet` has public fields, so a consumer can +build one as a literal in its own tests instead of shipping binary fixtures. + +`biff8.isCompoundFile(bytes)` is a cheap signature check for content +sniffing. Every `.xls` passes it, but so does every other legacy Office +file, so it is not proof of a workbook. + +## What is decoded + +| Record | Becomes | +|---|---| +| LABELSST, LABEL, RSTRING | `.text` (UTF-8) | +| NUMBER, RK, MULRK | `.number` | +| BOOLERR | `.boolean` or `.error_code` | +| FORMULA (+ STRING) | the cached result, as any of the above | + +Shared strings split across CONTINUE records are handled, including the +case where the split falls mid-string and the remainder switches between +1-byte and 2-byte characters. Unpaired UTF-16 surrogates decode as U+FFFD. + +## What is not + +- **Formatting.** Numbers are returned as stored, so a date-formatted cell + is an Excel serial day number; telling dates apart needs the number + format, which is not decoded. +- **Formulas.** Only their cached results. +- **Anything but BIFF8.** Excel 95 and earlier fail with + `error.UnsupportedBiffVersion`. `.xlsx` is a different format (zipped + XML) and fails with `error.NotCompoundFile`. +- **Encrypted workbooks.** `error.Encrypted`. +- **Writing.** + +## Errors + +`Workbook.parse` returns `biff8.ParseError`: + +| Error | Meaning | +|---|---| +| `NotCompoundFile` | Not an OLE2 file at all | +| `NoWorkbookStream` | An OLE2 file, but not a workbook (a `.doc`, an `.msg`, ...) | +| `UnsupportedBiffVersion` | Excel 95 or older | +| `Encrypted` | Password-protected | +| `Truncated` | The file ends early, usually an interrupted download | +| `CorruptFile` | Internally inconsistent structure | +| `OutOfMemory` | | + +Every sector chain walk is bounded and every declared size is checked +before allocating, so hostile input produces an error rather than a hang or +a huge allocation. + +## Specs + +- [MS-CFB] Compound File Binary File Format +- [MS-XLS] Excel Binary File Format (.xls) Structure + +## Development + +```sh +zig build test +zig build coverage # kcov, Linux x86_64/aarch64 +``` + +Test fixtures are built in code by `src/test_writer.zig` rather than checked +in as binary files. diff --git a/build.zig b/build.zig new file mode 100644 index 0000000..523ae21 --- /dev/null +++ b/build.zig @@ -0,0 +1,47 @@ +const std = @import("std"); +const Coverage = @import("build/Coverage.zig"); + +pub fn build(b: *std.Build) void { + const target = b.standardTargetOptions(.{}); + const optimize = b.standardOptimizeOption(.{}); + + // The public module. Consumers `@import("biff8")`. + const mod = b.addModule("biff8", .{ + .root_source_file = b.path("src/root.zig"), + .target = target, + .optimize = optimize, + }); + + // Tests: one binary rooted at src/root.zig. `refAllDecls` in + // root.zig's test block pulls in every file's tests through the + // import graph. + const tests = b.addTest(.{ .root_module = mod }); + const test_step = b.step("test", "Run all tests"); + test_step.dependOn(&b.addRunArtifact(tests).step); + + const lib = b.addLibrary(.{ + .name = "biff8", + .root_module = b.createModule(.{ + .root_source_file = b.path("src/root.zig"), + .target = target, + .optimize = optimize, + }), + }); + const docs_step = b.step("docs", "Generate documentation"); + docs_step.dependOn(&b.addInstallDirectory(.{ + .source_dir = lib.getEmittedDocs(), + .install_dir = .prefix, + .install_subdir = "docs", + }).step); + + // Coverage: `zig build coverage` (kcov, Linux x86_64/aarch64 only) + { + var cov = Coverage.init(b); + const cov_mod = b.createModule(.{ + .root_source_file = b.path("src/root.zig"), + .target = target, + .optimize = optimize, + }); + _ = cov.addModule(cov_mod, "biff8"); + } +} diff --git a/build.zig.zon b/build.zig.zon new file mode 100644 index 0000000..f1c01a2 --- /dev/null +++ b/build.zig.zon @@ -0,0 +1,14 @@ +.{ + .name = .biff8, + .version = "0.0.0", + .fingerprint = 0x3401a894ea1f9c5a, // Changing this has security and trust implications. + .minimum_zig_version = "0.16.0", + .dependencies = .{}, + .paths = .{ + "build.zig", + "build.zig.zon", + "src", + "LICENSE", + "README.md", + }, +} diff --git a/build/Coverage.zig b/build/Coverage.zig new file mode 100644 index 0000000..7d04b06 --- /dev/null +++ b/build/Coverage.zig @@ -0,0 +1,239 @@ +const builtin = @import("builtin"); +const std = @import("std"); +const Build = std.Build; + +const Coverage = @This(); + +/// Whether the host platform supports kcov-based coverage. +/// Only x86_64 and aarch64 Linux are supported (kcov binary availability). +/// On unsupported platforms, the coverage step will fail at runtime with +/// a clear error from the kcov download or execution step. +// pub const supported = builtin.os.tag == .linux and +// (builtin.cpu.arch == .x86_64 or builtin.cpu.arch == .aarch64); + +/// Initialize coverage infrastructure. Creates the "coverage" build step, +/// registers build options (-Dcoverage-threshold, -Dcoverage-dir), +/// and sets up the kcov download step. The kcov binary is downloaded into the +/// zig cache on first use and reused thereafter. +/// +/// Use `zig build coverage --verbose` to see per-file coverage breakdown. +/// +/// Call `addModule()` on the returned value to add the test module to the +/// coverage run. +/// +/// Because addModule creates a new test executable from the root module provided, +/// if there are any linking steps being done to your test executable, those +/// must also be done to the test_exe returned by addModule. +pub fn init(b: *Build) Coverage { + // Add options + const coverage_threshold = b.option(u7, "coverage-threshold", "Minimum coverage percentage required") orelse 0; + const coverage_dir = b.option([]const u8, "coverage-dir", "Coverage output directory") orelse + b.pathJoin(&.{ b.build_root.path orelse ".", "coverage" }); + const coverage_step = b.step("coverage", "Generate test coverage report"); + + // Set up kcov download. + // We can't download directly because we are sandboxed during build, but + // we can create a helper program and run it. First we need the destination + // directory, keyed by architecture. + const arch_name = switch (builtin.cpu.arch) { + .x86_64 => "x86_64", + .aarch64 => "aarch64", + else => @tagName(builtin.cpu.arch), + }; + + const Algo = std.crypto.hash.sha2.Sha256; + var hasher = Algo.init(.{}); + hasher.update("kcov-"); + hasher.update(arch_name); + var cache_hash: [Algo.digest_length]u8 = undefined; + hasher.final(&cache_hash); + + const cache_dir = b.pathJoin(&.{ + b.cache_root.path.?, + "o", + b.fmt("{s}", .{std.fmt.bytesToHex(cache_hash, .lower)}), + }); + + const kcov_path = b.pathJoin(&.{ cache_dir, b.fmt("kcov-{s}", .{arch_name}) }); + + // Create the download helper executable + const download_exe = b.addExecutable(.{ + .name = "download-kcov", + .root_module = b.createModule(.{ + .root_source_file = b.path("build/download_kcov.zig"), + .target = b.resolveTargetQuery(.{}), + }), + }); + + const run_download = b.addRunArtifact(download_exe); + run_download.addArg(kcov_path); + run_download.addArg(arch_name); + + return .{ + .b = b, + .coverage_step = coverage_step, + .coverage_dir = coverage_dir, + .coverage_threshold = coverage_threshold, + .kcov_path = kcov_path, + .run_download = run_download, + }; +} + +/// Add a test module to the coverage run. Runs kcov on the test binary, +/// then reads the coverage JSON and prints a summary (with per-file +/// breakdown if --verbose). Fails if below -Dcoverage-threshold. +/// +/// Returns the test executable so the caller can add any extra linking steps. +pub fn addModule(self: *Coverage, root_module: *Build.Module, name: []const u8) *Build.Step.Compile { + const b = self.b; + + // Set up kcov run: filter to src/ only, use custom CSS for HTML report + const run_coverage = b.addSystemCommand(&.{self.kcov_path}); + const include_path = b.pathJoin(&.{ b.build_root.path.?, "src" }); + run_coverage.addArgs(&.{ "--include-path", include_path }); + const css_file = b.pathJoin(&.{ b.build_root.path.?, "build", "bcov.css" }); + run_coverage.addArg(b.fmt("--configure=css-file={s}", .{css_file})); + run_coverage.addArg(self.coverage_dir); + + // Create a test executable for this module. + // We need to set use_llvm because the self-hosted backend + // does not emit the DWARF data that kcov needs. + const test_exe = b.addTest(.{ + .name = name, + .root_module = root_module, + .use_llvm = true, + }); + run_coverage.addArtifactArg(test_exe); + run_coverage.step.dependOn(&test_exe.step); + run_coverage.step.dependOn(&self.run_download.step); + + // Wire up the threshold check step after kcov completes + const check = b.allocator.create(Check) catch @panic("OOM"); + check.* = .{ + .step = Build.Step.init(.{ + .id = .custom, + .name = "check coverage", + .owner = b, + .makeFn = make, + }), + .json_path = b.fmt("{s}/{s}/coverage.json", .{ self.coverage_dir, name }), + .threshold = self.coverage_threshold, + }; + check.step.dependOn(&run_coverage.step); + self.coverage_step.dependOn(&check.step); + + return test_exe; +} + +// ── Coverage struct fields ────────────────────────────────── + +// Fields used by init() to configure the shared coverage infrastructure +b: *Build, +coverage_step: *Build.Step, +coverage_dir: []const u8, +coverage_threshold: u7, +kcov_path: []const u8, +run_download: *Build.Step.Run, + +// Per-module threshold-check step. Created in `addModule`; `make` +// recovers the instance via `@fieldParentPtr("step", ...)`. +const Check = struct { + step: Build.Step, + json_path: []const u8, + threshold: u7, +}; + +// This must be kept in step with kcov per-binary coverage.json format +const CoverageReport = struct { + files: []const CoverageFile, +}; + +const CoverageFile = struct { + file: []const u8, + covered_lines: usize, + total_lines: usize, +}; + +const File = struct { + file: []const u8, + percent_covered: f64, + covered_lines: usize, + total_lines: usize, + + pub fn coverageLessThanDesc(_: void, lhs: File, rhs: File) bool { + return lhs.percent_covered > rhs.percent_covered; + } +}; + +/// Build step make function: reads kcov JSON output, prints a summary +/// (with per-file breakdown if verbose), and fails if below threshold. +fn make(step: *Build.Step, options: Build.Step.MakeOptions) !void { + _ = options; + const check: *Check = @fieldParentPtr("step", step); + const allocator = step.owner.allocator; + const io = step.owner.graph.io; + + const file = std.Io.Dir.cwd().openFile(io, check.json_path, .{}) catch |err| { + return step.fail("Failed to open coverage report {s}: {}", .{ check.json_path, err }); + }; + defer file.close(io); + + var file_reader = file.reader(io, &.{}); + const content = try file_reader.interface.allocRemaining(allocator, .limited(10 * 1024 * 1024)); + defer allocator.free(content); + + const json = std.json.parseFromSlice(CoverageReport, allocator, content, .{ + .ignore_unknown_fields = true, + }) catch |err| { + return step.fail("Failed to parse coverage JSON: {}", .{err}); + }; + defer json.deinit(); + + var total_covered: usize = 0; + var total_lines: usize = 0; + + var file_list = std.ArrayList(File).empty; + defer file_list.deinit(allocator); + + for (json.value.files) |f| { + const pct: f64 = if (f.total_lines > 0) + @as(f64, @floatFromInt(f.covered_lines)) / @as(f64, @floatFromInt(f.total_lines)) * 100.0 + else + 0; + try file_list.append(allocator, .{ + .file = f.file, + .covered_lines = f.covered_lines, + .total_lines = f.total_lines, + .percent_covered = pct, + }); + total_covered += f.covered_lines; + total_lines += f.total_lines; + } + + std.mem.sort(File, file_list.items, {}, File.coverageLessThanDesc); + + var stdout_buffer: [1024]u8 = undefined; + var stdout_writer = std.Io.File.stdout().writer(io, &stdout_buffer); + const stdout = &stdout_writer.interface; + if (step.owner.verbose) { + for (file_list.items) |f| { + try stdout.print( + "{d: >5.1}% {d: >5}/{d: <5}:{s}\n", + .{ f.percent_covered, f.covered_lines, f.total_lines, f.file }, + ); + } + } + + const total_pct: f64 = if (total_lines > 0) + @as(f64, @floatFromInt(total_covered)) / @as(f64, @floatFromInt(total_lines)) * 100.0 + else + 0; + try stdout.print( + "Total test coverage: {d:.2}% ({d}/{d})\n", + .{ total_pct, total_covered, total_lines }, + ); + try stdout.flush(); + + if (@as(u7, @intFromFloat(@floor(total_pct))) < check.threshold) + return step.fail("Coverage {d:.2}% is below threshold {d}%", .{ total_pct, check.threshold }); +} diff --git a/build/bcov.css b/build/bcov.css new file mode 100644 index 0000000..861cfb1 --- /dev/null +++ b/build/bcov.css @@ -0,0 +1,46 @@ +/* Based upon the lcov CSS style, style files can be reused - Dark Theme */ +body { color: #e0e0e0; background-color: #1e1e1e; } +a:link { color: #6b9aff; text-decoration: underline; } +a:visited { color: #4dbb7a; text-decoration: underline; } +a:active { color: #ff6b8a; text-decoration: underline; } +td.title { text-align: center; padding-bottom: 10px; font-size: 20pt; font-weight: bold; } +td.ruler { background-color: #4a6ba8; } +td.headerItem { text-align: right; padding-right: 6px; font-family: sans-serif; font-weight: bold; } +td.headerValue { text-align: left; color: #6b9aff; font-family: sans-serif; font-weight: bold; } +td.versionInfo { text-align: center; padding-top: 2px; } +th.headerItem { text-align: right; padding-right: 6px; font-family: sans-serif; font-weight: bold; } +th.headerValue { text-align: left; color: #6b9aff; font-family: sans-serif; font-weight: bold; } +pre.source { font-family: monospace; white-space: pre; overflow: hidden; text-overflow: ellipsis; } +span.lineNum { background-color: #5a5a2a; } +span.lineNumLegend { background-color: #5a5a2a; width: 96px; font-weight: bold ;} +span.lineCov { background-color: #2d5a2d; } +span.linePartCov { background-color: #707000; } +span.lineNoCov { background-color: #762c2c; } +span.orderNum { background-color: #5a4a2a; float: right; width:5em; text-align: left; } +span.orderNumLegend { background-color: #5a4a2a; width: 96px; font-weight: bold ;} +span.coverHits { background-color: #4a4a2a; padding-left: 3px; padding-right: 1px; text-align: right; list-style-type: none; display: inline-block; width: 5em; } +span.coverHitsLegend { background-color: #4a4a2a; width: 96px; font-weight: bold; margin: 0 auto;} +td.tableHead { text-align: center; color: #e0e0e0; background-color: #4a6ba8; font-family: sans-serif; font-size: 120%; font-weight: bold; } +td.coverFile { text-align: left; padding-left: 10px; padding-right: 20px; color: #6b9aff; font-family: monospace; background-color: #3a3a3a; } +td.coverBar { padding-left: 10px; padding-right: 10px; background-color: #3a3a3a; } +td.coverBarOutline { background-color: #4a4a4a; } +td.coverPer { text-align: left; padding-left: 10px; padding-right: 10px; font-weight: bold; background-color: #3a3a3a; color: #e0e0e0; } +td.coverPerLeftMed { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #5a5a00; font-weight: bold; color: #e0e0e0; } +td.coverPerLeftLo { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #5a2d2d; font-weight: bold; color: #e0e0e0; } +td.coverPerLeftHi { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #2d5a2d; font-weight: bold; color: #e0e0e0; } +td.coverNum { text-align: right; padding-left: 10px; padding-right: 10px; background-color: #3a3a3a; color: #e0e0e0; } + +/* Override tablesorter hover styles for dark theme */ +.tablesorter-blue tbody > tr:hover > td, +.tablesorter-blue tbody > tr:hover + tr.tablesorter-childRow > td, +.tablesorter-blue tbody > tr:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td, +.tablesorter-blue tbody > tr.even:hover > td, +.tablesorter-blue tbody > tr.even:hover + tr.tablesorter-childRow > td, +.tablesorter-blue tbody > tr.even:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td { + background: #4a4a4a; +} +.tablesorter-blue tbody > tr.odd:hover > td, +.tablesorter-blue tbody > tr.odd:hover + tr.tablesorter-childRow > td, +.tablesorter-blue tbody > tr.odd:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td { + background: #4a4a4a; +} diff --git a/build/download_kcov.zig b/build/download_kcov.zig new file mode 100644 index 0000000..9ef6cd3 --- /dev/null +++ b/build/download_kcov.zig @@ -0,0 +1,85 @@ +const std = @import("std"); + +pub fn main(init: std.process.Init) !void { + // Build-time helper: short-lived process that downloads a single + // file. Arena lets us skip per-allocation `defer free(...)` and + // amortizes the allocation cost across the run via the arena's + // exponential block growth. Process exit reclaims everything. + const allocator = init.arena.allocator(); + const io = init.io; + + const args = try init.minimal.args.toSlice(allocator); + + if (args.len != 3) return error.InvalidArgs; + + const kcov_path = args[1]; + const arch_name = args[2]; + + // Check to see if file exists. If it does, we have nothing more to do + const stat = std.Io.Dir.cwd().statFile(io, kcov_path, .{}) catch |err| blk: { + if (err == error.FileNotFound) break :blk null else return err; + }; + // This might be better checking whether it's executable and >= 7MB, but + // for now, we'll do a simple exists check + if (stat != null) return; + var stdout_buffer: [1024]u8 = undefined; + var stdout_writer = std.Io.File.stdout().writer(io, &stdout_buffer); + const stdout = &stdout_writer.interface; + + try stdout.writeAll("Determining latest kcov version\n"); + try stdout.flush(); + + var client = std.http.Client{ .allocator = allocator, .io = io }; + defer client.deinit(); + + // Get redirect to find latest version + const list_uri = try std.Uri.parse("https://git.lerch.org/lobo/-/packages/generic/kcov/"); + var req = try client.request(.GET, list_uri, .{ .redirect_behavior = .unhandled }); + defer req.deinit(); + + try req.sendBodiless(); + var redirect_buf: [1024]u8 = undefined; + const response = try req.receiveHead(&redirect_buf); + + if (response.head.status != .see_other) return error.UnexpectedResponse; + + const location = response.head.location orelse return error.NoLocation; + const version_start = std.mem.lastIndexOfScalar(u8, location, '/') orelse return error.InvalidLocation; + const version = location[version_start + 1 ..]; + + try stdout.print( + "Downloading kcov version {s} for {s} to {s}...", + .{ version, arch_name, kcov_path }, + ); + try stdout.flush(); + + const binary_url = try std.fmt.allocPrint( + allocator, + "https://git.lerch.org/api/packages/lobo/generic/kcov/{s}/kcov-{s}", + .{ version, arch_name }, + ); + + const cache_dir = std.fs.path.dirname(kcov_path) orelse return error.InvalidPath; + std.Io.Dir.cwd().createDir(io, cache_dir, std.Io.File.Permissions.default_dir) catch |e| switch (e) { + error.PathAlreadyExists => {}, + else => return e, + }; + + const uri = try std.Uri.parse(binary_url); + const file = try std.Io.Dir.cwd().createFile(io, kcov_path, .{}); + defer file.close(io); + try file.setPermissions(io, @enumFromInt(0o755)); + + var buffer: [8192]u8 = undefined; + var writer = file.writer(io, &buffer); + const result = try client.fetch(.{ + .location = .{ .uri = uri }, + .response_writer = &writer.interface, + }); + + if (result.status != .ok) return error.DownloadFailed; + try writer.interface.flush(); + + try stdout.writeAll("done\n"); + try stdout.flush(); +} diff --git a/src/biff.zig b/src/biff.zig new file mode 100644 index 0000000..df2602b --- /dev/null +++ b/src/biff.zig @@ -0,0 +1,910 @@ +//! BIFF8 decoding: turns the `Workbook` stream of a legacy `.xls` file +//! (Excel 97 through 2003) into sheets of cell values. +//! +//! The stream is a flat sequence of records (`u16` type, `u16` length, +//! body). It opens with a "workbook globals" substream (BOF ... EOF) +//! holding the shared string table and one BOUNDSHEET8 record per +//! sheet; each BOUNDSHEET8 gives the stream offset of that sheet's own +//! BOF ... EOF substream, which holds its cell records. +//! +//! Decoded: LABELSST, LABEL, RSTRING (text), NUMBER, RK, MULRK +//! (numbers), BOOLERR (booleans and error codes), and FORMULA cached +//! results (with the STRING record that carries a text result). +//! Formatting, formulas themselves, merged cells, comments and charts +//! are ignored. Numbers are returned raw: a date-formatted cell is an +//! Excel serial number, because interpreting it needs the cell's +//! number format, which this reader does not decode. +//! +//! The shared string table (SST) is the one tricky part. A record body +//! is capped at 8224 bytes, so a long table spills into CONTINUE +//! records, and a string may be split across that boundary. When the +//! split falls inside a string's characters, the CONTINUE body begins +//! with a fresh "high byte" flag saying whether the rest of the +//! characters are 1-byte (Latin-1) or 2-byte (UTF-16), and it can +//! differ from the flag the string started with. Splits elsewhere +//! (string header, rich-text runs, phonetic data) carry no flag byte. +//! +//! Spec: [MS-XLS] Excel Binary File Format (.xls) Structure. + +const std = @import("std"); +const Allocator = std.mem.Allocator; + +pub const Error = error{ + /// Records are inconsistent: a length past the end of the stream, + /// a string index outside the shared string table, a malformed + /// record body. + CorruptFile, + /// The stream ends before a substream's EOF record. + Truncated, + /// Not BIFF8. Excel 95 and earlier (BIFF5 and below) are out of + /// scope. + UnsupportedBiffVersion, + /// The workbook is password-protected (a FILEPASS record). + Encrypted, + OutOfMemory, +}; + +/// One cell's value. +pub const Cell = union(enum) { + empty, + /// UTF-8. + text: []const u8, + /// Raw double. Date cells are Excel serial numbers (see module doc). + number: f64, + boolean: bool, + /// Excel error code: 0x00 #NULL!, 0x07 #DIV/0!, 0x0F #VALUE!, + /// 0x17 #REF!, 0x1D #NAME?, 0x24 #NUM!, 0x2A #N/A. + error_code: u8, + + pub fn asText(self: Cell) ?[]const u8 { + return switch (self) { + .text => |t| t, + else => null, + }; + } + + pub fn asNumber(self: Cell) ?f64 { + return switch (self) { + .number => |n| n, + else => null, + }; + } +}; + +/// A worksheet's cells, row-major. `rows[r]` is as wide as the +/// rightmost populated cell of row `r` (possibly empty), so a sparse +/// sheet costs memory proportional to its content. Use `cell` for +/// bounds-safe access. +/// +/// The fields are public so callers can build a `Sheet` literal in +/// their own tests without producing a binary file. +pub const Sheet = struct { + name: []const u8, + rows: []const []const Cell, + + /// The cell at zero-based (`row`, `col`), or `.empty` when out of + /// range. + pub fn cell(self: Sheet, row: usize, col: usize) Cell { + if (row >= self.rows.len) return .empty; + const r = self.rows[row]; + if (col >= r.len) return .empty; + return r[col]; + } +}; + +const rt = struct { + const bof: u16 = 0x0809; + const eof: u16 = 0x000A; + const filepass: u16 = 0x002F; + const boundsheet: u16 = 0x0085; + const sst: u16 = 0x00FC; + const @"continue": u16 = 0x003C; + const labelsst: u16 = 0x00FD; + const number: u16 = 0x0203; + const rk: u16 = 0x027E; + const mulrk: u16 = 0x00BD; + const label: u16 = 0x0204; + const rstring: u16 = 0x00D6; + const boolerr: u16 = 0x0205; + const formula: u16 = 0x0006; + const string: u16 = 0x0207; + // BOF record numbers of BIFF2, BIFF3 and BIFF4. + const bof_biff2: u16 = 0x0009; + const bof_biff3: u16 = 0x0209; + const bof_biff4: u16 = 0x0409; +}; + +const biff8_version: u16 = 0x0600; +const dt_globals: u16 = 0x0005; +const dt_worksheet: u16 = 0x0010; +/// BOUNDSHEET8 `dt` for a worksheet (or dialog sheet). +const sheet_type_worksheet: u8 = 0; + +/// Decode every worksheet in a BIFF8 `Workbook` stream, in workbook +/// order. Charts, macro sheets and VB modules are skipped. +/// +/// Everything returned is allocated from `arena` and never freed +/// individually. `scratch` holds temporaries, all freed before return. +pub fn parseSheets(arena: Allocator, scratch: Allocator, stream: []const u8) Error![]const Sheet { + var recs: Records = .{ .stream = stream }; + const first = (try recs.next()) orelse return error.Truncated; + try expectBof(first, dt_globals); + + var bounds: std.ArrayList(BoundSheet) = .empty; + defer bounds.deinit(scratch); + var sst: []const []const u8 = &.{}; + + while (true) { + const rec = (try recs.next()) orelse return error.Truncated; + switch (rec.kind) { + rt.eof => break, + rt.filepass => return error.Encrypted, + rt.boundsheet => try bounds.append(scratch, try parseBoundSheet(arena, scratch, rec.data)), + rt.sst => sst = try parseSst(arena, scratch, &recs, rec.data), + else => {}, + } + } + + var sheets: std.ArrayList(Sheet) = .empty; + for (bounds.items) |b| { + if (b.kind != sheet_type_worksheet) continue; + try sheets.append(arena, try parseSheet(arena, scratch, stream, b, sst)); + } + return sheets.items; +} + +const BoundSheet = struct { + name: []const u8, + offset: u32, + kind: u8, +}; + +fn parseBoundSheet(arena: Allocator, scratch: Allocator, data: []const u8) Error!BoundSheet { + if (data.len < 8) return error.CorruptFile; + var r: Chunks = .{ .chunks = &.{data[6..]} }; + // ShortXLUnicodeString: one-byte length, then flags and characters. + const cch = try r.byte(); + return .{ + .offset = readInt(u32, data, 0), + .kind = data[5], + .name = try r.characters(arena, scratch, cch, (try r.byte()) & 1 != 0), + }; +} + +/// SST body plus every CONTINUE record that follows it. +fn parseSst(arena: Allocator, scratch: Allocator, recs: *Records, first: []const u8) Error![]const []const u8 { + var chunks: std.ArrayList([]const u8) = .empty; + defer chunks.deinit(scratch); + try chunks.append(scratch, first); + while (recs.peekKind() == rt.@"continue") try chunks.append(scratch, (try recs.next()).?.data); + + var r: Chunks = .{ .chunks = chunks.items }; + _ = try r.int(u32); // total references; irrelevant to a reader + const unique = try r.int(u32); + // Every string costs at least three bytes (length + flags), which + // bounds the allocation a corrupt count can request. + var total_len: usize = 0; + for (chunks.items) |c| total_len += c.len; + if (unique > total_len / 3) return error.CorruptFile; + + const strings = try arena.alloc([]const u8, unique); + for (strings) |*s| { + // XLUnicodeRichExtendedString. + const cch = try r.int(u16); + const flags = try r.byte(); + const runs: usize = if (flags & 0x08 != 0) try r.int(u16) else 0; + const ext: usize = if (flags & 0x04 != 0) try r.int(u32) else 0; + s.* = try r.characters(arena, scratch, cch, flags & 1 != 0); + try r.skip(4 * runs); + try r.skip(ext); + } + return strings; +} + +const Placed = struct { + row: u16, + col: u16, + cell: Cell, +}; + +fn parseSheet(arena: Allocator, scratch: Allocator, stream: []const u8, b: BoundSheet, sst: []const []const u8) Error!Sheet { + if (b.offset >= stream.len) return error.CorruptFile; + var recs: Records = .{ .stream = stream, .pos = b.offset }; + const first = (try recs.next()) orelse return error.CorruptFile; + try expectBof(first, dt_worksheet); + + var cells: std.ArrayList(Placed) = .empty; + defer cells.deinit(scratch); + + // Embedded charts are complete BOF ... EOF substreams nested in the + // sheet's; their records are not this sheet's cells. + var depth: usize = 0; + // A FORMULA whose cached result is text is followed by a STRING + // record carrying that text. + var pending_string: ?Placed = null; + + while (true) { + const rec = (try recs.next()) orelse return error.Truncated; + const d = rec.data; + switch (rec.kind) { + rt.bof => { + depth += 1; + continue; + }, + rt.eof => { + if (depth == 0) break; + depth -= 1; + continue; + }, + else => if (depth > 0) continue, + } + + switch (rec.kind) { + rt.labelsst => { + if (d.len < 10) return error.CorruptFile; + const index = readInt(u32, d, 6); + if (index >= sst.len) return error.CorruptFile; + try put(scratch, &cells, d, .{ .text = sst[index] }); + }, + rt.number => { + if (d.len < 14) return error.CorruptFile; + try put(scratch, &cells, d, .{ .number = @bitCast(readInt(u64, d, 6)) }); + }, + rt.rk => { + if (d.len < 10) return error.CorruptFile; + try put(scratch, &cells, d, .{ .number = decodeRk(readInt(u32, d, 6)) }); + }, + rt.mulrk => { + // row, first col, N x (xf index, RK), last col. + if (d.len < 12 or (d.len - 6) % 6 != 0) return error.CorruptFile; + const row = readInt(u16, d, 0); + const first_col = readInt(u16, d, 2); + const n = (d.len - 6) / 6; + if (@as(usize, readInt(u16, d, d.len - 2)) + 1 != @as(usize, first_col) + n) return error.CorruptFile; + for (0..n) |i| { + try cells.append(scratch, .{ + .row = row, + .col = first_col + @as(u16, @intCast(i)), + .cell = .{ .number = decodeRk(readInt(u32, d, 4 + 6 * i + 2)) }, + }); + } + }, + // Both carry an XLUnicodeString after the cell header; + // RSTRING's trailing formatting runs are ignored. + rt.label, rt.rstring => { + if (d.len < 9) return error.CorruptFile; + var r: Chunks = .{ .chunks = &.{d[6..]} }; + try put(scratch, &cells, d, .{ .text = try r.unicodeString(arena, scratch) }); + }, + rt.boolerr => { + if (d.len < 8) return error.CorruptFile; + const cell: Cell = if (d[7] != 0) .{ .error_code = d[6] } else .{ .boolean = d[6] != 0 }; + try put(scratch, &cells, d, cell); + }, + rt.formula => { + if (d.len < 20) return error.CorruptFile; + pending_string = null; + const v = d[6..14]; + // 0xFFFF in the top two bytes marks a non-numeric result. + if (readInt(u16, v, 6) != 0xFFFF) { + try put(scratch, &cells, d, .{ .number = @bitCast(readInt(u64, v, 0)) }); + } else switch (v[0]) { + 0 => pending_string = .{ .row = readInt(u16, d, 0), .col = readInt(u16, d, 2), .cell = .empty }, + 1 => try put(scratch, &cells, d, .{ .boolean = v[2] != 0 }), + 2 => try put(scratch, &cells, d, .{ .error_code = v[2] }), + 3 => try put(scratch, &cells, d, .{ .text = "" }), + else => return error.CorruptFile, + } + }, + rt.string => if (pending_string) |p| { + var chunks: std.ArrayList([]const u8) = .empty; + defer chunks.deinit(scratch); + try chunks.append(scratch, d); + while (recs.peekKind() == rt.@"continue") try chunks.append(scratch, (try recs.next()).?.data); + var r: Chunks = .{ .chunks = chunks.items }; + try cells.append(scratch, .{ .row = p.row, .col = p.col, .cell = .{ .text = try r.unicodeString(arena, scratch) } }); + pending_string = null; + }, + else => {}, + } + } + + return .{ .name = b.name, .rows = try buildRows(arena, scratch, cells.items) }; +} + +/// Record a cell whose row and column are the first four bytes of `d`. +fn put(scratch: Allocator, cells: *std.ArrayList(Placed), d: []const u8, cell: Cell) Error!void { + try cells.append(scratch, .{ .row = readInt(u16, d, 0), .col = readInt(u16, d, 2), .cell = cell }); +} + +/// Lay placed cells out as rows. A later record for the same cell +/// wins, matching how Excel would overwrite it. +fn buildRows(arena: Allocator, scratch: Allocator, cells: []const Placed) Error![]const []const Cell { + var row_count: usize = 0; + for (cells) |c| row_count = @max(row_count, @as(usize, c.row) + 1); + + const widths = try scratch.alloc(usize, row_count); + defer scratch.free(widths); + @memset(widths, 0); + for (cells) |c| widths[c.row] = @max(widths[c.row], @as(usize, c.col) + 1); + + const rows = try arena.alloc([]Cell, row_count); + for (rows, widths) |*r, w| { + r.* = try arena.alloc(Cell, w); + @memset(r.*, .empty); + } + for (cells) |c| rows[c.row][c.col] = c.cell; + return rows; +} + +/// RK: a compressed number. Bit 1 selects a 30-bit signed integer over +/// the high 30 bits of an IEEE double; bit 0 divides by 100. +fn decodeRk(raw: u32) f64 { + const value: f64 = if (raw & 2 != 0) + @floatFromInt(@as(i32, @bitCast(raw)) >> 2) + else + @bitCast(@as(u64, raw & 0xFFFFFFFC) << 32); + return if (raw & 1 != 0) value / 100 else value; +} + +fn expectBof(rec: Record, dt: u16) Error!void { + switch (rec.kind) { + rt.bof => {}, + rt.bof_biff2, rt.bof_biff3, rt.bof_biff4 => return error.UnsupportedBiffVersion, + else => return error.CorruptFile, + } + if (rec.data.len < 4) return error.CorruptFile; + if (readInt(u16, rec.data, 0) != biff8_version) return error.UnsupportedBiffVersion; + if (readInt(u16, rec.data, 2) != dt) return error.CorruptFile; +} + +const Record = struct { + kind: u16, + data: []const u8, +}; + +const Records = struct { + stream: []const u8, + pos: usize = 0, + + fn next(self: *Records) Error!?Record { + if (self.pos == self.stream.len) return null; + if (self.stream.len - self.pos < 4) return error.CorruptFile; + const kind = readInt(u16, self.stream, self.pos); + const len = readInt(u16, self.stream, self.pos + 2); + const start = self.pos + 4; + if (len > self.stream.len - start) return error.CorruptFile; + self.pos = start + len; + return .{ .kind = kind, .data = self.stream[start..][0..len] }; + } + + fn peekKind(self: Records) ?u16 { + if (self.stream.len - self.pos < 4) return null; + return readInt(u16, self.stream, self.pos); + } +}; + +/// Reads across a record body and the CONTINUE bodies after it. Plain +/// reads (`byte`, `int`, `skip`) step across a boundary transparently; +/// `characters` consumes the high-byte flag a boundary inside character +/// data introduces. +const Chunks = struct { + chunks: []const []const u8, + index: usize = 0, + pos: usize = 0, + + fn current(self: Chunks) []const u8 { + return self.chunks[self.index]; + } + + fn nextChunk(self: *Chunks) Error!void { + if (self.index + 1 >= self.chunks.len) return error.CorruptFile; + self.index += 1; + self.pos = 0; + } + + fn byte(self: *Chunks) Error!u8 { + while (self.pos == self.current().len) try self.nextChunk(); + const b = self.current()[self.pos]; + self.pos += 1; + return b; + } + + fn int(self: *Chunks, comptime T: type) Error!T { + var v: T = 0; + for (0..@sizeOf(T)) |i| v |= @as(T, try self.byte()) << @intCast(8 * i); + return v; + } + + fn skip(self: *Chunks, n: usize) Error!void { + var left = n; + while (left > 0) { + if (self.pos == self.current().len) try self.nextChunk(); + const k = @min(left, self.current().len - self.pos); + self.pos += k; + left -= k; + } + } + + /// XLUnicodeString: `u16` length, flags, characters. + fn unicodeString(self: *Chunks, arena: Allocator, scratch: Allocator) Error![]const u8 { + const cch = try self.int(u16); + const flags = try self.byte(); + return self.characters(arena, scratch, cch, flags & 1 != 0); + } + + /// `count` characters, 1-byte (Latin-1) or 2-byte (UTF-16LE) per + /// `high`, re-reading the flag whenever the characters cross into + /// the next chunk. Returns UTF-8 allocated from `arena`. + fn characters(self: *Chunks, arena: Allocator, scratch: Allocator, count: usize, high_start: bool) Error![]const u8 { + var units: std.ArrayList(u16) = .empty; + defer units.deinit(scratch); + try units.ensureTotalCapacity(scratch, count); + + var high = high_start; + var left = count; + while (left > 0) { + if (self.pos == self.current().len) { + try self.nextChunk(); + if (self.current().len == 0) return error.CorruptFile; + high = self.current()[0] & 1 != 0; + self.pos = 1; + } + const avail = self.current()[self.pos..]; + if (high) { + const k = @min(left, avail.len / 2); + // A lone trailing byte cannot hold a UTF-16 unit. + if (k == 0) return error.CorruptFile; + for (0..k) |i| units.appendAssumeCapacity(readInt(u16, avail, 2 * i)); + self.pos += 2 * k; + left -= k; + } else { + const k = @min(left, avail.len); + for (avail[0..k]) |c| units.appendAssumeCapacity(c); + self.pos += k; + left -= k; + } + } + return utf8FromUtf16(arena, units.items); + } +}; + +/// UTF-16 to UTF-8. Unpaired surrogates become U+FFFD rather than an +/// error: a stray half in a cell should not make the workbook +/// unreadable. +fn utf8FromUtf16(arena: Allocator, units: []const u16) Error![]const u8 { + var len: usize = 0; + var i: usize = 0; + while (i < units.len) len += utf8Len(nextCodepoint(units, &i)); + + const out = try arena.alloc(u8, len); + var o: usize = 0; + i = 0; + while (i < units.len) o += encodeUtf8(nextCodepoint(units, &i), out[o..]); + return out; +} + +fn nextCodepoint(units: []const u16, i: *usize) u21 { + const u = units[i.*]; + i.* += 1; + if (u >= 0xD800 and u <= 0xDBFF and i.* < units.len and units[i.*] >= 0xDC00 and units[i.*] <= 0xDFFF) { + const lo = units[i.*]; + i.* += 1; + return 0x10000 + ((@as(u21, u) - 0xD800) << 10) + (lo - 0xDC00); + } + if (u >= 0xD800 and u <= 0xDFFF) return 0xFFFD; + return u; +} + +fn utf8Len(c: u21) usize { + if (c < 0x80) return 1; + if (c < 0x800) return 2; + if (c < 0x10000) return 3; + return 4; +} + +fn encodeUtf8(c: u21, out: []u8) usize { + switch (utf8Len(c)) { + 1 => out[0] = @intCast(c), + 2 => { + out[0] = @intCast(0xC0 | (c >> 6)); + out[1] = @intCast(0x80 | (c & 0x3F)); + }, + 3 => { + out[0] = @intCast(0xE0 | (c >> 12)); + out[1] = @intCast(0x80 | ((c >> 6) & 0x3F)); + out[2] = @intCast(0x80 | (c & 0x3F)); + }, + else => { + out[0] = @intCast(0xF0 | (c >> 18)); + out[1] = @intCast(0x80 | ((c >> 12) & 0x3F)); + out[2] = @intCast(0x80 | ((c >> 6) & 0x3F)); + out[3] = @intCast(0x80 | (c & 0x3F)); + }, + } + return utf8Len(c); +} + +fn readInt(comptime T: type, bytes: []const u8, off: usize) T { + return std.mem.readInt(T, bytes[off..][0..@sizeOf(T)], .little); +} + +// ---- Tests ---- + +const testing = std.testing; +const tw = @import("test_writer.zig"); + +/// Parse a workbook stream with a leak-checked scratch allocator and an +/// arena for the results. +const Parsed = struct { + arena: std.heap.ArenaAllocator, + sheets: []const Sheet, + + fn init(stream: []const u8) Error!Parsed { + var arena = std.heap.ArenaAllocator.init(testing.allocator); + errdefer arena.deinit(); + const sheets = try parseSheets(arena.allocator(), testing.allocator, stream); + return .{ .arena = arena, .sheets = sheets }; + } + + fn deinit(self: *Parsed) void { + self.arena.deinit(); + } +}; + +fn expectParseError(expected: Error, spec: tw.WorkbookSpec) !void { + const stream = try tw.workbook(testing.allocator, spec); + defer testing.allocator.free(stream); + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + try testing.expectError(expected, parseSheets(arena.allocator(), testing.allocator, stream)); +} + +test "decodes the common cell records" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + + const sst = try tw.sst(a, &.{ .{ .text = "Description" }, .{ .text = "Sample Brokerage" } }, 8224); + const stream = try tw.workbook(a, .{ + .globals = sst, + .sheets = &.{.{ + .name = "Positions", + .records = &.{ + try tw.labelSst(a, 0, 0, 0), + try tw.labelSst(a, 1, 0, 1), + try tw.number(a, 1, 1, 1234.5678), + try tw.rk(a, 1, 2, (25 << 2) | 2), // integer 25 + try tw.boolErr(a, 2, 0, 1, false), + try tw.boolErr(a, 2, 1, 0x07, true), + try tw.label(a, tw.rt.label, 3, 0, "inline label"), + try tw.label(a, tw.rt.rstring, 3, 1, "rich label"), + try tw.blank(a, 4, 3), + }, + }}, + }); + + var p = try Parsed.init(stream); + defer p.deinit(); + try testing.expectEqual(@as(usize, 1), p.sheets.len); + const s = p.sheets[0]; + try testing.expectEqualStrings("Positions", s.name); + try testing.expectEqualStrings("Description", s.cell(0, 0).asText().?); + try testing.expectEqualStrings("Sample Brokerage", s.cell(1, 0).asText().?); + try testing.expectEqual(@as(f64, 1234.5678), s.cell(1, 1).asNumber().?); + try testing.expectEqual(@as(f64, 25), s.cell(1, 2).asNumber().?); + try testing.expectEqual(Cell{ .boolean = true }, s.cell(2, 0)); + try testing.expectEqual(Cell{ .error_code = 0x07 }, s.cell(2, 1)); + try testing.expectEqualStrings("inline label", s.cell(3, 0).asText().?); + try testing.expectEqualStrings("rich label", s.cell(3, 1).asText().?); + // BLANK carries formatting only, so it adds no cell (or row). + try testing.expectEqual(@as(usize, 4), s.rows.len); + try testing.expectEqual(Cell.empty, s.cell(4, 3)); + // Out of range in either direction is empty, not a crash. + try testing.expectEqual(Cell.empty, s.cell(0, 200)); + try testing.expectEqual(Cell.empty, s.cell(9999, 0)); + try testing.expect(s.cell(1, 1).asText() == null); + try testing.expect(s.cell(0, 0).asNumber() == null); +} + +test "RK encodings" { + // Integer, integer / 100, float (high 30 bits of a double), float / 100. + try testing.expectEqual(@as(f64, 25), decodeRk((25 << 2) | 2)); + try testing.expectEqual(@as(f64, -25), decodeRk(@as(u32, @bitCast(@as(i32, -25) << 2)) | 2)); + try testing.expectEqual(@as(f64, 12.34), decodeRk((1234 << 2) | 3)); + const one_and_half: u64 = @bitCast(@as(f64, 1.5)); + const hi: u32 = @intCast(one_and_half >> 32); + try testing.expectEqual(@as(f64, 1.5), decodeRk(hi)); + try testing.expectEqual(@as(f64, 0.015), decodeRk(hi | 1)); +} + +test "MULRK fans out across columns" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{ + try tw.mulrk(a, 5, 2, &.{ (1 << 2) | 2, (2 << 2) | 2, (300 << 2) | 3 }), + } }} }); + var p = try Parsed.init(stream); + defer p.deinit(); + const s = p.sheets[0]; + try testing.expectEqual(@as(f64, 1), s.cell(5, 2).asNumber().?); + try testing.expectEqual(@as(f64, 2), s.cell(5, 3).asNumber().?); + try testing.expectEqual(@as(f64, 3), s.cell(5, 4).asNumber().?); + try testing.expectEqual(Cell.empty, s.cell(5, 1)); +} + +test "MULRK whose last column disagrees with its length is corrupt" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + const rec = try tw.mulrk(a, 0, 0, &.{ 2, 6 }); + const bad = try a.dupe(u8, rec.data); + std.mem.writeInt(u16, bad[bad.len - 2 ..][0..2], 7, .little); + try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{.{ .kind = tw.rt.mulrk, .data = bad }} }} }); +} + +test "formula cached results" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + const stream = try tw.workbook(a, .{ + .sheets = &.{.{ + .name = "S", + .records = &.{ + try tw.formula(a, 0, 0, tw.formulaNumber(42.5)), + try tw.formula(a, 0, 1, tw.formulaSpecial(0, 0)), + // Records such as SHRFMLA may sit between FORMULA and STRING. + .{ .kind = 0x04BC, .data = "\x00\x00" }, + try tw.string(a, "computed text"), + try tw.formula(a, 0, 2, tw.formulaSpecial(1, 1)), + try tw.formula(a, 0, 3, tw.formulaSpecial(2, 0x2A)), + try tw.formula(a, 0, 4, tw.formulaSpecial(3, 0)), + // A STRING with no text formula pending is ignored. + try tw.string(a, "orphan"), + // A text formula whose STRING never arrives leaves the cell empty. + try tw.formula(a, 0, 5, tw.formulaSpecial(0, 0)), + try tw.number(a, 1, 0, 1), + }, + }}, + }); + var p = try Parsed.init(stream); + defer p.deinit(); + const s = p.sheets[0]; + try testing.expectEqual(@as(f64, 42.5), s.cell(0, 0).asNumber().?); + try testing.expectEqualStrings("computed text", s.cell(0, 1).asText().?); + try testing.expectEqual(Cell{ .boolean = true }, s.cell(0, 2)); + try testing.expectEqual(Cell{ .error_code = 0x2A }, s.cell(0, 3)); + try testing.expectEqualStrings("", s.cell(0, 4).asText().?); + try testing.expectEqual(Cell.empty, s.cell(0, 5)); +} + +test "formula string result continued across a CONTINUE record" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + // STRING body: cch=6, compressed, "abc" | CONTINUE: flag 1 (UTF-16), "def". + const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{ + try tw.formula(a, 0, 0, tw.formulaSpecial(0, 0)), + .{ .kind = tw.rt.string, .data = "\x06\x00\x00abc" }, + .{ .kind = tw.rt.@"continue", .data = "\x01d\x00e\x00f\x00" }, + } }} }); + var p = try Parsed.init(stream); + defer p.deinit(); + try testing.expectEqualStrings("abcdef", p.sheets[0].cell(0, 0).asText().?); +} + +test "unknown formula result type is corrupt" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{ + try tw.formula(a, 0, 0, tw.formulaSpecial(9, 0)), + } }} }); +} + +test "shared strings split across CONTINUE records at every offset" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + + const strings = [_]tw.SstString{ + .{ .text = "Bank Deposit Sweep" }, + .{ .text = "caf\u{e9} au lait" }, // Latin-1 but not ASCII: still compressed + .{ .text = "\u{201c}buy\u{201d} or \u{201c}sell\u{201d}" }, // needs UTF-16 + .{ .text = "rich text", .runs = 3 }, + .{ .text = "phonetic", .ext = "\x01\x02\x03\x04\x05\x06\x07" }, + .{ .text = "" }, + .{ .text = "emoji \u{1F600} end" }, // surrogate pair + .{ .text = "\u{3042} then plain ascii after the switch" }, // UTF-16 start, compressed tail + }; + + // Every chunk size from tiny (splits everywhere, including inside + // headers) to large (no split) must decode identically. + var max_chunk: usize = 3; + while (max_chunk <= 200) : (max_chunk += 1) { + const sst = try tw.sst(a, &strings, max_chunk); + var records: std.ArrayList(tw.Record) = .empty; + for (0..strings.len) |i| try records.append(a, try tw.labelSst(a, @intCast(i), 0, @intCast(i))); + const stream = try tw.workbook(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} }); + + var p = try Parsed.init(stream); + defer p.deinit(); + for (strings, 0..) |s, i| { + testing.expectEqualStrings(s.text, p.sheets[0].cell(i, 0).asText().?) catch |err| { + std.debug.print("max_chunk={d} string={d}\n", .{ max_chunk, i }); + return err; + }; + } + } +} + +test "unpaired surrogates decode as U+FFFD" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + const sst = try tw.sst(a, &.{ + .{ .units = &.{ 'a', 0xD800, 'b' } }, + .{ .units = &.{ 'a', 0xDC00 } }, + .{ .units = &.{0xD83D} }, + }, 8224); + const stream = try tw.workbook(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = &.{ + try tw.labelSst(a, 0, 0, 0), + try tw.labelSst(a, 1, 0, 1), + try tw.labelSst(a, 2, 0, 2), + } }} }); + var p = try Parsed.init(stream); + defer p.deinit(); + try testing.expectEqualStrings("a\u{FFFD}b", p.sheets[0].cell(0, 0).asText().?); + try testing.expectEqualStrings("a\u{FFFD}", p.sheets[0].cell(1, 0).asText().?); + try testing.expectEqualStrings("\u{FFFD}", p.sheets[0].cell(2, 0).asText().?); +} + +test "UTF-8 encoding covers every sequence length" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const units = [_]u16{ 'A', 0x00E9, 0x20AC, 0xD83D, 0xDE00 }; + try testing.expectEqualStrings("A\u{e9}\u{20ac}\u{1F600}", try utf8FromUtf16(fx.allocator(), &units)); +} + +test "multiple sheets keep workbook order; non-worksheets are skipped" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + const stream = try tw.workbook(a, .{ .sheets = &.{ + .{ .name = "First", .records = &.{try tw.number(a, 0, 0, 1)} }, + .{ .name = "Chart1", .boundsheet_type = 2, .bof_type = 0x0020 }, + .{ .name = "Second", .records = &.{try tw.number(a, 0, 0, 2)} }, + } }); + var p = try Parsed.init(stream); + defer p.deinit(); + try testing.expectEqual(@as(usize, 2), p.sheets.len); + try testing.expectEqualStrings("First", p.sheets[0].name); + try testing.expectEqualStrings("Second", p.sheets[1].name); + try testing.expectEqual(@as(f64, 2), p.sheets[1].cell(0, 0).asNumber().?); +} + +test "embedded chart substreams inside a sheet are not its cells" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + const chart_bof = tw.bof(0x0600, 0x0020); + const stream = try tw.workbook(a, .{ + .sheets = &.{.{ + .name = "S", + .records = &.{ + try tw.number(a, 0, 0, 1), + .{ .kind = tw.rt.bof, .data = &chart_bof }, + try tw.number(a, 0, 1, 99), // belongs to the chart + .{ .kind = tw.rt.eof, .data = "" }, + try tw.number(a, 0, 2, 3), + }, + }}, + }); + var p = try Parsed.init(stream); + defer p.deinit(); + const s = p.sheets[0]; + try testing.expectEqual(@as(f64, 1), s.cell(0, 0).asNumber().?); + try testing.expectEqual(Cell.empty, s.cell(0, 1)); + try testing.expectEqual(@as(f64, 3), s.cell(0, 2).asNumber().?); +} + +test "a repeated cell keeps the last value; sparse rows stay empty" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{ + try tw.number(a, 3, 1, 1), + try tw.number(a, 3, 1, 2), + try tw.number(a, 0, 4, 7), + } }} }); + var p = try Parsed.init(stream); + defer p.deinit(); + const s = p.sheets[0]; + try testing.expectEqual(@as(usize, 4), s.rows.len); + try testing.expectEqual(@as(f64, 2), s.cell(3, 1).asNumber().?); + try testing.expectEqual(@as(usize, 0), s.rows[1].len); + try testing.expectEqual(@as(usize, 5), s.rows[0].len); +} + +test "a sheet with no cells has no rows" { + const stream = try tw.workbook(testing.allocator, .{ .sheets = &.{.{ .name = "Empty" }} }); + defer testing.allocator.free(stream); + var p = try Parsed.init(stream); + defer p.deinit(); + try testing.expectEqual(@as(usize, 0), p.sheets[0].rows.len); +} + +test "older BIFF versions are unsupported" { + try expectParseError(error.UnsupportedBiffVersion, .{ .version = 0x0500 }); + // BIFF2-4 use different BOF record numbers entirely. + const biff4_bof = "\x09\x04\x06\x00\x00\x00\x10\x00\x00\x00"; + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + try testing.expectError(error.UnsupportedBiffVersion, parseSheets(arena.allocator(), testing.allocator, biff4_bof)); +} + +test "password-protected workbooks report Encrypted" { + try expectParseError(error.Encrypted, .{ .globals = &.{.{ .kind = tw.rt.filepass, .data = "\x01\x00" }} }); +} + +test "structural corruption is reported" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + + // LABELSST pointing past the shared string table. + try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{try tw.labelSst(a, 0, 0, 5)} }} }); + // BOUNDSHEET offset past the end of the stream. + try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S" }}, .offset_override = 0xFFFFFF }); + // BOUNDSHEET offset that lands on something other than a BOF. + try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S" }}, .offset_override = 0 }); + // Short record bodies. + const short = [_]u16{ tw.rt.labelsst, tw.rt.number, tw.rt.rk, tw.rt.mulrk, tw.rt.label, tw.rt.boolerr, tw.rt.formula }; + for (short) |kind| { + try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{.{ .kind = kind, .data = "\x00\x00" }} }} }); + } + // An SST claiming more strings than its bytes could hold. + try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.sst, .data = "\x00\x00\x00\x00\xFF\xFF\x00\x00" }} }); + // An SST whose last string runs off the end of its records. + try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x05\x00\x00ab" }} }); + // A UTF-16 string that leaves one odd byte at a record boundary. + try expectParseError(error.CorruptFile, .{ .globals = &.{ + .{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x02\x00\x01a" }, + .{ .kind = tw.rt.@"continue", .data = "\x01b\x00" }, + } }); + // A CONTINUE with no room for the high-byte flag. + try expectParseError(error.CorruptFile, .{ .globals = &.{ + .{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x02\x00\x00a" }, + .{ .kind = tw.rt.@"continue", .data = "" }, + } }); + // A BOUNDSHEET too short to hold its header. + try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.boundsheet, .data = "\x00\x00" }} }); +} + +test "a substream without its EOF is truncated" { + try expectParseError(error.Truncated, .{ .sheets = &.{.{ .name = "S" }}, .omit_last_eof = true }); + + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + // Globals that never reach EOF, and an empty stream. + const globals_bof = comptime tw.bof(0x0600, 0x0005); + const no_eof = "\x09\x08\x10\x00" ++ globals_bof; + try testing.expectError(error.Truncated, parseSheets(arena.allocator(), testing.allocator, no_eof)); + try testing.expectError(error.Truncated, parseSheets(arena.allocator(), testing.allocator, "")); +} + +test "record framing errors" { + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + // A record header cut short, and a length past the end. + try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08")); + try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\xFF\x00")); + // First record is not a BOF at all. + try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x0A\x00\x00\x00")); + // A BOF too short to carry a version. + try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\x02\x00\x00\x06")); + // A BOF whose substream type is wrong for its position. + const ws_bof = comptime tw.bof(0x0600, 0x0010); + try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\x10\x00" ++ ws_bof)); +} diff --git a/src/cfb.zig b/src/cfb.zig new file mode 100644 index 0000000..6ad24c9 --- /dev/null +++ b/src/cfb.zig @@ -0,0 +1,671 @@ +//! Read-only reader for the Compound File Binary format (MS-CFB), also +//! known as "OLE2 structured storage". It is a FAT-style filesystem +//! inside one file, and it is the container a legacy `.xls` workbook +//! lives in: the spreadsheet itself is the `Workbook` stream inside it. +//! +//! Only what reading a stream needs is implemented: +//! +//! - header validation (signature, byte order, sector sizes for +//! major versions 3 and 4) +//! - the DIFAT -> FAT sector-allocation table, including DIFAT +//! sectors beyond the 109 entries the header holds +//! - the directory, scanned linearly (the red-black tree is ignored; +//! a name lookup over a few dozen entries does not need it) +//! - regular streams (FAT chains) and small streams (mini FAT chains +//! inside the root entry's mini stream) +//! +//! Every chain walk is bounded by the size of the table it walks, so a +//! cyclic chain is reported as `CorruptFile` instead of looping, and a +//! declared size larger than the file is rejected before allocating. +//! +//! Spec: [MS-CFB] Compound File Binary File Format. + +const std = @import("std"); + +pub const Error = error{ + /// The bytes do not start with the compound-file signature. + NotCompoundFile, + /// The structure is internally inconsistent: bad header fields, + /// a chain that loops or points outside its table, an entry larger + /// than the file can hold. + CorruptFile, + /// The file ends before a sector it references. Usually an + /// interrupted download. + Truncated, + OutOfMemory, +}; + +/// First eight bytes of every compound file. +pub const signature = [8]u8{ 0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1 }; + +/// Special sector numbers (MS-CFB 2.1). Any value above `max_reg_sect` +/// is not a real sector. +const max_reg_sect: u32 = 0xFFFFFFFA; +const end_of_chain: u32 = 0xFFFFFFFE; +const free_sect: u32 = 0xFFFFFFFF; + +const header_size = 512; +const dir_entry_size = 128; +const mini_sector_size = 64; +/// DIFAT entries stored in the header itself. +const header_difat_count = 109; + +/// True when `bytes` starts with the compound-file signature. Cheap +/// enough for content sniffing; does not validate anything else. +pub fn isCompoundFile(bytes: []const u8) bool { + return bytes.len >= signature.len and std.mem.eql(u8, bytes[0..signature.len], &signature); +} + +pub const EntryType = enum(u8) { + unknown = 0, + storage = 1, + stream = 2, + root = 5, + _, +}; + +/// One directory entry. Only the fields a reader needs. +pub const Entry = struct { + /// UTF-16 code units of the name, without the terminating NUL. + name: [31]u16, + name_len: u8, + kind: EntryType, + start_sector: u32, + size: u64, + + /// Case-insensitive comparison against an ASCII name. CFB names + /// compare case-insensitively (MS-CFB 2.6.4), and every stream name + /// a reader looks up by literal is ASCII. + pub fn nameEql(self: Entry, ascii: []const u8) bool { + if (ascii.len != self.name_len) return false; + for (self.name[0..self.name_len], ascii) |unit, c| { + if (unit > 0x7F) return false; + if (std.ascii.toLower(@intCast(unit)) != std.ascii.toLower(c)) return false; + } + return true; + } +}; + +pub const File = struct { + allocator: std.mem.Allocator, + bytes: []const u8, + sector_shift: u4, + mini_cutoff: u32, + fat: []u32, + mini_fat: []u32, + entries: []Entry, + /// Contents of the root entry's stream, which holds every small + /// stream. Empty when the file has no small streams. + mini_stream: []u8, + + /// Parse the header, allocation tables and directory. `bytes` is + /// borrowed and must outlive the `File`. + pub fn init(allocator: std.mem.Allocator, bytes: []const u8) Error!File { + if (!isCompoundFile(bytes)) return error.NotCompoundFile; + if (bytes.len < header_size) return error.Truncated; + + if (readU16(bytes, 28) != 0xFFFE) return error.CorruptFile; // byte order mark + const major = readU16(bytes, 26); + const sector_shift: u4 = switch (major) { + 3 => 9, + 4 => 12, + else => return error.CorruptFile, + }; + if (readU16(bytes, 30) != sector_shift) return error.CorruptFile; + if (readU16(bytes, 32) != 6) return error.CorruptFile; // 64-byte mini sectors + const sector_size = @as(usize, 1) << sector_shift; + // Version 4 pads the header out to a full 4096-byte sector. + if (bytes.len < sector_size) return error.Truncated; + + const num_fat = readU32(bytes, 44); + const first_dir = readU32(bytes, 48); + const mini_cutoff = readU32(bytes, 56); + const first_mini_fat = readU32(bytes, 60); + const first_difat = readU32(bytes, 68); + + // Upper bound on sectors this file can address. Every table + // size below is checked against it before allocating, so a + // hostile header cannot request a huge allocation. + const total_sectors = (bytes.len - sector_size + sector_size - 1) >> sector_shift; + if (num_fat > total_sectors) return error.CorruptFile; + + var self: File = .{ + .allocator = allocator, + .bytes = bytes, + .sector_shift = sector_shift, + .mini_cutoff = mini_cutoff, + .fat = &.{}, + .mini_fat = &.{}, + .entries = &.{}, + .mini_stream = &.{}, + }; + errdefer self.deinit(); + + self.fat = try self.readFat(num_fat, first_difat, total_sectors); + self.entries = try self.readDirectory(first_dir); + if (self.entries.len == 0 or self.entries[0].kind != .root) return error.CorruptFile; + self.mini_fat = try self.readMiniFat(first_mini_fat); + + const root = self.entries[0]; + if (root.size > 0) { + self.mini_stream = try self.readRegular(allocator, root.start_sector, root.size); + } + return self; + } + + pub fn deinit(self: *File) void { + self.allocator.free(self.fat); + self.allocator.free(self.mini_fat); + self.allocator.free(self.entries); + self.allocator.free(self.mini_stream); + } + + /// First stream entry whose name matches `ascii` case-insensitively. + pub fn find(self: File, ascii: []const u8) ?Entry { + for (self.entries) |e| { + if (e.kind == .stream and e.nameEql(ascii)) return e; + } + return null; + } + + /// Read a stream's full contents. Caller owns the returned bytes. + pub fn readStream(self: File, allocator: std.mem.Allocator, entry: Entry) Error![]u8 { + if (entry.size < self.mini_cutoff) return self.readMini(allocator, entry.start_sector, entry.size); + return self.readRegular(allocator, entry.start_sector, entry.size); + } + + fn sectorSize(self: File) usize { + return @as(usize, 1) << self.sector_shift; + } + + /// Bytes of sector `index`. The final sector of a file may be short + /// (some writers do not pad it), so the slice can be shorter than a + /// sector; callers that need a whole sector use `fullSector`. + fn sector(self: File, index: u32) Error![]const u8 { + if (index > max_reg_sect) return error.CorruptFile; + const start = (@as(usize, index) + 1) << self.sector_shift; + if (start >= self.bytes.len) return error.Truncated; + const end = @min(start + self.sectorSize(), self.bytes.len); + return self.bytes[start..end]; + } + + fn fullSector(self: File, index: u32) Error![]const u8 { + const s = try self.sector(index); + if (s.len != self.sectorSize()) return error.Truncated; + return s; + } + + /// Collect the FAT sector numbers (header DIFAT, then the DIFAT + /// sector chain) and concatenate those sectors into one table. + fn readFat(self: File, num_fat: u32, first_difat: u32, total_sectors: usize) Error![]u32 { + const sector_size = self.sectorSize(); + const per_sector = sector_size / 4; + + const fat_sectors = try self.allocator.alloc(u32, num_fat); + defer self.allocator.free(fat_sectors); + + const in_header = @min(num_fat, header_difat_count); + for (0..in_header) |i| fat_sectors[i] = readU32(self.bytes, 76 + 4 * i); + + var filled: usize = in_header; + var difat = first_difat; + var steps: usize = 0; + while (filled < num_fat) { + steps += 1; + if (difat > max_reg_sect or steps > total_sectors) return error.CorruptFile; + const s = try self.fullSector(difat); + // The last entry of a DIFAT sector links to the next one. + const take = @min(num_fat - filled, per_sector - 1); + for (0..take) |i| fat_sectors[filled + i] = readU32(s, 4 * i); + filled += take; + difat = readU32(s, sector_size - 4); + } + + const fat = try self.allocator.alloc(u32, @as(usize, num_fat) * per_sector); + errdefer self.allocator.free(fat); + for (fat_sectors, 0..) |fs, n| { + const s = try self.fullSector(fs); + for (0..per_sector) |i| fat[n * per_sector + i] = readU32(s, 4 * i); + } + return fat; + } + + fn readDirectory(self: File, first_dir: u32) Error![]Entry { + var entries: std.ArrayList(Entry) = .empty; + errdefer entries.deinit(self.allocator); + + var it: ChainIterator = .{ .table = self.fat, .next_index = first_dir }; + while (try it.next()) |index| { + const s = try self.fullSector(index); + var off: usize = 0; + while (off + dir_entry_size <= s.len) : (off += dir_entry_size) { + try entries.append(self.allocator, try parseEntry(s[off..][0..dir_entry_size], self.sector_shift == 9)); + } + } + return entries.toOwnedSlice(self.allocator); + } + + fn readMiniFat(self: File, first_mini_fat: u32) Error![]u32 { + var table: std.ArrayList(u32) = .empty; + errdefer table.deinit(self.allocator); + + if (first_mini_fat == end_of_chain or first_mini_fat == free_sect) return table.toOwnedSlice(self.allocator); + var it: ChainIterator = .{ .table = self.fat, .next_index = first_mini_fat }; + while (try it.next()) |index| { + const s = try self.fullSector(index); + var off: usize = 0; + while (off + 4 <= s.len) : (off += 4) try table.append(self.allocator, readU32(s, off)); + } + return table.toOwnedSlice(self.allocator); + } + + /// Read `size` bytes following the FAT chain from `start`. + fn readRegular(self: File, allocator: std.mem.Allocator, start: u32, size: u64) Error![]u8 { + // Beyond what the FAT can address is impossible; beyond the end + // of the bytes we have means the file was cut short. + if (size > @as(u64, self.fat.len) << self.sector_shift) return error.CorruptFile; + if (size > self.bytes.len) return error.Truncated; + const out = try allocator.alloc(u8, @intCast(size)); + errdefer allocator.free(out); + + var filled: usize = 0; + var it: ChainIterator = .{ .table = self.fat, .next_index = start }; + while (filled < out.len) { + const index = (try it.next()) orelse return error.CorruptFile; // chain shorter than size + const s = try self.sector(index); + const n = @min(s.len, out.len - filled); + // A short final sector is only acceptable when it holds the + // rest of the stream. + if (n < self.sectorSize() and filled + n < out.len) return error.Truncated; + @memcpy(out[filled..][0..n], s[0..n]); + filled += n; + } + try it.finish(); + return out; + } + + /// Read `size` bytes following the mini FAT chain from `start` + /// inside the mini stream. + fn readMini(self: File, allocator: std.mem.Allocator, start: u32, size: u64) Error![]u8 { + if (size > self.mini_stream.len) return error.CorruptFile; + const out = try allocator.alloc(u8, @intCast(size)); + errdefer allocator.free(out); + + var filled: usize = 0; + var it: ChainIterator = .{ .table = self.mini_fat, .next_index = start }; + while (filled < out.len) { + const index = (try it.next()) orelse return error.CorruptFile; + const off = @as(usize, index) * mini_sector_size; + if (off >= self.mini_stream.len) return error.CorruptFile; + const n = @min(mini_sector_size, out.len - filled, self.mini_stream.len - off); + // The mini stream ran out before the stream did. + if (n < mini_sector_size and filled + n < out.len) return error.CorruptFile; + @memcpy(out[filled..][0..n], self.mini_stream[off..][0..n]); + filled += n; + } + try it.finish(); + return out; + } +}; + +/// Walks a sector chain through a FAT or mini FAT. Each step must land +/// inside the table, and a chain can visit at most `table.len` sectors, +/// so loops and dangling links are reported instead of followed. +const ChainIterator = struct { + table: []const u32, + next_index: u32, + steps: usize = 0, + + fn next(it: *ChainIterator) Error!?u32 { + if (it.next_index == end_of_chain) return null; + if (it.next_index >= it.table.len) return error.CorruptFile; + it.steps += 1; + if (it.steps > it.table.len) return error.CorruptFile; + const current = it.next_index; + it.next_index = it.table[current]; + return current; + } + + /// Walk whatever is left of the chain. A reader stops once it has + /// a stream's declared size, so a loop late in the chain would + /// otherwise go unnoticed and its repeated sectors would be + /// returned as data. Extra sectors that do end are tolerated. + fn finish(it: *ChainIterator) Error!void { + while (try it.next()) |_| {} + } +}; + +fn parseEntry(raw: *const [dir_entry_size]u8, is_v3: bool) Error!Entry { + // Name length is in bytes and includes the UTF-16 NUL terminator. + const name_bytes = readU16(raw, 64); + if (name_bytes > 64 or name_bytes % 2 != 0) return error.CorruptFile; + const units: u8 = if (name_bytes == 0) 0 else @intCast(name_bytes / 2 - 1); + + var e: Entry = .{ + // SAFETY: the first `units` code units are written by the loop + // below, and nothing reads past `name_len`. + .name = undefined, + .name_len = units, + .kind = @enumFromInt(raw[66]), + .start_sector = readU32(raw, 116), + .size = std.mem.readInt(u64, raw[120..128], .little), + }; + // Version 3 files only define the low 32 bits of the size; writers + // are allowed to leave junk in the high half (MS-CFB 2.6.3). + if (is_v3) e.size &= 0xFFFFFFFF; + for (0..units) |i| e.name[i] = readU16(raw, 2 * i); + return e; +} + +fn readU16(bytes: []const u8, off: usize) u16 { + return std.mem.readInt(u16, bytes[off..][0..2], .little); +} + +fn readU32(bytes: []const u8, off: usize) u32 { + return std.mem.readInt(u32, bytes[off..][0..4], .little); +} + +// ---- Tests ---- + +const testing = std.testing; +const test_writer = @import("test_writer.zig"); + +test "isCompoundFile" { + try testing.expect(isCompoundFile(&signature)); + try testing.expect(!isCompoundFile(signature[0..7])); + try testing.expect(!isCompoundFile("PK\x03\x04 a zip file, i.e. xlsx")); +} + +test "reads a regular stream through the FAT" { + const allocator = testing.allocator; + const big = try allocator.alloc(u8, 10_000); + defer allocator.free(big); + for (big, 0..) |*b, i| b.* = @truncate(i *% 7); + + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{}); + defer allocator.free(bytes); + + var file = try File.init(allocator, bytes); + defer file.deinit(); + const entry = file.find("Workbook").?; + const got = try file.readStream(allocator, entry); + defer allocator.free(got); + try testing.expectEqualSlices(u8, big, got); +} + +test "reads small streams through the mini FAT" { + const allocator = testing.allocator; + const bytes = try test_writer.buildCfb(allocator, &.{ + .{ .name = "First", .data = "a small stream, well under the 4096-byte cutoff" }, + .{ .name = "Second", .data = "x" ** 130 }, // spans three mini sectors + }, .{}); + defer allocator.free(bytes); + + var file = try File.init(allocator, bytes); + defer file.deinit(); + + const first = try file.readStream(allocator, file.find("First").?); + defer allocator.free(first); + try testing.expectEqualStrings("a small stream, well under the 4096-byte cutoff", first); + + const second = try file.readStream(allocator, file.find("Second").?); + defer allocator.free(second); + try testing.expectEqualStrings("x" ** 130, second); +} + +test "version 4 files use 4096-byte sectors" { + const allocator = testing.allocator; + const big = try allocator.alloc(u8, 9_000); + defer allocator.free(big); + @memset(big, 0x5A); + + const bytes = try test_writer.buildCfb(allocator, &.{ + .{ .name = "Workbook", .data = big }, + .{ .name = "Tiny", .data = "tiny" }, + }, .{ .major_version = 4 }); + defer allocator.free(bytes); + + var file = try File.init(allocator, bytes); + defer file.deinit(); + const got = try file.readStream(allocator, file.find("Workbook").?); + defer allocator.free(got); + try testing.expectEqualSlices(u8, big, got); + const tiny = try file.readStream(allocator, file.find("tiny").?); + defer allocator.free(tiny); + try testing.expectEqualStrings("tiny", tiny); +} + +test "FAT sectors beyond the header's 109 are found through DIFAT sectors" { + // 109 FAT sectors address 109 * 128 sectors (~7 MiB). A stream + // larger than that forces the writer to spill into a DIFAT sector. + const allocator = testing.allocator; + const big = try allocator.alloc(u8, 8 * 1024 * 1024); + defer allocator.free(big); + for (big, 0..) |*b, i| b.* = @truncate(i >> 9); + + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{}); + defer allocator.free(bytes); + try testing.expect(readU32(bytes, 72) > 0); // the writer really did use DIFAT sectors + + var file = try File.init(allocator, bytes); + defer file.deinit(); + const got = try file.readStream(allocator, file.find("Workbook").?); + defer allocator.free(got); + try testing.expectEqualSlices(u8, big, got); +} + +test "stream lookup is case-insensitive and type-aware" { + const allocator = testing.allocator; + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = "data" }}, .{}); + defer allocator.free(bytes); + var file = try File.init(allocator, bytes); + defer file.deinit(); + + try testing.expect(file.find("WORKBOOK") != null); + try testing.expect(file.find("workbook") != null); + try testing.expect(file.find("Workboo") == null); + // The root entry is not a stream, so its name never matches. + try testing.expect(file.find("Root Entry") == null); +} + +test "Entry.nameEql rejects non-ASCII code units" { + var e: Entry = .{ .name = @splat(0), .name_len = 1, .kind = .stream, .start_sector = 0, .size = 0 }; + e.name[0] = 0x00E9; // e-acute + try testing.expect(!e.nameEql("e")); +} + +test "rejects a non-compound file" { + try testing.expectError(error.NotCompoundFile, File.init(testing.allocator, "not an ole file at all")); +} + +test "rejects a header shorter than 512 bytes" { + try testing.expectError(error.Truncated, File.init(testing.allocator, &signature)); +} + +test "rejects bad header fields" { + const allocator = testing.allocator; + const good = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{}); + defer allocator.free(good); + + const Patch = struct { off: usize, value: u16 }; + const patches = [_]Patch{ + .{ .off = 28, .value = 0xFEFF }, // byte order + .{ .off = 26, .value = 5 }, // major version + .{ .off = 30, .value = 12 }, // sector shift does not match version 3 + .{ .off = 32, .value = 7 }, // mini sector shift + }; + for (patches) |p| { + const bad = try allocator.dupe(u8, good); + defer allocator.free(bad); + std.mem.writeInt(u16, bad[p.off..][0..2], p.value, .little); + try testing.expectError(error.CorruptFile, File.init(allocator, bad)); + } +} + +test "rejects a FAT sector count the file cannot hold" { + const allocator = testing.allocator; + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{}); + defer allocator.free(bytes); + std.mem.writeInt(u32, bytes[44..48], 0xFFFFFF, .little); + try testing.expectError(error.CorruptFile, File.init(allocator, bytes)); +} + +test "rejects a version 4 file cut off inside its header sector" { + const allocator = testing.allocator; + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{ .major_version = 4 }); + defer allocator.free(bytes); + try testing.expectError(error.Truncated, File.init(allocator, bytes[0..1000])); +} + +test "a FAT sector listed past the end of the file is truncation" { + const allocator = testing.allocator; + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{}); + defer allocator.free(bytes); + std.mem.writeInt(u32, bytes[76..80], 5000, .little); // header DIFAT[0] + try testing.expectError(error.Truncated, File.init(allocator, bytes)); +} + +test "a cyclic FAT chain is reported, not followed" { + const allocator = testing.allocator; + const big = try allocator.alloc(u8, 5000); + defer allocator.free(big); + @memset(big, 1); + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{}); + defer allocator.free(bytes); + + var file = try File.init(allocator, bytes); + defer file.deinit(); + const entry = file.find("Workbook").?; + // Point the stream's first sector back at itself. + file.fat[entry.start_sector] = entry.start_sector; + try testing.expectError(error.CorruptFile, file.readStream(allocator, entry)); +} + +test "a chain shorter than the declared size is corrupt" { + const allocator = testing.allocator; + const big = try allocator.alloc(u8, 5000); + defer allocator.free(big); + @memset(big, 1); + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{}); + defer allocator.free(bytes); + + var file = try File.init(allocator, bytes); + defer file.deinit(); + const entry = file.find("Workbook").?; + file.fat[entry.start_sector] = end_of_chain; + try testing.expectError(error.CorruptFile, file.readStream(allocator, entry)); + + var mini = entry; + mini.size = 10; // below the cutoff, so read through the (empty) mini stream + try testing.expectError(error.CorruptFile, file.readStream(allocator, mini)); +} + +test "a stream larger than the file is rejected before allocating" { + const allocator = testing.allocator; + const big = try allocator.alloc(u8, 5000); + defer allocator.free(big); + @memset(big, 1); + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{}); + defer allocator.free(bytes); + + var file = try File.init(allocator, bytes); + defer file.deinit(); + var entry = file.find("Workbook").?; + entry.size = 1 << 40; + try testing.expectError(error.CorruptFile, file.readStream(allocator, entry)); +} + +test "a mini chain pointing outside the mini stream is corrupt" { + const allocator = testing.allocator; + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "x" ** 100 }}, .{}); + defer allocator.free(bytes); + var file = try File.init(allocator, bytes); + defer file.deinit(); + const entry = file.find("S").?; + // The mini FAT has a full sector of entries (128) but the mini + // stream only holds two mini sectors, so entry 100 is in the table + // yet outside the stream. + file.mini_fat[entry.start_sector] = 100; + try testing.expectError(error.CorruptFile, file.readStream(allocator, entry)); +} + +test "a mini stream that ends mid-chain is corrupt" { + const allocator = testing.allocator; + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "x" ** 100 }}, .{}); + defer allocator.free(bytes); + var file = try File.init(allocator, bytes); + defer file.deinit(); + var entry = file.find("S").?; + // Start the chain at the second mini sector and pretend the mini + // stream ends 36 bytes into it: the first hop yields a short read + // that cannot be the end of a 100-byte stream. + entry.start_sector = 1; + const full_len = file.mini_stream.len; + file.mini_stream.len = 100; + defer file.mini_stream.len = full_len; + try testing.expectError(error.CorruptFile, file.readStream(allocator, entry)); +} + +test "a file cut short mid-stream reports Truncated" { + const allocator = testing.allocator; + const big = try allocator.alloc(u8, 20_000); + defer allocator.free(big); + @memset(big, 3); + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{}); + defer allocator.free(bytes); + + // The writer places the stream last, so dropping the tail cuts it. + var file = try File.init(allocator, bytes); + defer file.deinit(); + const cut_at = bytes.len - 4096; + file.bytes = bytes[0..cut_at]; + try testing.expectError(error.Truncated, file.readStream(allocator, file.find("Workbook").?)); + + // A cut that lands inside a sector leaves a short sector that is not + // the stream's last: also Truncated. + file.bytes = bytes[0 .. cut_at + 100]; + try testing.expectError(error.Truncated, file.readStream(allocator, file.find("Workbook").?)); +} + +test "an unpadded final sector is accepted when it ends the stream" { + const allocator = testing.allocator; + const big = try allocator.alloc(u8, 5000); + defer allocator.free(big); + @memset(big, 9); + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{}); + defer allocator.free(bytes); + + // 5000 bytes = 9 full sectors + 392 bytes; drop the padding. + const unpadded = bytes[0 .. bytes.len - (512 - 392)]; + var file = try File.init(allocator, unpadded); + defer file.deinit(); + const got = try file.readStream(allocator, file.find("Workbook").?); + defer allocator.free(got); + try testing.expectEqualSlices(u8, big, got); +} + +test "a directory entry with an impossible name length is corrupt" { + var raw: [dir_entry_size]u8 = @splat(0); + std.mem.writeInt(u16, raw[64..66], 66, .little); + try testing.expectError(error.CorruptFile, parseEntry(&raw, true)); + std.mem.writeInt(u16, raw[64..66], 3, .little); + try testing.expectError(error.CorruptFile, parseEntry(&raw, true)); +} + +test "version 3 entries ignore the high half of the size" { + var raw: [dir_entry_size]u8 = @splat(0); + std.mem.writeInt(u64, raw[120..128], 0xDEADBEEF_00000010, .little); + try testing.expectEqual(@as(u64, 0x10), (try parseEntry(&raw, true)).size); + try testing.expectEqual(@as(u64, 0xDEADBEEF_00000010), (try parseEntry(&raw, false)).size); +} + +test "the root entry must come first" { + const allocator = testing.allocator; + const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{}); + defer allocator.free(bytes); + // Root is the first entry of the first directory sector, which the + // writer places right after the FAT. Retype it as a stream. + const dir_off = (@as(usize, readU32(bytes, 48)) + 1) * 512; + bytes[dir_off + 66] = @intFromEnum(EntryType.stream); + try testing.expectError(error.CorruptFile, File.init(allocator, bytes)); +} diff --git a/src/root.zig b/src/root.zig new file mode 100644 index 0000000..6d02b97 --- /dev/null +++ b/src/root.zig @@ -0,0 +1,183 @@ +//! biff8: a read-only reader for legacy binary Excel workbooks (`.xls`, +//! Excel 97 through 2003). +//! +//! These files are BIFF8 record streams inside a Compound File Binary +//! ("OLE2") container. This library unwraps the container and decodes +//! each worksheet's cell values: text, numbers, booleans, error codes, +//! and the cached results of formulas. It does not decode formatting, +//! evaluate formulas, read `.xlsx` (a different format entirely), or +//! write anything. +//! +//! ```zig +//! var wb = try biff8.Workbook.parse(allocator, bytes); +//! defer wb.deinit(); +//! const sheet = wb.sheets[0]; +//! switch (sheet.cell(row, col)) { +//! .text => |t| ..., +//! .number => |n| ..., +//! else => {}, +//! } +//! ``` +//! +//! Numbers are returned exactly as stored, so a date-formatted cell is +//! an Excel serial day number. Converting it requires knowing the cell +//! is a date, which lives in the number format this library skips. + +const std = @import("std"); +const cfb = @import("cfb.zig"); +const biff = @import("biff.zig"); + +pub const Cell = biff.Cell; +pub const Sheet = biff.Sheet; + +pub const ParseError = cfb.Error || biff.Error || error{ + /// The container holds no `Workbook` stream, so it is some other + /// kind of compound file (a `.doc`, an `.msg`, ...). + NoWorkbookStream, +}; + +/// True when `bytes` starts with the compound-file signature. Every +/// `.xls` does, but so do other legacy Office files; use it for cheap +/// content sniffing, not as proof the bytes are a workbook. +pub const isCompoundFile = cfb.isCompoundFile; + +pub const Workbook = struct { + arena: std.heap.ArenaAllocator, + /// Worksheets in workbook order. Chart sheets, macro sheets and VB + /// modules are omitted. + sheets: []const Sheet, + + /// Parse a workbook. `bytes` is only read during the call; the + /// result owns everything it references. + pub fn parse(allocator: std.mem.Allocator, bytes: []const u8) ParseError!Workbook { + var file = try cfb.File.init(allocator, bytes); + defer file.deinit(); + + const entry = file.find("Workbook") orelse { + // Excel 5 and 95 (BIFF5) named the stream "Book". + if (file.find("Book") != null) return error.UnsupportedBiffVersion; + return error.NoWorkbookStream; + }; + const stream = try file.readStream(allocator, entry); + defer allocator.free(stream); + + var arena = std.heap.ArenaAllocator.init(allocator); + errdefer arena.deinit(); + const sheets = try biff.parseSheets(arena.allocator(), allocator, stream); + return .{ .arena = arena, .sheets = sheets }; + } + + pub fn deinit(self: *Workbook) void { + self.arena.deinit(); + } + + /// The first worksheet named exactly `name`. + pub fn sheet(self: *const Workbook, name: []const u8) ?*const Sheet { + for (self.sheets) |*s| { + if (std.mem.eql(u8, s.name, name)) return s; + } + return null; + } +}; + +// ---- Tests ---- + +const testing = std.testing; +const tw = @import("test_writer.zig"); + +test { + std.testing.refAllDecls(@This()); + _ = cfb; + _ = biff; +} + +test "parse: end to end through the compound file" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + + const sst = try tw.sst(a, &.{ .{ .text = "Symbol" }, .{ .text = "SAMPLE" } }, 8224); + const bytes = try tw.xls(a, .{ + .globals = sst, + .sheets = &.{ + .{ .name = "Sample_Positions", .records = &.{ + try tw.labelSst(a, 0, 0, 0), + try tw.labelSst(a, 1, 0, 1), + try tw.number(a, 1, 1, 100.25), + } }, + .{ .name = "Notes" }, + }, + }); + try testing.expect(isCompoundFile(bytes)); + + var wb = try Workbook.parse(testing.allocator, bytes); + defer wb.deinit(); + try testing.expectEqual(@as(usize, 2), wb.sheets.len); + const s = wb.sheet("Sample_Positions").?; + try testing.expectEqualStrings("Symbol", s.cell(0, 0).asText().?); + try testing.expectEqualStrings("SAMPLE", s.cell(1, 0).asText().?); + try testing.expectEqual(@as(f64, 100.25), s.cell(1, 1).asNumber().?); + try testing.expect(wb.sheet("Notes") != null); + try testing.expect(wb.sheet("notes") == null); +} + +test "parse: a large workbook stream lives outside the mini stream" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + + var records: std.ArrayList(tw.Record) = .empty; + for (0..1000) |i| try records.append(a, try tw.number(a, @intCast(i), 0, @floatFromInt(i))); + const bytes = try tw.xls(a, .{ .sheets = &.{.{ .name = "Big", .records = records.items }} }); + + var wb = try Workbook.parse(testing.allocator, bytes); + defer wb.deinit(); + try testing.expectEqual(@as(usize, 1000), wb.sheets[0].rows.len); + try testing.expectEqual(@as(f64, 999), wb.sheets[0].cell(999, 0).asNumber().?); +} + +test "parse: compound files that are not BIFF8 workbooks" { + const allocator = testing.allocator; + + const doc = try tw.buildCfb(allocator, &.{.{ .name = "WordDocument", .data = "not a workbook" }}, .{}); + defer allocator.free(doc); + try testing.expectError(error.NoWorkbookStream, Workbook.parse(allocator, doc)); + + const biff5 = try tw.buildCfb(allocator, &.{.{ .name = "Book", .data = "excel 95" }}, .{}); + defer allocator.free(biff5); + try testing.expectError(error.UnsupportedBiffVersion, Workbook.parse(allocator, biff5)); + + try testing.expectError(error.NotCompoundFile, Workbook.parse(allocator, "PK\x03\x04 xlsx is a zip")); +} + +test "parse: errors after the container is read release everything" { + // testing.allocator fails the test on any leak along the error path. + const allocator = testing.allocator; + const bytes = try tw.buildCfb(allocator, &.{.{ .name = "Workbook", .data = "\x09\x08\x02\x00" }}, .{}); + defer allocator.free(bytes); + try testing.expectError(error.CorruptFile, Workbook.parse(allocator, bytes)); +} + +fn parseAndRelease(allocator: std.mem.Allocator, bytes: []const u8) !void { + var wb = try Workbook.parse(allocator, bytes); + wb.deinit(); +} + +test "parse: every allocation failure is clean" { + var fx = std.heap.ArenaAllocator.init(testing.allocator); + defer fx.deinit(); + const a = fx.allocator(); + + // Small (mini stream) and large (regular sectors) workbooks take + // different allocation paths through the container reader. + const sst = try tw.sst(a, &.{ .{ .text = "alpha" }, .{ .text = "\u{3042}" } }, 8224); + var records: std.ArrayList(tw.Record) = .empty; + try records.append(a, try tw.labelSst(a, 0, 0, 0)); + try records.append(a, try tw.labelSst(a, 0, 1, 1)); + const small = try tw.xls(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} }); + for (1..600) |i| try records.append(a, try tw.number(a, @intCast(i), 0, 1)); + const large = try tw.xls(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} }); + + try testing.checkAllAllocationFailures(testing.allocator, parseAndRelease, .{small}); + try testing.checkAllAllocationFailures(testing.allocator, parseAndRelease, .{large}); +} diff --git a/src/test_writer.zig b/src/test_writer.zig new file mode 100644 index 0000000..b8fe655 --- /dev/null +++ b/src/test_writer.zig @@ -0,0 +1,508 @@ +//! Test-only builders for compound files and BIFF8 record streams. +//! +//! The reader's tests describe their fixtures in code with these +//! helpers instead of checking in binary `.xls` files. That keeps every +//! fixture readable and placeholder-only, and lets a test produce +//! layouts that real writers rarely emit (DIFAT sectors, strings split +//! across CONTINUE records at every possible point, unpaired UTF-16 +//! surrogates). +//! +//! Nothing here is validated against a second implementation; the +//! reader's tests against it check self-consistency. The real-world +//! check is parsing an actual Excel-produced file, which lives outside +//! the test suite because such files carry personal data. + +const std = @import("std"); +const Allocator = std.mem.Allocator; + +// ---- Compound file ---- + +pub const Stream = struct { + name: []const u8, + data: []const u8, +}; + +pub const CfbOptions = struct { + /// 3 (512-byte sectors) or 4 (4096-byte sectors). + major_version: u16 = 3, +}; + +const free_sect: u32 = 0xFFFFFFFF; +const end_of_chain: u32 = 0xFFFFFFFE; +const fat_sect: u32 = 0xFFFFFFFD; +const difat_sect: u32 = 0xFFFFFFFC; +const no_stream: u32 = 0xFFFFFFFF; +const mini_cutoff = 4096; +const mini_sector_size = 64; + +/// Build a compound file holding `streams` in its root storage. +/// Streams under 4096 bytes go in the mini stream, larger ones get +/// their own sector chains. Layout, in sector order: FAT, DIFAT, +/// directory, mini FAT, mini stream, then each large stream (so the +/// last stream's data ends the file). +pub fn buildCfb(allocator: Allocator, streams: []const Stream, opts: CfbOptions) ![]u8 { + const shift: u5 = switch (opts.major_version) { + 3 => 9, + 4 => 12, + else => unreachable, + }; + const sector_size: usize = @as(usize, 1) << shift; + const per_fat = sector_size / 4; + + // Mini stream contents and per-stream placement. + var mini: std.ArrayList(u8) = .empty; + defer mini.deinit(allocator); + const starts = try allocator.alloc(u32, streams.len); + defer allocator.free(starts); + var mini_sectors: usize = 0; + var regular_sectors: usize = 0; + for (streams, 0..) |s, i| { + if (s.data.len < mini_cutoff) { + const n = ceilDiv(s.data.len, mini_sector_size); + starts[i] = if (n == 0) end_of_chain else @intCast(mini_sectors); + mini_sectors += n; + try mini.appendSlice(allocator, s.data); + try mini.appendNTimes(allocator, 0, n * mini_sector_size - s.data.len); + } else { + regular_sectors += ceilDiv(s.data.len, sector_size); + } + } + + const dir_sectors = ceilDiv((streams.len + 1) * 128, sector_size); + const mini_fat_sectors = ceilDiv(mini_sectors * 4, sector_size); + const mini_stream_sectors = ceilDiv(mini.items.len, sector_size); + const data_sectors = dir_sectors + mini_fat_sectors + mini_stream_sectors + regular_sectors; + + // Smallest FAT (plus the DIFAT sectors it needs) that addresses + // every sector including itself. + var fat_sectors: usize = 1; + var difat_sectors: usize = 0; + while (true) : (fat_sectors += 1) { + difat_sectors = if (fat_sectors > 109) ceilDiv(fat_sectors - 109, per_fat - 1) else 0; + if (fat_sectors * per_fat >= data_sectors + fat_sectors + difat_sectors) break; + } + + const total_sectors = fat_sectors + difat_sectors + data_sectors; + const out = try allocator.alloc(u8, (total_sectors + 1) * sector_size); + errdefer allocator.free(out); + @memset(out, 0); + + const fat = try allocator.alloc(u32, fat_sectors * per_fat); + defer allocator.free(fat); + @memset(fat, free_sect); + + var next: usize = 0; + const fat_first = next; + for (0..fat_sectors) |i| fat[fat_first + i] = fat_sect; + next += fat_sectors; + const difat_first = next; + for (0..difat_sectors) |i| fat[difat_first + i] = difat_sect; + next += difat_sectors; + const dir_first = chain(fat, &next, dir_sectors); + const mini_fat_first = chain(fat, &next, mini_fat_sectors); + const mini_stream_first = chain(fat, &next, mini_stream_sectors); + for (streams, 0..) |s, i| { + if (s.data.len >= mini_cutoff) starts[i] = chain(fat, &next, ceilDiv(s.data.len, sector_size)); + } + + // Header. + const hdr = out[0..512]; + @memcpy(hdr[0..8], &[_]u8{ 0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1 }); + put16(hdr, 24, 0x003E); + put16(hdr, 26, opts.major_version); + put16(hdr, 28, 0xFFFE); + put16(hdr, 30, shift); + put16(hdr, 32, 6); + put32(hdr, 40, if (opts.major_version == 4) @intCast(dir_sectors) else 0); + put32(hdr, 44, @intCast(fat_sectors)); + put32(hdr, 48, dir_first); + put32(hdr, 56, mini_cutoff); + put32(hdr, 60, mini_fat_first); + put32(hdr, 64, @intCast(mini_fat_sectors)); + put32(hdr, 68, if (difat_sectors > 0) @intCast(difat_first) else end_of_chain); + put32(hdr, 72, @intCast(difat_sectors)); + for (0..109) |i| put32(hdr, 76 + 4 * i, if (i < fat_sectors) @intCast(fat_first + i) else free_sect); + + // DIFAT sectors: FAT sector numbers past the first 109, last slot + // links to the next DIFAT sector. + for (0..difat_sectors) |d| { + const s = sectorBytes(out, sector_size, difat_first + d); + for (0..per_fat - 1) |i| { + const n = 109 + d * (per_fat - 1) + i; + put32(s, 4 * i, if (n < fat_sectors) @intCast(fat_first + n) else free_sect); + } + put32(s, sector_size - 4, if (d + 1 < difat_sectors) @intCast(difat_first + d + 1) else end_of_chain); + } + + // FAT. + for (fat, 0..) |v, i| put32(out[sector_size..], 4 * i, v); + + // Directory: root, then each stream linked as a right-sibling list. + { + const dir = try allocator.alloc(u8, dir_sectors * sector_size); + defer allocator.free(dir); + @memset(dir, 0); + var e: usize = 0; + while (e * 128 < dir.len) : (e += 1) { + const raw = dir[e * 128 ..][0..128]; + put32(raw, 68, no_stream); + put32(raw, 72, no_stream); + put32(raw, 76, no_stream); + } + writeEntry(dir[0..128], "Root Entry", 5, if (streams.len > 0) 1 else no_stream, no_stream, if (mini.items.len > 0) mini_stream_first else end_of_chain, mini.items.len); + for (streams, 0..) |s, i| { + const right: u32 = if (i + 1 < streams.len) @intCast(i + 2) else no_stream; + writeEntry(dir[(i + 1) * 128 ..][0..128], s.name, 2, no_stream, right, starts[i], s.data.len); + } + writeChain(out, sector_size, dir_first, dir); + } + + // Mini FAT and mini stream. + { + const table = try allocator.alloc(u8, mini_fat_sectors * sector_size); + defer allocator.free(table); + @memset(table, 0xFF); + var idx: usize = 0; + for (streams) |s| { + if (s.data.len >= mini_cutoff) continue; + const n = ceilDiv(s.data.len, mini_sector_size); + for (0..n) |k| put32(table, 4 * (idx + k), if (k + 1 < n) @intCast(idx + k + 1) else end_of_chain); + idx += n; + } + writeChain(out, sector_size, mini_fat_first, table); + writeChain(out, sector_size, mini_stream_first, mini.items); + } + + for (streams, 0..) |s, i| { + if (s.data.len >= mini_cutoff) writeChain(out, sector_size, starts[i], s.data); + } + return out; +} + +/// Allocate `n` consecutive sectors as one chain. Returns the first +/// sector, or ENDOFCHAIN for an empty chain. +fn chain(fat: []u32, next: *usize, n: usize) u32 { + if (n == 0) return end_of_chain; + const first = next.*; + for (0..n) |i| fat[first + i] = if (i + 1 < n) @intCast(first + i + 1) else end_of_chain; + next.* += n; + return @intCast(first); +} + +/// Copy `data` into consecutive sectors starting at `first` (the +/// builder always allocates chains contiguously). +fn writeChain(out: []u8, sector_size: usize, first: u32, data: []const u8) void { + if (data.len == 0) return; + const off = (@as(usize, first) + 1) * sector_size; + @memcpy(out[off..][0..data.len], data); +} + +fn sectorBytes(out: []u8, sector_size: usize, index: usize) []u8 { + return out[(index + 1) * sector_size ..][0..sector_size]; +} + +fn writeEntry(raw: *[128]u8, name: []const u8, kind: u8, child: u32, right: u32, start: u32, size: usize) void { + for (name, 0..) |c, i| put16(raw, 2 * i, c); + put16(raw, 64, @intCast((name.len + 1) * 2)); + raw[66] = kind; + raw[67] = 1; // black + put32(raw, 72, right); + put32(raw, 76, child); + put32(raw, 116, start); + std.mem.writeInt(u64, raw[120..128], size, .little); +} + +// ---- BIFF8 records ---- + +pub const Record = struct { + kind: u16, + data: []const u8, +}; + +pub const rt = struct { + pub const bof: u16 = 0x0809; + pub const eof: u16 = 0x000A; + pub const filepass: u16 = 0x002F; + pub const boundsheet: u16 = 0x0085; + pub const sst: u16 = 0x00FC; + pub const @"continue": u16 = 0x003C; + pub const labelsst: u16 = 0x00FD; + pub const number: u16 = 0x0203; + pub const rk: u16 = 0x027E; + pub const mulrk: u16 = 0x00BD; + pub const label: u16 = 0x0204; + pub const rstring: u16 = 0x00D6; + pub const boolerr: u16 = 0x0205; + pub const formula: u16 = 0x0006; + pub const string: u16 = 0x0207; + pub const blank: u16 = 0x0201; + pub const dimensions: u16 = 0x0200; +}; + +pub const SheetSpec = struct { + name: []const u8, + records: []const Record = &.{}, + /// BOUNDSHEET8 `dt`: 0 worksheet, 2 chart, 6 VB module. + boundsheet_type: u8 = 0, + /// BOF `dt` for the substream; 0x0010 worksheet, 0x0020 chart. + bof_type: u16 = 0x0010, +}; + +pub const WorkbookSpec = struct { + version: u16 = 0x0600, + /// Records between the globals BOF and the BOUNDSHEET records + /// (SST, FILEPASS, ...). + globals: []const Record = &.{}, + sheets: []const SheetSpec = &.{}, + /// Leave the final sheet without its EOF record. + omit_last_eof: bool = false, + /// Overrides every BOUNDSHEET offset (for out-of-range tests). + offset_override: ?u32 = null, +}; + +/// Assemble a Workbook stream: globals BOF, `globals`, one BOUNDSHEET +/// per sheet (offsets patched once the substreams are placed), EOF, +/// then each sheet substream. +pub fn workbook(allocator: Allocator, spec: WorkbookSpec) ![]u8 { + var out: std.ArrayList(u8) = .empty; + errdefer out.deinit(allocator); + + try appendRecord(allocator, &out, rt.bof, &bof(spec.version, 0x0005)); + for (spec.globals) |r| try appendRecord(allocator, &out, r.kind, r.data); + + const offset_slots = try allocator.alloc(usize, spec.sheets.len); + defer allocator.free(offset_slots); + for (spec.sheets, 0..) |s, i| { + var data: std.ArrayList(u8) = .empty; + defer data.deinit(allocator); + try data.appendNTimes(allocator, 0, 4); // lbPlyPos, patched below + try data.append(allocator, 0); // visible + try data.append(allocator, s.boundsheet_type); + try data.append(allocator, @intCast(s.name.len)); + try data.append(allocator, 0); // compressed characters + try data.appendSlice(allocator, s.name); + offset_slots[i] = out.items.len + 4; + try appendRecord(allocator, &out, rt.boundsheet, data.items); + } + try appendRecord(allocator, &out, rt.eof, ""); + + for (spec.sheets, 0..) |s, i| { + const offset: u32 = spec.offset_override orelse @intCast(out.items.len); + put32(out.items, offset_slots[i], offset); + try appendRecord(allocator, &out, rt.bof, &bof(spec.version, s.bof_type)); + for (s.records) |r| try appendRecord(allocator, &out, r.kind, r.data); + if (!(spec.omit_last_eof and i + 1 == spec.sheets.len)) try appendRecord(allocator, &out, rt.eof, ""); + } + return out.toOwnedSlice(allocator); +} + +/// Workbook stream wrapped in a compound file as the `Workbook` stream. +pub fn xls(allocator: Allocator, spec: WorkbookSpec) ![]u8 { + const stream = try workbook(allocator, spec); + defer allocator.free(stream); + return buildCfb(allocator, &.{.{ .name = "Workbook", .data = stream }}, .{}); +} + +fn appendRecord(allocator: Allocator, out: *std.ArrayList(u8), kind: u16, data: []const u8) !void { + // SAFETY: both halves are written immediately below. + var head: [4]u8 = undefined; + put16(&head, 0, kind); + put16(&head, 2, @intCast(data.len)); + try out.appendSlice(allocator, &head); + try out.appendSlice(allocator, data); +} + +pub fn bof(version: u16, dt: u16) [16]u8 { + var b: [16]u8 = @splat(0); + put16(&b, 0, version); + put16(&b, 2, dt); + return b; +} + +// The single-cell builders allocate their record bodies so the +// returned `Record` stays valid; tests pass an arena. + +pub fn labelSst(allocator: Allocator, row: u16, col: u16, index: u32) !Record { + return cellRecord(allocator, rt.labelsst, row, col, u32, index); +} + +pub fn number(allocator: Allocator, row: u16, col: u16, value: f64) !Record { + return cellRecord(allocator, rt.number, row, col, u64, @bitCast(value)); +} + +pub fn rk(allocator: Allocator, row: u16, col: u16, raw: u32) !Record { + return cellRecord(allocator, rt.rk, row, col, u32, raw); +} + +pub fn boolErr(allocator: Allocator, row: u16, col: u16, value: u8, is_error: bool) !Record { + return cellRecord(allocator, rt.boolerr, row, col, u16, @as(u16, value) | (@as(u16, @intFromBool(is_error)) << 8)); +} + +pub fn blank(allocator: Allocator, row: u16, col: u16) !Record { + return cellRecord(allocator, rt.blank, row, col, void, {}); +} + +/// FORMULA record with an 8-byte cached value (see `formulaNumber`, +/// `formulaSpecial`) and an empty expression. +pub fn formula(allocator: Allocator, row: u16, col: u16, value: [8]u8) !Record { + const b = try allocator.alloc(u8, 6 + 8 + 2 + 4 + 2); + @memset(b, 0); + put16(b, 0, row); + put16(b, 2, col); + @memcpy(b[6..14], &value); + return .{ .kind = rt.formula, .data = b }; +} + +pub fn formulaNumber(value: f64) [8]u8 { + // SAFETY: writeInt fills all eight bytes. + var v: [8]u8 = undefined; + std.mem.writeInt(u64, &v, @bitCast(value), .little); + return v; +} + +/// Non-numeric cached formula result: 0 string (in a following STRING +/// record), 1 boolean, 2 error, 3 empty string. +pub fn formulaSpecial(kind: u8, payload: u8) [8]u8 { + return .{ kind, 0, payload, 0, 0, 0, 0xFF, 0xFF }; +} + +/// MULRK: consecutive RK cells starting at `first_col`. +pub fn mulrk(allocator: Allocator, row: u16, first_col: u16, raws: []const u32) !Record { + const b = try allocator.alloc(u8, 4 + 6 * raws.len + 2); + put16(b, 0, row); + put16(b, 2, first_col); + for (raws, 0..) |r, i| { + put16(b, 4 + 6 * i, 0); + put32(b, 4 + 6 * i + 2, r); + } + put16(b, b.len - 2, first_col + @as(u16, @intCast(raws.len)) - 1); + return .{ .kind = rt.mulrk, .data = b }; +} + +/// LABEL / RSTRING with an inline compressed (Latin-1) string. +pub fn label(allocator: Allocator, kind: u16, row: u16, col: u16, text: []const u8) !Record { + const b = try allocator.alloc(u8, 6 + 3 + text.len); + @memset(b, 0); + put16(b, 0, row); + put16(b, 2, col); + put16(b, 6, @intCast(text.len)); + @memcpy(b[9..], text); + return .{ .kind = kind, .data = b }; +} + +/// STRING record (formula string result), compressed characters. +pub fn string(allocator: Allocator, text: []const u8) !Record { + const b = try allocator.alloc(u8, 3 + text.len); + put16(b, 0, @intCast(text.len)); + b[2] = 0; + @memcpy(b[3..], text); + return .{ .kind = rt.string, .data = b }; +} + +fn cellRecord(allocator: Allocator, kind: u16, row: u16, col: u16, comptime T: type, value: T) !Record { + const b = try allocator.alloc(u8, 6 + @sizeOf(T)); + @memset(b, 0); + put16(b, 0, row); + put16(b, 2, col); + if (T != void) std.mem.writeInt(T, b[6..][0..@sizeOf(T)], value, .little); + return .{ .kind = kind, .data = b }; +} + +pub const SstString = struct { + /// UTF-8 text. Ignored when `units` is set. + text: []const u8 = "", + /// Raw UTF-16 code units, for strings UTF-8 cannot express (an + /// unpaired surrogate). + units: ?[]const u16 = null, + /// Rich-text run count; the runs themselves are filler bytes. + runs: u16 = 0, + /// Phonetic (ExtRst) bytes. + ext: []const u8 = "", +}; + +/// SST plus CONTINUE records, split so no record body exceeds +/// `max_chunk` bytes. Splits land wherever the limit falls: inside +/// a string's header, characters, runs or ExtRst. A split inside the +/// characters starts the next record with a fresh high-byte flag +/// chosen for the remaining characters, as Excel does, so one string +/// can switch between compressed and UTF-16 storage mid-way. +pub fn sst(allocator: Allocator, strings: []const SstString, max_chunk: usize) ![]Record { + var w: ChunkWriter = .{ .allocator = allocator, .max = max_chunk }; + defer w.chunks.deinit(allocator); + + try w.int(u32, @intCast(strings.len)); + try w.int(u32, @intCast(strings.len)); + for (strings) |s| { + const owned = if (s.units == null) try std.unicode.utf8ToUtf16LeAlloc(allocator, s.text) else null; + defer if (owned) |o| allocator.free(o); + const units = s.units orelse owned.?; + + var flags: u8 = if (anyHigh(units)) 1 else 0; + if (s.ext.len > 0) flags |= 0x04; + if (s.runs > 0) flags |= 0x08; + try w.int(u16, @intCast(units.len)); + try w.byte(flags); + if (s.runs > 0) try w.int(u16, s.runs); + if (s.ext.len > 0) try w.int(u32, @intCast(s.ext.len)); + + var high = flags & 1 != 0; + for (units, 0..) |u, i| { + const width: usize = if (high) 2 else 1; + if (w.current.items.len + width > w.max) { + try w.finish(); + high = anyHigh(units[i..]); + try w.current.append(allocator, @intFromBool(high)); + } + if (high) { + try w.current.append(allocator, @truncate(u)); + try w.current.append(allocator, @truncate(u >> 8)); + } else { + try w.current.append(allocator, @intCast(u)); + } + } + for (0..@as(usize, s.runs) * 4) |_| try w.byte(0xAB); + for (s.ext) |b| try w.byte(b); + } + try w.finish(); + + const records = try allocator.alloc(Record, w.chunks.items.len); + for (w.chunks.items, 0..) |c, i| records[i] = .{ .kind = if (i == 0) rt.sst else rt.@"continue", .data = c }; + return records; +} + +const ChunkWriter = struct { + allocator: Allocator, + max: usize, + current: std.ArrayList(u8) = .empty, + chunks: std.ArrayList([]u8) = .empty, + + fn byte(w: *ChunkWriter, b: u8) !void { + if (w.current.items.len == w.max) try w.finish(); + try w.current.append(w.allocator, b); + } + + fn int(w: *ChunkWriter, comptime T: type, v: T) !void { + for (0..@sizeOf(T)) |i| try w.byte(@truncate(v >> @intCast(8 * i))); + } + + fn finish(w: *ChunkWriter) !void { + try w.chunks.append(w.allocator, try w.current.toOwnedSlice(w.allocator)); + } +}; + +fn anyHigh(units: []const u16) bool { + for (units) |u| if (u > 0xFF) return true; + return false; +} + +fn ceilDiv(a: usize, b: usize) usize { + return (a + b - 1) / b; +} + +fn put16(b: []u8, off: usize, v: u16) void { + std.mem.writeInt(u16, b[off..][0..2], v, .little); +} + +fn put32(b: []u8, off: usize, v: u32) void { + std.mem.writeInt(u32, b[off..][0..4], v, .little); +}