initial vibe coded commit

This commit is contained in:
Emil Lerch 2026-10-03 12:30:20 -07:00
parent 6903039a2a
commit 4f48adaa18
Signed by: lobo
GPG key ID: A7B62D657EF764F8
14 changed files with 2880 additions and 0 deletions

5
.gitignore vendored Normal file
View file

@ -0,0 +1,5 @@
.zig-cache/
zig-out/
zig-pkg/
coverage/
.tmp/

7
.mise.toml Normal file
View file

@ -0,0 +1,7 @@
[tools]
zig = "0.16.0"
zls = "0.16.0"
"github:j178/prek" = "0.4.1"
[tools."github:DonIsaac/zlint"]
version = "0.9.0"

51
.pre-commit-config.yaml Normal file
View file

@ -0,0 +1,51 @@
# See https://pre-commit.com for more information
# See https://pre-commit.com/hooks.html for more hooks
repos:
- repo: https://github.com/pre-commit/pre-commit-hooks
rev: v6.0.0
hooks:
- id: trailing-whitespace
- id: end-of-file-fixer
- id: check-yaml
- id: check-added-large-files
- repo: local
hooks:
- id: forbid-ai-punctuation
name: Forbid smart punctuation (en/figure dash, minus, ellipsis, arrows, smart quotes)
language: pygrep
entry: '(–|‒|―|−|…|→|⇐|⇒|⇔|“|”|‘|’)'
files: '\.(zig|zon|md|txt|toml|ya?ml)$'
exclude: '^\.pre-commit-config\.yaml$'
- id: forbid-prose-em-dash
name: Forbid prose em-dash (use ASCII hyphen)
language: pygrep
entry: ' — '
files: '\.(zig|zon|md|txt|toml|ya?ml)$'
exclude: '^\.pre-commit-config\.yaml$'
- repo: https://github.com/batmac/pre-commit-zig
rev: v0.3.0
hooks:
- id: zig-fmt
- repo: local
hooks:
- id: zlint
name: Run zlint
# zlint accepts file paths only via stdin (-S); positional
# args are interpreted as directory names and silently
# produce no output.
entry: bash -c 'printf "%s\n" "$@" | zlint --deny-warnings --fix -S' --
language: system
types: [zig]
- repo: https://github.com/batmac/pre-commit-zig
rev: v0.3.0
hooks:
- id: zig-build
- repo: local
hooks:
- id: test
name: Run zig build coverage
entry: zig
args: ["build", "coverage", "-Dcoverage-threshold=99"]
language: system
types: [file]
pass_filenames: false

21
LICENSE Normal file
View file

@ -0,0 +1,21 @@
MIT License
Copyright (c) 2026 Emil Lerch
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.

93
README.md Normal file
View file

@ -0,0 +1,93 @@
# biff8
A read-only Zig reader for legacy binary Excel workbooks: `.xls` files from
Excel 97 through 2003 (BIFF8), which some sites still export as their only
"spreadsheet" download.
It unwraps the Compound File Binary ("OLE2") container, decodes the BIFF8
record stream, and gives you each worksheet's cell values. That is all it
does.
## Usage
```zig
const biff8 = @import("biff8");
var wb = try biff8.Workbook.parse(allocator, bytes);
defer wb.deinit();
const sheet = wb.sheet("Positions") orelse return error.NoSuchSheet;
for (sheet.rows, 0..) |row, r| {
for (row, 0..) |cell, c| switch (cell) {
.text => |t| std.debug.print("{d},{d}: {s}\n", .{ r, c, t }),
.number => |n| std.debug.print("{d},{d}: {d}\n", .{ r, c, n }),
.boolean, .error_code, .empty => {},
};
}
```
`sheet.cell(row, col)` does bounds-safe random access and returns `.empty`
outside the populated area. `Sheet` has public fields, so a consumer can
build one as a literal in its own tests instead of shipping binary fixtures.
`biff8.isCompoundFile(bytes)` is a cheap signature check for content
sniffing. Every `.xls` passes it, but so does every other legacy Office
file, so it is not proof of a workbook.
## What is decoded
| Record | Becomes |
|---|---|
| LABELSST, LABEL, RSTRING | `.text` (UTF-8) |
| NUMBER, RK, MULRK | `.number` |
| BOOLERR | `.boolean` or `.error_code` |
| FORMULA (+ STRING) | the cached result, as any of the above |
Shared strings split across CONTINUE records are handled, including the
case where the split falls mid-string and the remainder switches between
1-byte and 2-byte characters. Unpaired UTF-16 surrogates decode as U+FFFD.
## What is not
- **Formatting.** Numbers are returned as stored, so a date-formatted cell
is an Excel serial day number; telling dates apart needs the number
format, which is not decoded.
- **Formulas.** Only their cached results.
- **Anything but BIFF8.** Excel 95 and earlier fail with
`error.UnsupportedBiffVersion`. `.xlsx` is a different format (zipped
XML) and fails with `error.NotCompoundFile`.
- **Encrypted workbooks.** `error.Encrypted`.
- **Writing.**
## Errors
`Workbook.parse` returns `biff8.ParseError`:
| Error | Meaning |
|---|---|
| `NotCompoundFile` | Not an OLE2 file at all |
| `NoWorkbookStream` | An OLE2 file, but not a workbook (a `.doc`, an `.msg`, ...) |
| `UnsupportedBiffVersion` | Excel 95 or older |
| `Encrypted` | Password-protected |
| `Truncated` | The file ends early, usually an interrupted download |
| `CorruptFile` | Internally inconsistent structure |
| `OutOfMemory` | |
Every sector chain walk is bounded and every declared size is checked
before allocating, so hostile input produces an error rather than a hang or
a huge allocation.
## Specs
- [MS-CFB] Compound File Binary File Format
- [MS-XLS] Excel Binary File Format (.xls) Structure
## Development
```sh
zig build test
zig build coverage # kcov, Linux x86_64/aarch64
```
Test fixtures are built in code by `src/test_writer.zig` rather than checked
in as binary files.

47
build.zig Normal file
View file

@ -0,0 +1,47 @@
const std = @import("std");
const Coverage = @import("build/Coverage.zig");
pub fn build(b: *std.Build) void {
const target = b.standardTargetOptions(.{});
const optimize = b.standardOptimizeOption(.{});
// The public module. Consumers `@import("biff8")`.
const mod = b.addModule("biff8", .{
.root_source_file = b.path("src/root.zig"),
.target = target,
.optimize = optimize,
});
// Tests: one binary rooted at src/root.zig. `refAllDecls` in
// root.zig's test block pulls in every file's tests through the
// import graph.
const tests = b.addTest(.{ .root_module = mod });
const test_step = b.step("test", "Run all tests");
test_step.dependOn(&b.addRunArtifact(tests).step);
const lib = b.addLibrary(.{
.name = "biff8",
.root_module = b.createModule(.{
.root_source_file = b.path("src/root.zig"),
.target = target,
.optimize = optimize,
}),
});
const docs_step = b.step("docs", "Generate documentation");
docs_step.dependOn(&b.addInstallDirectory(.{
.source_dir = lib.getEmittedDocs(),
.install_dir = .prefix,
.install_subdir = "docs",
}).step);
// Coverage: `zig build coverage` (kcov, Linux x86_64/aarch64 only)
{
var cov = Coverage.init(b);
const cov_mod = b.createModule(.{
.root_source_file = b.path("src/root.zig"),
.target = target,
.optimize = optimize,
});
_ = cov.addModule(cov_mod, "biff8");
}
}

14
build.zig.zon Normal file
View file

@ -0,0 +1,14 @@
.{
.name = .biff8,
.version = "0.0.0",
.fingerprint = 0x3401a894ea1f9c5a, // Changing this has security and trust implications.
.minimum_zig_version = "0.16.0",
.dependencies = .{},
.paths = .{
"build.zig",
"build.zig.zon",
"src",
"LICENSE",
"README.md",
},
}

239
build/Coverage.zig Normal file
View file

@ -0,0 +1,239 @@
const builtin = @import("builtin");
const std = @import("std");
const Build = std.Build;
const Coverage = @This();
/// Whether the host platform supports kcov-based coverage.
/// Only x86_64 and aarch64 Linux are supported (kcov binary availability).
/// On unsupported platforms, the coverage step will fail at runtime with
/// a clear error from the kcov download or execution step.
// pub const supported = builtin.os.tag == .linux and
// (builtin.cpu.arch == .x86_64 or builtin.cpu.arch == .aarch64);
/// Initialize coverage infrastructure. Creates the "coverage" build step,
/// registers build options (-Dcoverage-threshold, -Dcoverage-dir),
/// and sets up the kcov download step. The kcov binary is downloaded into the
/// zig cache on first use and reused thereafter.
///
/// Use `zig build coverage --verbose` to see per-file coverage breakdown.
///
/// Call `addModule()` on the returned value to add the test module to the
/// coverage run.
///
/// Because addModule creates a new test executable from the root module provided,
/// if there are any linking steps being done to your test executable, those
/// must also be done to the test_exe returned by addModule.
pub fn init(b: *Build) Coverage {
// Add options
const coverage_threshold = b.option(u7, "coverage-threshold", "Minimum coverage percentage required") orelse 0;
const coverage_dir = b.option([]const u8, "coverage-dir", "Coverage output directory") orelse
b.pathJoin(&.{ b.build_root.path orelse ".", "coverage" });
const coverage_step = b.step("coverage", "Generate test coverage report");
// Set up kcov download.
// We can't download directly because we are sandboxed during build, but
// we can create a helper program and run it. First we need the destination
// directory, keyed by architecture.
const arch_name = switch (builtin.cpu.arch) {
.x86_64 => "x86_64",
.aarch64 => "aarch64",
else => @tagName(builtin.cpu.arch),
};
const Algo = std.crypto.hash.sha2.Sha256;
var hasher = Algo.init(.{});
hasher.update("kcov-");
hasher.update(arch_name);
var cache_hash: [Algo.digest_length]u8 = undefined;
hasher.final(&cache_hash);
const cache_dir = b.pathJoin(&.{
b.cache_root.path.?,
"o",
b.fmt("{s}", .{std.fmt.bytesToHex(cache_hash, .lower)}),
});
const kcov_path = b.pathJoin(&.{ cache_dir, b.fmt("kcov-{s}", .{arch_name}) });
// Create the download helper executable
const download_exe = b.addExecutable(.{
.name = "download-kcov",
.root_module = b.createModule(.{
.root_source_file = b.path("build/download_kcov.zig"),
.target = b.resolveTargetQuery(.{}),
}),
});
const run_download = b.addRunArtifact(download_exe);
run_download.addArg(kcov_path);
run_download.addArg(arch_name);
return .{
.b = b,
.coverage_step = coverage_step,
.coverage_dir = coverage_dir,
.coverage_threshold = coverage_threshold,
.kcov_path = kcov_path,
.run_download = run_download,
};
}
/// Add a test module to the coverage run. Runs kcov on the test binary,
/// then reads the coverage JSON and prints a summary (with per-file
/// breakdown if --verbose). Fails if below -Dcoverage-threshold.
///
/// Returns the test executable so the caller can add any extra linking steps.
pub fn addModule(self: *Coverage, root_module: *Build.Module, name: []const u8) *Build.Step.Compile {
const b = self.b;
// Set up kcov run: filter to src/ only, use custom CSS for HTML report
const run_coverage = b.addSystemCommand(&.{self.kcov_path});
const include_path = b.pathJoin(&.{ b.build_root.path.?, "src" });
run_coverage.addArgs(&.{ "--include-path", include_path });
const css_file = b.pathJoin(&.{ b.build_root.path.?, "build", "bcov.css" });
run_coverage.addArg(b.fmt("--configure=css-file={s}", .{css_file}));
run_coverage.addArg(self.coverage_dir);
// Create a test executable for this module.
// We need to set use_llvm because the self-hosted backend
// does not emit the DWARF data that kcov needs.
const test_exe = b.addTest(.{
.name = name,
.root_module = root_module,
.use_llvm = true,
});
run_coverage.addArtifactArg(test_exe);
run_coverage.step.dependOn(&test_exe.step);
run_coverage.step.dependOn(&self.run_download.step);
// Wire up the threshold check step after kcov completes
const check = b.allocator.create(Check) catch @panic("OOM");
check.* = .{
.step = Build.Step.init(.{
.id = .custom,
.name = "check coverage",
.owner = b,
.makeFn = make,
}),
.json_path = b.fmt("{s}/{s}/coverage.json", .{ self.coverage_dir, name }),
.threshold = self.coverage_threshold,
};
check.step.dependOn(&run_coverage.step);
self.coverage_step.dependOn(&check.step);
return test_exe;
}
// ── Coverage struct fields ──────────────────────────────────
// Fields used by init() to configure the shared coverage infrastructure
b: *Build,
coverage_step: *Build.Step,
coverage_dir: []const u8,
coverage_threshold: u7,
kcov_path: []const u8,
run_download: *Build.Step.Run,
// Per-module threshold-check step. Created in `addModule`; `make`
// recovers the instance via `@fieldParentPtr("step", ...)`.
const Check = struct {
step: Build.Step,
json_path: []const u8,
threshold: u7,
};
// This must be kept in step with kcov per-binary coverage.json format
const CoverageReport = struct {
files: []const CoverageFile,
};
const CoverageFile = struct {
file: []const u8,
covered_lines: usize,
total_lines: usize,
};
const File = struct {
file: []const u8,
percent_covered: f64,
covered_lines: usize,
total_lines: usize,
pub fn coverageLessThanDesc(_: void, lhs: File, rhs: File) bool {
return lhs.percent_covered > rhs.percent_covered;
}
};
/// Build step make function: reads kcov JSON output, prints a summary
/// (with per-file breakdown if verbose), and fails if below threshold.
fn make(step: *Build.Step, options: Build.Step.MakeOptions) !void {
_ = options;
const check: *Check = @fieldParentPtr("step", step);
const allocator = step.owner.allocator;
const io = step.owner.graph.io;
const file = std.Io.Dir.cwd().openFile(io, check.json_path, .{}) catch |err| {
return step.fail("Failed to open coverage report {s}: {}", .{ check.json_path, err });
};
defer file.close(io);
var file_reader = file.reader(io, &.{});
const content = try file_reader.interface.allocRemaining(allocator, .limited(10 * 1024 * 1024));
defer allocator.free(content);
const json = std.json.parseFromSlice(CoverageReport, allocator, content, .{
.ignore_unknown_fields = true,
}) catch |err| {
return step.fail("Failed to parse coverage JSON: {}", .{err});
};
defer json.deinit();
var total_covered: usize = 0;
var total_lines: usize = 0;
var file_list = std.ArrayList(File).empty;
defer file_list.deinit(allocator);
for (json.value.files) |f| {
const pct: f64 = if (f.total_lines > 0)
@as(f64, @floatFromInt(f.covered_lines)) / @as(f64, @floatFromInt(f.total_lines)) * 100.0
else
0;
try file_list.append(allocator, .{
.file = f.file,
.covered_lines = f.covered_lines,
.total_lines = f.total_lines,
.percent_covered = pct,
});
total_covered += f.covered_lines;
total_lines += f.total_lines;
}
std.mem.sort(File, file_list.items, {}, File.coverageLessThanDesc);
var stdout_buffer: [1024]u8 = undefined;
var stdout_writer = std.Io.File.stdout().writer(io, &stdout_buffer);
const stdout = &stdout_writer.interface;
if (step.owner.verbose) {
for (file_list.items) |f| {
try stdout.print(
"{d: >5.1}% {d: >5}/{d: <5}:{s}\n",
.{ f.percent_covered, f.covered_lines, f.total_lines, f.file },
);
}
}
const total_pct: f64 = if (total_lines > 0)
@as(f64, @floatFromInt(total_covered)) / @as(f64, @floatFromInt(total_lines)) * 100.0
else
0;
try stdout.print(
"Total test coverage: {d:.2}% ({d}/{d})\n",
.{ total_pct, total_covered, total_lines },
);
try stdout.flush();
if (@as(u7, @intFromFloat(@floor(total_pct))) < check.threshold)
return step.fail("Coverage {d:.2}% is below threshold {d}%", .{ total_pct, check.threshold });
}

46
build/bcov.css Normal file
View file

@ -0,0 +1,46 @@
/* Based upon the lcov CSS style, style files can be reused - Dark Theme */
body { color: #e0e0e0; background-color: #1e1e1e; }
a:link { color: #6b9aff; text-decoration: underline; }
a:visited { color: #4dbb7a; text-decoration: underline; }
a:active { color: #ff6b8a; text-decoration: underline; }
td.title { text-align: center; padding-bottom: 10px; font-size: 20pt; font-weight: bold; }
td.ruler { background-color: #4a6ba8; }
td.headerItem { text-align: right; padding-right: 6px; font-family: sans-serif; font-weight: bold; }
td.headerValue { text-align: left; color: #6b9aff; font-family: sans-serif; font-weight: bold; }
td.versionInfo { text-align: center; padding-top: 2px; }
th.headerItem { text-align: right; padding-right: 6px; font-family: sans-serif; font-weight: bold; }
th.headerValue { text-align: left; color: #6b9aff; font-family: sans-serif; font-weight: bold; }
pre.source { font-family: monospace; white-space: pre; overflow: hidden; text-overflow: ellipsis; }
span.lineNum { background-color: #5a5a2a; }
span.lineNumLegend { background-color: #5a5a2a; width: 96px; font-weight: bold ;}
span.lineCov { background-color: #2d5a2d; }
span.linePartCov { background-color: #707000; }
span.lineNoCov { background-color: #762c2c; }
span.orderNum { background-color: #5a4a2a; float: right; width:5em; text-align: left; }
span.orderNumLegend { background-color: #5a4a2a; width: 96px; font-weight: bold ;}
span.coverHits { background-color: #4a4a2a; padding-left: 3px; padding-right: 1px; text-align: right; list-style-type: none; display: inline-block; width: 5em; }
span.coverHitsLegend { background-color: #4a4a2a; width: 96px; font-weight: bold; margin: 0 auto;}
td.tableHead { text-align: center; color: #e0e0e0; background-color: #4a6ba8; font-family: sans-serif; font-size: 120%; font-weight: bold; }
td.coverFile { text-align: left; padding-left: 10px; padding-right: 20px; color: #6b9aff; font-family: monospace; background-color: #3a3a3a; }
td.coverBar { padding-left: 10px; padding-right: 10px; background-color: #3a3a3a; }
td.coverBarOutline { background-color: #4a4a4a; }
td.coverPer { text-align: left; padding-left: 10px; padding-right: 10px; font-weight: bold; background-color: #3a3a3a; color: #e0e0e0; }
td.coverPerLeftMed { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #5a5a00; font-weight: bold; color: #e0e0e0; }
td.coverPerLeftLo { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #5a2d2d; font-weight: bold; color: #e0e0e0; }
td.coverPerLeftHi { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #2d5a2d; font-weight: bold; color: #e0e0e0; }
td.coverNum { text-align: right; padding-left: 10px; padding-right: 10px; background-color: #3a3a3a; color: #e0e0e0; }
/* Override tablesorter hover styles for dark theme */
.tablesorter-blue tbody > tr:hover > td,
.tablesorter-blue tbody > tr:hover + tr.tablesorter-childRow > td,
.tablesorter-blue tbody > tr:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td,
.tablesorter-blue tbody > tr.even:hover > td,
.tablesorter-blue tbody > tr.even:hover + tr.tablesorter-childRow > td,
.tablesorter-blue tbody > tr.even:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td {
background: #4a4a4a;
}
.tablesorter-blue tbody > tr.odd:hover > td,
.tablesorter-blue tbody > tr.odd:hover + tr.tablesorter-childRow > td,
.tablesorter-blue tbody > tr.odd:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td {
background: #4a4a4a;
}

85
build/download_kcov.zig Normal file
View file

@ -0,0 +1,85 @@
const std = @import("std");
pub fn main(init: std.process.Init) !void {
// Build-time helper: short-lived process that downloads a single
// file. Arena lets us skip per-allocation `defer free(...)` and
// amortizes the allocation cost across the run via the arena's
// exponential block growth. Process exit reclaims everything.
const allocator = init.arena.allocator();
const io = init.io;
const args = try init.minimal.args.toSlice(allocator);
if (args.len != 3) return error.InvalidArgs;
const kcov_path = args[1];
const arch_name = args[2];
// Check to see if file exists. If it does, we have nothing more to do
const stat = std.Io.Dir.cwd().statFile(io, kcov_path, .{}) catch |err| blk: {
if (err == error.FileNotFound) break :blk null else return err;
};
// This might be better checking whether it's executable and >= 7MB, but
// for now, we'll do a simple exists check
if (stat != null) return;
var stdout_buffer: [1024]u8 = undefined;
var stdout_writer = std.Io.File.stdout().writer(io, &stdout_buffer);
const stdout = &stdout_writer.interface;
try stdout.writeAll("Determining latest kcov version\n");
try stdout.flush();
var client = std.http.Client{ .allocator = allocator, .io = io };
defer client.deinit();
// Get redirect to find latest version
const list_uri = try std.Uri.parse("https://git.lerch.org/lobo/-/packages/generic/kcov/");
var req = try client.request(.GET, list_uri, .{ .redirect_behavior = .unhandled });
defer req.deinit();
try req.sendBodiless();
var redirect_buf: [1024]u8 = undefined;
const response = try req.receiveHead(&redirect_buf);
if (response.head.status != .see_other) return error.UnexpectedResponse;
const location = response.head.location orelse return error.NoLocation;
const version_start = std.mem.lastIndexOfScalar(u8, location, '/') orelse return error.InvalidLocation;
const version = location[version_start + 1 ..];
try stdout.print(
"Downloading kcov version {s} for {s} to {s}...",
.{ version, arch_name, kcov_path },
);
try stdout.flush();
const binary_url = try std.fmt.allocPrint(
allocator,
"https://git.lerch.org/api/packages/lobo/generic/kcov/{s}/kcov-{s}",
.{ version, arch_name },
);
const cache_dir = std.fs.path.dirname(kcov_path) orelse return error.InvalidPath;
std.Io.Dir.cwd().createDir(io, cache_dir, std.Io.File.Permissions.default_dir) catch |e| switch (e) {
error.PathAlreadyExists => {},
else => return e,
};
const uri = try std.Uri.parse(binary_url);
const file = try std.Io.Dir.cwd().createFile(io, kcov_path, .{});
defer file.close(io);
try file.setPermissions(io, @enumFromInt(0o755));
var buffer: [8192]u8 = undefined;
var writer = file.writer(io, &buffer);
const result = try client.fetch(.{
.location = .{ .uri = uri },
.response_writer = &writer.interface,
});
if (result.status != .ok) return error.DownloadFailed;
try writer.interface.flush();
try stdout.writeAll("done\n");
try stdout.flush();
}

910
src/biff.zig Normal file
View file

@ -0,0 +1,910 @@
//! BIFF8 decoding: turns the `Workbook` stream of a legacy `.xls` file
//! (Excel 97 through 2003) into sheets of cell values.
//!
//! The stream is a flat sequence of records (`u16` type, `u16` length,
//! body). It opens with a "workbook globals" substream (BOF ... EOF)
//! holding the shared string table and one BOUNDSHEET8 record per
//! sheet; each BOUNDSHEET8 gives the stream offset of that sheet's own
//! BOF ... EOF substream, which holds its cell records.
//!
//! Decoded: LABELSST, LABEL, RSTRING (text), NUMBER, RK, MULRK
//! (numbers), BOOLERR (booleans and error codes), and FORMULA cached
//! results (with the STRING record that carries a text result).
//! Formatting, formulas themselves, merged cells, comments and charts
//! are ignored. Numbers are returned raw: a date-formatted cell is an
//! Excel serial number, because interpreting it needs the cell's
//! number format, which this reader does not decode.
//!
//! The shared string table (SST) is the one tricky part. A record body
//! is capped at 8224 bytes, so a long table spills into CONTINUE
//! records, and a string may be split across that boundary. When the
//! split falls inside a string's characters, the CONTINUE body begins
//! with a fresh "high byte" flag saying whether the rest of the
//! characters are 1-byte (Latin-1) or 2-byte (UTF-16), and it can
//! differ from the flag the string started with. Splits elsewhere
//! (string header, rich-text runs, phonetic data) carry no flag byte.
//!
//! Spec: [MS-XLS] Excel Binary File Format (.xls) Structure.
const std = @import("std");
const Allocator = std.mem.Allocator;
pub const Error = error{
/// Records are inconsistent: a length past the end of the stream,
/// a string index outside the shared string table, a malformed
/// record body.
CorruptFile,
/// The stream ends before a substream's EOF record.
Truncated,
/// Not BIFF8. Excel 95 and earlier (BIFF5 and below) are out of
/// scope.
UnsupportedBiffVersion,
/// The workbook is password-protected (a FILEPASS record).
Encrypted,
OutOfMemory,
};
/// One cell's value.
pub const Cell = union(enum) {
empty,
/// UTF-8.
text: []const u8,
/// Raw double. Date cells are Excel serial numbers (see module doc).
number: f64,
boolean: bool,
/// Excel error code: 0x00 #NULL!, 0x07 #DIV/0!, 0x0F #VALUE!,
/// 0x17 #REF!, 0x1D #NAME?, 0x24 #NUM!, 0x2A #N/A.
error_code: u8,
pub fn asText(self: Cell) ?[]const u8 {
return switch (self) {
.text => |t| t,
else => null,
};
}
pub fn asNumber(self: Cell) ?f64 {
return switch (self) {
.number => |n| n,
else => null,
};
}
};
/// A worksheet's cells, row-major. `rows[r]` is as wide as the
/// rightmost populated cell of row `r` (possibly empty), so a sparse
/// sheet costs memory proportional to its content. Use `cell` for
/// bounds-safe access.
///
/// The fields are public so callers can build a `Sheet` literal in
/// their own tests without producing a binary file.
pub const Sheet = struct {
name: []const u8,
rows: []const []const Cell,
/// The cell at zero-based (`row`, `col`), or `.empty` when out of
/// range.
pub fn cell(self: Sheet, row: usize, col: usize) Cell {
if (row >= self.rows.len) return .empty;
const r = self.rows[row];
if (col >= r.len) return .empty;
return r[col];
}
};
const rt = struct {
const bof: u16 = 0x0809;
const eof: u16 = 0x000A;
const filepass: u16 = 0x002F;
const boundsheet: u16 = 0x0085;
const sst: u16 = 0x00FC;
const @"continue": u16 = 0x003C;
const labelsst: u16 = 0x00FD;
const number: u16 = 0x0203;
const rk: u16 = 0x027E;
const mulrk: u16 = 0x00BD;
const label: u16 = 0x0204;
const rstring: u16 = 0x00D6;
const boolerr: u16 = 0x0205;
const formula: u16 = 0x0006;
const string: u16 = 0x0207;
// BOF record numbers of BIFF2, BIFF3 and BIFF4.
const bof_biff2: u16 = 0x0009;
const bof_biff3: u16 = 0x0209;
const bof_biff4: u16 = 0x0409;
};
const biff8_version: u16 = 0x0600;
const dt_globals: u16 = 0x0005;
const dt_worksheet: u16 = 0x0010;
/// BOUNDSHEET8 `dt` for a worksheet (or dialog sheet).
const sheet_type_worksheet: u8 = 0;
/// Decode every worksheet in a BIFF8 `Workbook` stream, in workbook
/// order. Charts, macro sheets and VB modules are skipped.
///
/// Everything returned is allocated from `arena` and never freed
/// individually. `scratch` holds temporaries, all freed before return.
pub fn parseSheets(arena: Allocator, scratch: Allocator, stream: []const u8) Error![]const Sheet {
var recs: Records = .{ .stream = stream };
const first = (try recs.next()) orelse return error.Truncated;
try expectBof(first, dt_globals);
var bounds: std.ArrayList(BoundSheet) = .empty;
defer bounds.deinit(scratch);
var sst: []const []const u8 = &.{};
while (true) {
const rec = (try recs.next()) orelse return error.Truncated;
switch (rec.kind) {
rt.eof => break,
rt.filepass => return error.Encrypted,
rt.boundsheet => try bounds.append(scratch, try parseBoundSheet(arena, scratch, rec.data)),
rt.sst => sst = try parseSst(arena, scratch, &recs, rec.data),
else => {},
}
}
var sheets: std.ArrayList(Sheet) = .empty;
for (bounds.items) |b| {
if (b.kind != sheet_type_worksheet) continue;
try sheets.append(arena, try parseSheet(arena, scratch, stream, b, sst));
}
return sheets.items;
}
const BoundSheet = struct {
name: []const u8,
offset: u32,
kind: u8,
};
fn parseBoundSheet(arena: Allocator, scratch: Allocator, data: []const u8) Error!BoundSheet {
if (data.len < 8) return error.CorruptFile;
var r: Chunks = .{ .chunks = &.{data[6..]} };
// ShortXLUnicodeString: one-byte length, then flags and characters.
const cch = try r.byte();
return .{
.offset = readInt(u32, data, 0),
.kind = data[5],
.name = try r.characters(arena, scratch, cch, (try r.byte()) & 1 != 0),
};
}
/// SST body plus every CONTINUE record that follows it.
fn parseSst(arena: Allocator, scratch: Allocator, recs: *Records, first: []const u8) Error![]const []const u8 {
var chunks: std.ArrayList([]const u8) = .empty;
defer chunks.deinit(scratch);
try chunks.append(scratch, first);
while (recs.peekKind() == rt.@"continue") try chunks.append(scratch, (try recs.next()).?.data);
var r: Chunks = .{ .chunks = chunks.items };
_ = try r.int(u32); // total references; irrelevant to a reader
const unique = try r.int(u32);
// Every string costs at least three bytes (length + flags), which
// bounds the allocation a corrupt count can request.
var total_len: usize = 0;
for (chunks.items) |c| total_len += c.len;
if (unique > total_len / 3) return error.CorruptFile;
const strings = try arena.alloc([]const u8, unique);
for (strings) |*s| {
// XLUnicodeRichExtendedString.
const cch = try r.int(u16);
const flags = try r.byte();
const runs: usize = if (flags & 0x08 != 0) try r.int(u16) else 0;
const ext: usize = if (flags & 0x04 != 0) try r.int(u32) else 0;
s.* = try r.characters(arena, scratch, cch, flags & 1 != 0);
try r.skip(4 * runs);
try r.skip(ext);
}
return strings;
}
const Placed = struct {
row: u16,
col: u16,
cell: Cell,
};
fn parseSheet(arena: Allocator, scratch: Allocator, stream: []const u8, b: BoundSheet, sst: []const []const u8) Error!Sheet {
if (b.offset >= stream.len) return error.CorruptFile;
var recs: Records = .{ .stream = stream, .pos = b.offset };
const first = (try recs.next()) orelse return error.CorruptFile;
try expectBof(first, dt_worksheet);
var cells: std.ArrayList(Placed) = .empty;
defer cells.deinit(scratch);
// Embedded charts are complete BOF ... EOF substreams nested in the
// sheet's; their records are not this sheet's cells.
var depth: usize = 0;
// A FORMULA whose cached result is text is followed by a STRING
// record carrying that text.
var pending_string: ?Placed = null;
while (true) {
const rec = (try recs.next()) orelse return error.Truncated;
const d = rec.data;
switch (rec.kind) {
rt.bof => {
depth += 1;
continue;
},
rt.eof => {
if (depth == 0) break;
depth -= 1;
continue;
},
else => if (depth > 0) continue,
}
switch (rec.kind) {
rt.labelsst => {
if (d.len < 10) return error.CorruptFile;
const index = readInt(u32, d, 6);
if (index >= sst.len) return error.CorruptFile;
try put(scratch, &cells, d, .{ .text = sst[index] });
},
rt.number => {
if (d.len < 14) return error.CorruptFile;
try put(scratch, &cells, d, .{ .number = @bitCast(readInt(u64, d, 6)) });
},
rt.rk => {
if (d.len < 10) return error.CorruptFile;
try put(scratch, &cells, d, .{ .number = decodeRk(readInt(u32, d, 6)) });
},
rt.mulrk => {
// row, first col, N x (xf index, RK), last col.
if (d.len < 12 or (d.len - 6) % 6 != 0) return error.CorruptFile;
const row = readInt(u16, d, 0);
const first_col = readInt(u16, d, 2);
const n = (d.len - 6) / 6;
if (@as(usize, readInt(u16, d, d.len - 2)) + 1 != @as(usize, first_col) + n) return error.CorruptFile;
for (0..n) |i| {
try cells.append(scratch, .{
.row = row,
.col = first_col + @as(u16, @intCast(i)),
.cell = .{ .number = decodeRk(readInt(u32, d, 4 + 6 * i + 2)) },
});
}
},
// Both carry an XLUnicodeString after the cell header;
// RSTRING's trailing formatting runs are ignored.
rt.label, rt.rstring => {
if (d.len < 9) return error.CorruptFile;
var r: Chunks = .{ .chunks = &.{d[6..]} };
try put(scratch, &cells, d, .{ .text = try r.unicodeString(arena, scratch) });
},
rt.boolerr => {
if (d.len < 8) return error.CorruptFile;
const cell: Cell = if (d[7] != 0) .{ .error_code = d[6] } else .{ .boolean = d[6] != 0 };
try put(scratch, &cells, d, cell);
},
rt.formula => {
if (d.len < 20) return error.CorruptFile;
pending_string = null;
const v = d[6..14];
// 0xFFFF in the top two bytes marks a non-numeric result.
if (readInt(u16, v, 6) != 0xFFFF) {
try put(scratch, &cells, d, .{ .number = @bitCast(readInt(u64, v, 0)) });
} else switch (v[0]) {
0 => pending_string = .{ .row = readInt(u16, d, 0), .col = readInt(u16, d, 2), .cell = .empty },
1 => try put(scratch, &cells, d, .{ .boolean = v[2] != 0 }),
2 => try put(scratch, &cells, d, .{ .error_code = v[2] }),
3 => try put(scratch, &cells, d, .{ .text = "" }),
else => return error.CorruptFile,
}
},
rt.string => if (pending_string) |p| {
var chunks: std.ArrayList([]const u8) = .empty;
defer chunks.deinit(scratch);
try chunks.append(scratch, d);
while (recs.peekKind() == rt.@"continue") try chunks.append(scratch, (try recs.next()).?.data);
var r: Chunks = .{ .chunks = chunks.items };
try cells.append(scratch, .{ .row = p.row, .col = p.col, .cell = .{ .text = try r.unicodeString(arena, scratch) } });
pending_string = null;
},
else => {},
}
}
return .{ .name = b.name, .rows = try buildRows(arena, scratch, cells.items) };
}
/// Record a cell whose row and column are the first four bytes of `d`.
fn put(scratch: Allocator, cells: *std.ArrayList(Placed), d: []const u8, cell: Cell) Error!void {
try cells.append(scratch, .{ .row = readInt(u16, d, 0), .col = readInt(u16, d, 2), .cell = cell });
}
/// Lay placed cells out as rows. A later record for the same cell
/// wins, matching how Excel would overwrite it.
fn buildRows(arena: Allocator, scratch: Allocator, cells: []const Placed) Error![]const []const Cell {
var row_count: usize = 0;
for (cells) |c| row_count = @max(row_count, @as(usize, c.row) + 1);
const widths = try scratch.alloc(usize, row_count);
defer scratch.free(widths);
@memset(widths, 0);
for (cells) |c| widths[c.row] = @max(widths[c.row], @as(usize, c.col) + 1);
const rows = try arena.alloc([]Cell, row_count);
for (rows, widths) |*r, w| {
r.* = try arena.alloc(Cell, w);
@memset(r.*, .empty);
}
for (cells) |c| rows[c.row][c.col] = c.cell;
return rows;
}
/// RK: a compressed number. Bit 1 selects a 30-bit signed integer over
/// the high 30 bits of an IEEE double; bit 0 divides by 100.
fn decodeRk(raw: u32) f64 {
const value: f64 = if (raw & 2 != 0)
@floatFromInt(@as(i32, @bitCast(raw)) >> 2)
else
@bitCast(@as(u64, raw & 0xFFFFFFFC) << 32);
return if (raw & 1 != 0) value / 100 else value;
}
fn expectBof(rec: Record, dt: u16) Error!void {
switch (rec.kind) {
rt.bof => {},
rt.bof_biff2, rt.bof_biff3, rt.bof_biff4 => return error.UnsupportedBiffVersion,
else => return error.CorruptFile,
}
if (rec.data.len < 4) return error.CorruptFile;
if (readInt(u16, rec.data, 0) != biff8_version) return error.UnsupportedBiffVersion;
if (readInt(u16, rec.data, 2) != dt) return error.CorruptFile;
}
const Record = struct {
kind: u16,
data: []const u8,
};
const Records = struct {
stream: []const u8,
pos: usize = 0,
fn next(self: *Records) Error!?Record {
if (self.pos == self.stream.len) return null;
if (self.stream.len - self.pos < 4) return error.CorruptFile;
const kind = readInt(u16, self.stream, self.pos);
const len = readInt(u16, self.stream, self.pos + 2);
const start = self.pos + 4;
if (len > self.stream.len - start) return error.CorruptFile;
self.pos = start + len;
return .{ .kind = kind, .data = self.stream[start..][0..len] };
}
fn peekKind(self: Records) ?u16 {
if (self.stream.len - self.pos < 4) return null;
return readInt(u16, self.stream, self.pos);
}
};
/// Reads across a record body and the CONTINUE bodies after it. Plain
/// reads (`byte`, `int`, `skip`) step across a boundary transparently;
/// `characters` consumes the high-byte flag a boundary inside character
/// data introduces.
const Chunks = struct {
chunks: []const []const u8,
index: usize = 0,
pos: usize = 0,
fn current(self: Chunks) []const u8 {
return self.chunks[self.index];
}
fn nextChunk(self: *Chunks) Error!void {
if (self.index + 1 >= self.chunks.len) return error.CorruptFile;
self.index += 1;
self.pos = 0;
}
fn byte(self: *Chunks) Error!u8 {
while (self.pos == self.current().len) try self.nextChunk();
const b = self.current()[self.pos];
self.pos += 1;
return b;
}
fn int(self: *Chunks, comptime T: type) Error!T {
var v: T = 0;
for (0..@sizeOf(T)) |i| v |= @as(T, try self.byte()) << @intCast(8 * i);
return v;
}
fn skip(self: *Chunks, n: usize) Error!void {
var left = n;
while (left > 0) {
if (self.pos == self.current().len) try self.nextChunk();
const k = @min(left, self.current().len - self.pos);
self.pos += k;
left -= k;
}
}
/// XLUnicodeString: `u16` length, flags, characters.
fn unicodeString(self: *Chunks, arena: Allocator, scratch: Allocator) Error![]const u8 {
const cch = try self.int(u16);
const flags = try self.byte();
return self.characters(arena, scratch, cch, flags & 1 != 0);
}
/// `count` characters, 1-byte (Latin-1) or 2-byte (UTF-16LE) per
/// `high`, re-reading the flag whenever the characters cross into
/// the next chunk. Returns UTF-8 allocated from `arena`.
fn characters(self: *Chunks, arena: Allocator, scratch: Allocator, count: usize, high_start: bool) Error![]const u8 {
var units: std.ArrayList(u16) = .empty;
defer units.deinit(scratch);
try units.ensureTotalCapacity(scratch, count);
var high = high_start;
var left = count;
while (left > 0) {
if (self.pos == self.current().len) {
try self.nextChunk();
if (self.current().len == 0) return error.CorruptFile;
high = self.current()[0] & 1 != 0;
self.pos = 1;
}
const avail = self.current()[self.pos..];
if (high) {
const k = @min(left, avail.len / 2);
// A lone trailing byte cannot hold a UTF-16 unit.
if (k == 0) return error.CorruptFile;
for (0..k) |i| units.appendAssumeCapacity(readInt(u16, avail, 2 * i));
self.pos += 2 * k;
left -= k;
} else {
const k = @min(left, avail.len);
for (avail[0..k]) |c| units.appendAssumeCapacity(c);
self.pos += k;
left -= k;
}
}
return utf8FromUtf16(arena, units.items);
}
};
/// UTF-16 to UTF-8. Unpaired surrogates become U+FFFD rather than an
/// error: a stray half in a cell should not make the workbook
/// unreadable.
fn utf8FromUtf16(arena: Allocator, units: []const u16) Error![]const u8 {
var len: usize = 0;
var i: usize = 0;
while (i < units.len) len += utf8Len(nextCodepoint(units, &i));
const out = try arena.alloc(u8, len);
var o: usize = 0;
i = 0;
while (i < units.len) o += encodeUtf8(nextCodepoint(units, &i), out[o..]);
return out;
}
fn nextCodepoint(units: []const u16, i: *usize) u21 {
const u = units[i.*];
i.* += 1;
if (u >= 0xD800 and u <= 0xDBFF and i.* < units.len and units[i.*] >= 0xDC00 and units[i.*] <= 0xDFFF) {
const lo = units[i.*];
i.* += 1;
return 0x10000 + ((@as(u21, u) - 0xD800) << 10) + (lo - 0xDC00);
}
if (u >= 0xD800 and u <= 0xDFFF) return 0xFFFD;
return u;
}
fn utf8Len(c: u21) usize {
if (c < 0x80) return 1;
if (c < 0x800) return 2;
if (c < 0x10000) return 3;
return 4;
}
fn encodeUtf8(c: u21, out: []u8) usize {
switch (utf8Len(c)) {
1 => out[0] = @intCast(c),
2 => {
out[0] = @intCast(0xC0 | (c >> 6));
out[1] = @intCast(0x80 | (c & 0x3F));
},
3 => {
out[0] = @intCast(0xE0 | (c >> 12));
out[1] = @intCast(0x80 | ((c >> 6) & 0x3F));
out[2] = @intCast(0x80 | (c & 0x3F));
},
else => {
out[0] = @intCast(0xF0 | (c >> 18));
out[1] = @intCast(0x80 | ((c >> 12) & 0x3F));
out[2] = @intCast(0x80 | ((c >> 6) & 0x3F));
out[3] = @intCast(0x80 | (c & 0x3F));
},
}
return utf8Len(c);
}
fn readInt(comptime T: type, bytes: []const u8, off: usize) T {
return std.mem.readInt(T, bytes[off..][0..@sizeOf(T)], .little);
}
// ---- Tests ----
const testing = std.testing;
const tw = @import("test_writer.zig");
/// Parse a workbook stream with a leak-checked scratch allocator and an
/// arena for the results.
const Parsed = struct {
arena: std.heap.ArenaAllocator,
sheets: []const Sheet,
fn init(stream: []const u8) Error!Parsed {
var arena = std.heap.ArenaAllocator.init(testing.allocator);
errdefer arena.deinit();
const sheets = try parseSheets(arena.allocator(), testing.allocator, stream);
return .{ .arena = arena, .sheets = sheets };
}
fn deinit(self: *Parsed) void {
self.arena.deinit();
}
};
fn expectParseError(expected: Error, spec: tw.WorkbookSpec) !void {
const stream = try tw.workbook(testing.allocator, spec);
defer testing.allocator.free(stream);
var arena = std.heap.ArenaAllocator.init(testing.allocator);
defer arena.deinit();
try testing.expectError(expected, parseSheets(arena.allocator(), testing.allocator, stream));
}
test "decodes the common cell records" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const sst = try tw.sst(a, &.{ .{ .text = "Description" }, .{ .text = "Sample Brokerage" } }, 8224);
const stream = try tw.workbook(a, .{
.globals = sst,
.sheets = &.{.{
.name = "Positions",
.records = &.{
try tw.labelSst(a, 0, 0, 0),
try tw.labelSst(a, 1, 0, 1),
try tw.number(a, 1, 1, 1234.5678),
try tw.rk(a, 1, 2, (25 << 2) | 2), // integer 25
try tw.boolErr(a, 2, 0, 1, false),
try tw.boolErr(a, 2, 1, 0x07, true),
try tw.label(a, tw.rt.label, 3, 0, "inline label"),
try tw.label(a, tw.rt.rstring, 3, 1, "rich label"),
try tw.blank(a, 4, 3),
},
}},
});
var p = try Parsed.init(stream);
defer p.deinit();
try testing.expectEqual(@as(usize, 1), p.sheets.len);
const s = p.sheets[0];
try testing.expectEqualStrings("Positions", s.name);
try testing.expectEqualStrings("Description", s.cell(0, 0).asText().?);
try testing.expectEqualStrings("Sample Brokerage", s.cell(1, 0).asText().?);
try testing.expectEqual(@as(f64, 1234.5678), s.cell(1, 1).asNumber().?);
try testing.expectEqual(@as(f64, 25), s.cell(1, 2).asNumber().?);
try testing.expectEqual(Cell{ .boolean = true }, s.cell(2, 0));
try testing.expectEqual(Cell{ .error_code = 0x07 }, s.cell(2, 1));
try testing.expectEqualStrings("inline label", s.cell(3, 0).asText().?);
try testing.expectEqualStrings("rich label", s.cell(3, 1).asText().?);
// BLANK carries formatting only, so it adds no cell (or row).
try testing.expectEqual(@as(usize, 4), s.rows.len);
try testing.expectEqual(Cell.empty, s.cell(4, 3));
// Out of range in either direction is empty, not a crash.
try testing.expectEqual(Cell.empty, s.cell(0, 200));
try testing.expectEqual(Cell.empty, s.cell(9999, 0));
try testing.expect(s.cell(1, 1).asText() == null);
try testing.expect(s.cell(0, 0).asNumber() == null);
}
test "RK encodings" {
// Integer, integer / 100, float (high 30 bits of a double), float / 100.
try testing.expectEqual(@as(f64, 25), decodeRk((25 << 2) | 2));
try testing.expectEqual(@as(f64, -25), decodeRk(@as(u32, @bitCast(@as(i32, -25) << 2)) | 2));
try testing.expectEqual(@as(f64, 12.34), decodeRk((1234 << 2) | 3));
const one_and_half: u64 = @bitCast(@as(f64, 1.5));
const hi: u32 = @intCast(one_and_half >> 32);
try testing.expectEqual(@as(f64, 1.5), decodeRk(hi));
try testing.expectEqual(@as(f64, 0.015), decodeRk(hi | 1));
}
test "MULRK fans out across columns" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{
try tw.mulrk(a, 5, 2, &.{ (1 << 2) | 2, (2 << 2) | 2, (300 << 2) | 3 }),
} }} });
var p = try Parsed.init(stream);
defer p.deinit();
const s = p.sheets[0];
try testing.expectEqual(@as(f64, 1), s.cell(5, 2).asNumber().?);
try testing.expectEqual(@as(f64, 2), s.cell(5, 3).asNumber().?);
try testing.expectEqual(@as(f64, 3), s.cell(5, 4).asNumber().?);
try testing.expectEqual(Cell.empty, s.cell(5, 1));
}
test "MULRK whose last column disagrees with its length is corrupt" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const rec = try tw.mulrk(a, 0, 0, &.{ 2, 6 });
const bad = try a.dupe(u8, rec.data);
std.mem.writeInt(u16, bad[bad.len - 2 ..][0..2], 7, .little);
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{.{ .kind = tw.rt.mulrk, .data = bad }} }} });
}
test "formula cached results" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const stream = try tw.workbook(a, .{
.sheets = &.{.{
.name = "S",
.records = &.{
try tw.formula(a, 0, 0, tw.formulaNumber(42.5)),
try tw.formula(a, 0, 1, tw.formulaSpecial(0, 0)),
// Records such as SHRFMLA may sit between FORMULA and STRING.
.{ .kind = 0x04BC, .data = "\x00\x00" },
try tw.string(a, "computed text"),
try tw.formula(a, 0, 2, tw.formulaSpecial(1, 1)),
try tw.formula(a, 0, 3, tw.formulaSpecial(2, 0x2A)),
try tw.formula(a, 0, 4, tw.formulaSpecial(3, 0)),
// A STRING with no text formula pending is ignored.
try tw.string(a, "orphan"),
// A text formula whose STRING never arrives leaves the cell empty.
try tw.formula(a, 0, 5, tw.formulaSpecial(0, 0)),
try tw.number(a, 1, 0, 1),
},
}},
});
var p = try Parsed.init(stream);
defer p.deinit();
const s = p.sheets[0];
try testing.expectEqual(@as(f64, 42.5), s.cell(0, 0).asNumber().?);
try testing.expectEqualStrings("computed text", s.cell(0, 1).asText().?);
try testing.expectEqual(Cell{ .boolean = true }, s.cell(0, 2));
try testing.expectEqual(Cell{ .error_code = 0x2A }, s.cell(0, 3));
try testing.expectEqualStrings("", s.cell(0, 4).asText().?);
try testing.expectEqual(Cell.empty, s.cell(0, 5));
}
test "formula string result continued across a CONTINUE record" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
// STRING body: cch=6, compressed, "abc" | CONTINUE: flag 1 (UTF-16), "def".
const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{
try tw.formula(a, 0, 0, tw.formulaSpecial(0, 0)),
.{ .kind = tw.rt.string, .data = "\x06\x00\x00abc" },
.{ .kind = tw.rt.@"continue", .data = "\x01d\x00e\x00f\x00" },
} }} });
var p = try Parsed.init(stream);
defer p.deinit();
try testing.expectEqualStrings("abcdef", p.sheets[0].cell(0, 0).asText().?);
}
test "unknown formula result type is corrupt" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{
try tw.formula(a, 0, 0, tw.formulaSpecial(9, 0)),
} }} });
}
test "shared strings split across CONTINUE records at every offset" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const strings = [_]tw.SstString{
.{ .text = "Bank Deposit Sweep" },
.{ .text = "caf\u{e9} au lait" }, // Latin-1 but not ASCII: still compressed
.{ .text = "\u{201c}buy\u{201d} or \u{201c}sell\u{201d}" }, // needs UTF-16
.{ .text = "rich text", .runs = 3 },
.{ .text = "phonetic", .ext = "\x01\x02\x03\x04\x05\x06\x07" },
.{ .text = "" },
.{ .text = "emoji \u{1F600} end" }, // surrogate pair
.{ .text = "\u{3042} then plain ascii after the switch" }, // UTF-16 start, compressed tail
};
// Every chunk size from tiny (splits everywhere, including inside
// headers) to large (no split) must decode identically.
var max_chunk: usize = 3;
while (max_chunk <= 200) : (max_chunk += 1) {
const sst = try tw.sst(a, &strings, max_chunk);
var records: std.ArrayList(tw.Record) = .empty;
for (0..strings.len) |i| try records.append(a, try tw.labelSst(a, @intCast(i), 0, @intCast(i)));
const stream = try tw.workbook(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} });
var p = try Parsed.init(stream);
defer p.deinit();
for (strings, 0..) |s, i| {
testing.expectEqualStrings(s.text, p.sheets[0].cell(i, 0).asText().?) catch |err| {
std.debug.print("max_chunk={d} string={d}\n", .{ max_chunk, i });
return err;
};
}
}
}
test "unpaired surrogates decode as U+FFFD" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const sst = try tw.sst(a, &.{
.{ .units = &.{ 'a', 0xD800, 'b' } },
.{ .units = &.{ 'a', 0xDC00 } },
.{ .units = &.{0xD83D} },
}, 8224);
const stream = try tw.workbook(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = &.{
try tw.labelSst(a, 0, 0, 0),
try tw.labelSst(a, 1, 0, 1),
try tw.labelSst(a, 2, 0, 2),
} }} });
var p = try Parsed.init(stream);
defer p.deinit();
try testing.expectEqualStrings("a\u{FFFD}b", p.sheets[0].cell(0, 0).asText().?);
try testing.expectEqualStrings("a\u{FFFD}", p.sheets[0].cell(1, 0).asText().?);
try testing.expectEqualStrings("\u{FFFD}", p.sheets[0].cell(2, 0).asText().?);
}
test "UTF-8 encoding covers every sequence length" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const units = [_]u16{ 'A', 0x00E9, 0x20AC, 0xD83D, 0xDE00 };
try testing.expectEqualStrings("A\u{e9}\u{20ac}\u{1F600}", try utf8FromUtf16(fx.allocator(), &units));
}
test "multiple sheets keep workbook order; non-worksheets are skipped" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const stream = try tw.workbook(a, .{ .sheets = &.{
.{ .name = "First", .records = &.{try tw.number(a, 0, 0, 1)} },
.{ .name = "Chart1", .boundsheet_type = 2, .bof_type = 0x0020 },
.{ .name = "Second", .records = &.{try tw.number(a, 0, 0, 2)} },
} });
var p = try Parsed.init(stream);
defer p.deinit();
try testing.expectEqual(@as(usize, 2), p.sheets.len);
try testing.expectEqualStrings("First", p.sheets[0].name);
try testing.expectEqualStrings("Second", p.sheets[1].name);
try testing.expectEqual(@as(f64, 2), p.sheets[1].cell(0, 0).asNumber().?);
}
test "embedded chart substreams inside a sheet are not its cells" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const chart_bof = tw.bof(0x0600, 0x0020);
const stream = try tw.workbook(a, .{
.sheets = &.{.{
.name = "S",
.records = &.{
try tw.number(a, 0, 0, 1),
.{ .kind = tw.rt.bof, .data = &chart_bof },
try tw.number(a, 0, 1, 99), // belongs to the chart
.{ .kind = tw.rt.eof, .data = "" },
try tw.number(a, 0, 2, 3),
},
}},
});
var p = try Parsed.init(stream);
defer p.deinit();
const s = p.sheets[0];
try testing.expectEqual(@as(f64, 1), s.cell(0, 0).asNumber().?);
try testing.expectEqual(Cell.empty, s.cell(0, 1));
try testing.expectEqual(@as(f64, 3), s.cell(0, 2).asNumber().?);
}
test "a repeated cell keeps the last value; sparse rows stay empty" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{
try tw.number(a, 3, 1, 1),
try tw.number(a, 3, 1, 2),
try tw.number(a, 0, 4, 7),
} }} });
var p = try Parsed.init(stream);
defer p.deinit();
const s = p.sheets[0];
try testing.expectEqual(@as(usize, 4), s.rows.len);
try testing.expectEqual(@as(f64, 2), s.cell(3, 1).asNumber().?);
try testing.expectEqual(@as(usize, 0), s.rows[1].len);
try testing.expectEqual(@as(usize, 5), s.rows[0].len);
}
test "a sheet with no cells has no rows" {
const stream = try tw.workbook(testing.allocator, .{ .sheets = &.{.{ .name = "Empty" }} });
defer testing.allocator.free(stream);
var p = try Parsed.init(stream);
defer p.deinit();
try testing.expectEqual(@as(usize, 0), p.sheets[0].rows.len);
}
test "older BIFF versions are unsupported" {
try expectParseError(error.UnsupportedBiffVersion, .{ .version = 0x0500 });
// BIFF2-4 use different BOF record numbers entirely.
const biff4_bof = "\x09\x04\x06\x00\x00\x00\x10\x00\x00\x00";
var arena = std.heap.ArenaAllocator.init(testing.allocator);
defer arena.deinit();
try testing.expectError(error.UnsupportedBiffVersion, parseSheets(arena.allocator(), testing.allocator, biff4_bof));
}
test "password-protected workbooks report Encrypted" {
try expectParseError(error.Encrypted, .{ .globals = &.{.{ .kind = tw.rt.filepass, .data = "\x01\x00" }} });
}
test "structural corruption is reported" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
// LABELSST pointing past the shared string table.
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{try tw.labelSst(a, 0, 0, 5)} }} });
// BOUNDSHEET offset past the end of the stream.
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S" }}, .offset_override = 0xFFFFFF });
// BOUNDSHEET offset that lands on something other than a BOF.
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S" }}, .offset_override = 0 });
// Short record bodies.
const short = [_]u16{ tw.rt.labelsst, tw.rt.number, tw.rt.rk, tw.rt.mulrk, tw.rt.label, tw.rt.boolerr, tw.rt.formula };
for (short) |kind| {
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{.{ .kind = kind, .data = "\x00\x00" }} }} });
}
// An SST claiming more strings than its bytes could hold.
try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.sst, .data = "\x00\x00\x00\x00\xFF\xFF\x00\x00" }} });
// An SST whose last string runs off the end of its records.
try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x05\x00\x00ab" }} });
// A UTF-16 string that leaves one odd byte at a record boundary.
try expectParseError(error.CorruptFile, .{ .globals = &.{
.{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x02\x00\x01a" },
.{ .kind = tw.rt.@"continue", .data = "\x01b\x00" },
} });
// A CONTINUE with no room for the high-byte flag.
try expectParseError(error.CorruptFile, .{ .globals = &.{
.{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x02\x00\x00a" },
.{ .kind = tw.rt.@"continue", .data = "" },
} });
// A BOUNDSHEET too short to hold its header.
try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.boundsheet, .data = "\x00\x00" }} });
}
test "a substream without its EOF is truncated" {
try expectParseError(error.Truncated, .{ .sheets = &.{.{ .name = "S" }}, .omit_last_eof = true });
var arena = std.heap.ArenaAllocator.init(testing.allocator);
defer arena.deinit();
// Globals that never reach EOF, and an empty stream.
const globals_bof = comptime tw.bof(0x0600, 0x0005);
const no_eof = "\x09\x08\x10\x00" ++ globals_bof;
try testing.expectError(error.Truncated, parseSheets(arena.allocator(), testing.allocator, no_eof));
try testing.expectError(error.Truncated, parseSheets(arena.allocator(), testing.allocator, ""));
}
test "record framing errors" {
var arena = std.heap.ArenaAllocator.init(testing.allocator);
defer arena.deinit();
// A record header cut short, and a length past the end.
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08"));
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\xFF\x00"));
// First record is not a BOF at all.
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x0A\x00\x00\x00"));
// A BOF too short to carry a version.
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\x02\x00\x00\x06"));
// A BOF whose substream type is wrong for its position.
const ws_bof = comptime tw.bof(0x0600, 0x0010);
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\x10\x00" ++ ws_bof));
}

671
src/cfb.zig Normal file
View file

@ -0,0 +1,671 @@
//! Read-only reader for the Compound File Binary format (MS-CFB), also
//! known as "OLE2 structured storage". It is a FAT-style filesystem
//! inside one file, and it is the container a legacy `.xls` workbook
//! lives in: the spreadsheet itself is the `Workbook` stream inside it.
//!
//! Only what reading a stream needs is implemented:
//!
//! - header validation (signature, byte order, sector sizes for
//! major versions 3 and 4)
//! - the DIFAT -> FAT sector-allocation table, including DIFAT
//! sectors beyond the 109 entries the header holds
//! - the directory, scanned linearly (the red-black tree is ignored;
//! a name lookup over a few dozen entries does not need it)
//! - regular streams (FAT chains) and small streams (mini FAT chains
//! inside the root entry's mini stream)
//!
//! Every chain walk is bounded by the size of the table it walks, so a
//! cyclic chain is reported as `CorruptFile` instead of looping, and a
//! declared size larger than the file is rejected before allocating.
//!
//! Spec: [MS-CFB] Compound File Binary File Format.
const std = @import("std");
pub const Error = error{
/// The bytes do not start with the compound-file signature.
NotCompoundFile,
/// The structure is internally inconsistent: bad header fields,
/// a chain that loops or points outside its table, an entry larger
/// than the file can hold.
CorruptFile,
/// The file ends before a sector it references. Usually an
/// interrupted download.
Truncated,
OutOfMemory,
};
/// First eight bytes of every compound file.
pub const signature = [8]u8{ 0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1 };
/// Special sector numbers (MS-CFB 2.1). Any value above `max_reg_sect`
/// is not a real sector.
const max_reg_sect: u32 = 0xFFFFFFFA;
const end_of_chain: u32 = 0xFFFFFFFE;
const free_sect: u32 = 0xFFFFFFFF;
const header_size = 512;
const dir_entry_size = 128;
const mini_sector_size = 64;
/// DIFAT entries stored in the header itself.
const header_difat_count = 109;
/// True when `bytes` starts with the compound-file signature. Cheap
/// enough for content sniffing; does not validate anything else.
pub fn isCompoundFile(bytes: []const u8) bool {
return bytes.len >= signature.len and std.mem.eql(u8, bytes[0..signature.len], &signature);
}
pub const EntryType = enum(u8) {
unknown = 0,
storage = 1,
stream = 2,
root = 5,
_,
};
/// One directory entry. Only the fields a reader needs.
pub const Entry = struct {
/// UTF-16 code units of the name, without the terminating NUL.
name: [31]u16,
name_len: u8,
kind: EntryType,
start_sector: u32,
size: u64,
/// Case-insensitive comparison against an ASCII name. CFB names
/// compare case-insensitively (MS-CFB 2.6.4), and every stream name
/// a reader looks up by literal is ASCII.
pub fn nameEql(self: Entry, ascii: []const u8) bool {
if (ascii.len != self.name_len) return false;
for (self.name[0..self.name_len], ascii) |unit, c| {
if (unit > 0x7F) return false;
if (std.ascii.toLower(@intCast(unit)) != std.ascii.toLower(c)) return false;
}
return true;
}
};
pub const File = struct {
allocator: std.mem.Allocator,
bytes: []const u8,
sector_shift: u4,
mini_cutoff: u32,
fat: []u32,
mini_fat: []u32,
entries: []Entry,
/// Contents of the root entry's stream, which holds every small
/// stream. Empty when the file has no small streams.
mini_stream: []u8,
/// Parse the header, allocation tables and directory. `bytes` is
/// borrowed and must outlive the `File`.
pub fn init(allocator: std.mem.Allocator, bytes: []const u8) Error!File {
if (!isCompoundFile(bytes)) return error.NotCompoundFile;
if (bytes.len < header_size) return error.Truncated;
if (readU16(bytes, 28) != 0xFFFE) return error.CorruptFile; // byte order mark
const major = readU16(bytes, 26);
const sector_shift: u4 = switch (major) {
3 => 9,
4 => 12,
else => return error.CorruptFile,
};
if (readU16(bytes, 30) != sector_shift) return error.CorruptFile;
if (readU16(bytes, 32) != 6) return error.CorruptFile; // 64-byte mini sectors
const sector_size = @as(usize, 1) << sector_shift;
// Version 4 pads the header out to a full 4096-byte sector.
if (bytes.len < sector_size) return error.Truncated;
const num_fat = readU32(bytes, 44);
const first_dir = readU32(bytes, 48);
const mini_cutoff = readU32(bytes, 56);
const first_mini_fat = readU32(bytes, 60);
const first_difat = readU32(bytes, 68);
// Upper bound on sectors this file can address. Every table
// size below is checked against it before allocating, so a
// hostile header cannot request a huge allocation.
const total_sectors = (bytes.len - sector_size + sector_size - 1) >> sector_shift;
if (num_fat > total_sectors) return error.CorruptFile;
var self: File = .{
.allocator = allocator,
.bytes = bytes,
.sector_shift = sector_shift,
.mini_cutoff = mini_cutoff,
.fat = &.{},
.mini_fat = &.{},
.entries = &.{},
.mini_stream = &.{},
};
errdefer self.deinit();
self.fat = try self.readFat(num_fat, first_difat, total_sectors);
self.entries = try self.readDirectory(first_dir);
if (self.entries.len == 0 or self.entries[0].kind != .root) return error.CorruptFile;
self.mini_fat = try self.readMiniFat(first_mini_fat);
const root = self.entries[0];
if (root.size > 0) {
self.mini_stream = try self.readRegular(allocator, root.start_sector, root.size);
}
return self;
}
pub fn deinit(self: *File) void {
self.allocator.free(self.fat);
self.allocator.free(self.mini_fat);
self.allocator.free(self.entries);
self.allocator.free(self.mini_stream);
}
/// First stream entry whose name matches `ascii` case-insensitively.
pub fn find(self: File, ascii: []const u8) ?Entry {
for (self.entries) |e| {
if (e.kind == .stream and e.nameEql(ascii)) return e;
}
return null;
}
/// Read a stream's full contents. Caller owns the returned bytes.
pub fn readStream(self: File, allocator: std.mem.Allocator, entry: Entry) Error![]u8 {
if (entry.size < self.mini_cutoff) return self.readMini(allocator, entry.start_sector, entry.size);
return self.readRegular(allocator, entry.start_sector, entry.size);
}
fn sectorSize(self: File) usize {
return @as(usize, 1) << self.sector_shift;
}
/// Bytes of sector `index`. The final sector of a file may be short
/// (some writers do not pad it), so the slice can be shorter than a
/// sector; callers that need a whole sector use `fullSector`.
fn sector(self: File, index: u32) Error![]const u8 {
if (index > max_reg_sect) return error.CorruptFile;
const start = (@as(usize, index) + 1) << self.sector_shift;
if (start >= self.bytes.len) return error.Truncated;
const end = @min(start + self.sectorSize(), self.bytes.len);
return self.bytes[start..end];
}
fn fullSector(self: File, index: u32) Error![]const u8 {
const s = try self.sector(index);
if (s.len != self.sectorSize()) return error.Truncated;
return s;
}
/// Collect the FAT sector numbers (header DIFAT, then the DIFAT
/// sector chain) and concatenate those sectors into one table.
fn readFat(self: File, num_fat: u32, first_difat: u32, total_sectors: usize) Error![]u32 {
const sector_size = self.sectorSize();
const per_sector = sector_size / 4;
const fat_sectors = try self.allocator.alloc(u32, num_fat);
defer self.allocator.free(fat_sectors);
const in_header = @min(num_fat, header_difat_count);
for (0..in_header) |i| fat_sectors[i] = readU32(self.bytes, 76 + 4 * i);
var filled: usize = in_header;
var difat = first_difat;
var steps: usize = 0;
while (filled < num_fat) {
steps += 1;
if (difat > max_reg_sect or steps > total_sectors) return error.CorruptFile;
const s = try self.fullSector(difat);
// The last entry of a DIFAT sector links to the next one.
const take = @min(num_fat - filled, per_sector - 1);
for (0..take) |i| fat_sectors[filled + i] = readU32(s, 4 * i);
filled += take;
difat = readU32(s, sector_size - 4);
}
const fat = try self.allocator.alloc(u32, @as(usize, num_fat) * per_sector);
errdefer self.allocator.free(fat);
for (fat_sectors, 0..) |fs, n| {
const s = try self.fullSector(fs);
for (0..per_sector) |i| fat[n * per_sector + i] = readU32(s, 4 * i);
}
return fat;
}
fn readDirectory(self: File, first_dir: u32) Error![]Entry {
var entries: std.ArrayList(Entry) = .empty;
errdefer entries.deinit(self.allocator);
var it: ChainIterator = .{ .table = self.fat, .next_index = first_dir };
while (try it.next()) |index| {
const s = try self.fullSector(index);
var off: usize = 0;
while (off + dir_entry_size <= s.len) : (off += dir_entry_size) {
try entries.append(self.allocator, try parseEntry(s[off..][0..dir_entry_size], self.sector_shift == 9));
}
}
return entries.toOwnedSlice(self.allocator);
}
fn readMiniFat(self: File, first_mini_fat: u32) Error![]u32 {
var table: std.ArrayList(u32) = .empty;
errdefer table.deinit(self.allocator);
if (first_mini_fat == end_of_chain or first_mini_fat == free_sect) return table.toOwnedSlice(self.allocator);
var it: ChainIterator = .{ .table = self.fat, .next_index = first_mini_fat };
while (try it.next()) |index| {
const s = try self.fullSector(index);
var off: usize = 0;
while (off + 4 <= s.len) : (off += 4) try table.append(self.allocator, readU32(s, off));
}
return table.toOwnedSlice(self.allocator);
}
/// Read `size` bytes following the FAT chain from `start`.
fn readRegular(self: File, allocator: std.mem.Allocator, start: u32, size: u64) Error![]u8 {
// Beyond what the FAT can address is impossible; beyond the end
// of the bytes we have means the file was cut short.
if (size > @as(u64, self.fat.len) << self.sector_shift) return error.CorruptFile;
if (size > self.bytes.len) return error.Truncated;
const out = try allocator.alloc(u8, @intCast(size));
errdefer allocator.free(out);
var filled: usize = 0;
var it: ChainIterator = .{ .table = self.fat, .next_index = start };
while (filled < out.len) {
const index = (try it.next()) orelse return error.CorruptFile; // chain shorter than size
const s = try self.sector(index);
const n = @min(s.len, out.len - filled);
// A short final sector is only acceptable when it holds the
// rest of the stream.
if (n < self.sectorSize() and filled + n < out.len) return error.Truncated;
@memcpy(out[filled..][0..n], s[0..n]);
filled += n;
}
try it.finish();
return out;
}
/// Read `size` bytes following the mini FAT chain from `start`
/// inside the mini stream.
fn readMini(self: File, allocator: std.mem.Allocator, start: u32, size: u64) Error![]u8 {
if (size > self.mini_stream.len) return error.CorruptFile;
const out = try allocator.alloc(u8, @intCast(size));
errdefer allocator.free(out);
var filled: usize = 0;
var it: ChainIterator = .{ .table = self.mini_fat, .next_index = start };
while (filled < out.len) {
const index = (try it.next()) orelse return error.CorruptFile;
const off = @as(usize, index) * mini_sector_size;
if (off >= self.mini_stream.len) return error.CorruptFile;
const n = @min(mini_sector_size, out.len - filled, self.mini_stream.len - off);
// The mini stream ran out before the stream did.
if (n < mini_sector_size and filled + n < out.len) return error.CorruptFile;
@memcpy(out[filled..][0..n], self.mini_stream[off..][0..n]);
filled += n;
}
try it.finish();
return out;
}
};
/// Walks a sector chain through a FAT or mini FAT. Each step must land
/// inside the table, and a chain can visit at most `table.len` sectors,
/// so loops and dangling links are reported instead of followed.
const ChainIterator = struct {
table: []const u32,
next_index: u32,
steps: usize = 0,
fn next(it: *ChainIterator) Error!?u32 {
if (it.next_index == end_of_chain) return null;
if (it.next_index >= it.table.len) return error.CorruptFile;
it.steps += 1;
if (it.steps > it.table.len) return error.CorruptFile;
const current = it.next_index;
it.next_index = it.table[current];
return current;
}
/// Walk whatever is left of the chain. A reader stops once it has
/// a stream's declared size, so a loop late in the chain would
/// otherwise go unnoticed and its repeated sectors would be
/// returned as data. Extra sectors that do end are tolerated.
fn finish(it: *ChainIterator) Error!void {
while (try it.next()) |_| {}
}
};
fn parseEntry(raw: *const [dir_entry_size]u8, is_v3: bool) Error!Entry {
// Name length is in bytes and includes the UTF-16 NUL terminator.
const name_bytes = readU16(raw, 64);
if (name_bytes > 64 or name_bytes % 2 != 0) return error.CorruptFile;
const units: u8 = if (name_bytes == 0) 0 else @intCast(name_bytes / 2 - 1);
var e: Entry = .{
// SAFETY: the first `units` code units are written by the loop
// below, and nothing reads past `name_len`.
.name = undefined,
.name_len = units,
.kind = @enumFromInt(raw[66]),
.start_sector = readU32(raw, 116),
.size = std.mem.readInt(u64, raw[120..128], .little),
};
// Version 3 files only define the low 32 bits of the size; writers
// are allowed to leave junk in the high half (MS-CFB 2.6.3).
if (is_v3) e.size &= 0xFFFFFFFF;
for (0..units) |i| e.name[i] = readU16(raw, 2 * i);
return e;
}
fn readU16(bytes: []const u8, off: usize) u16 {
return std.mem.readInt(u16, bytes[off..][0..2], .little);
}
fn readU32(bytes: []const u8, off: usize) u32 {
return std.mem.readInt(u32, bytes[off..][0..4], .little);
}
// ---- Tests ----
const testing = std.testing;
const test_writer = @import("test_writer.zig");
test "isCompoundFile" {
try testing.expect(isCompoundFile(&signature));
try testing.expect(!isCompoundFile(signature[0..7]));
try testing.expect(!isCompoundFile("PK\x03\x04 a zip file, i.e. xlsx"));
}
test "reads a regular stream through the FAT" {
const allocator = testing.allocator;
const big = try allocator.alloc(u8, 10_000);
defer allocator.free(big);
for (big, 0..) |*b, i| b.* = @truncate(i *% 7);
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
const entry = file.find("Workbook").?;
const got = try file.readStream(allocator, entry);
defer allocator.free(got);
try testing.expectEqualSlices(u8, big, got);
}
test "reads small streams through the mini FAT" {
const allocator = testing.allocator;
const bytes = try test_writer.buildCfb(allocator, &.{
.{ .name = "First", .data = "a small stream, well under the 4096-byte cutoff" },
.{ .name = "Second", .data = "x" ** 130 }, // spans three mini sectors
}, .{});
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
const first = try file.readStream(allocator, file.find("First").?);
defer allocator.free(first);
try testing.expectEqualStrings("a small stream, well under the 4096-byte cutoff", first);
const second = try file.readStream(allocator, file.find("Second").?);
defer allocator.free(second);
try testing.expectEqualStrings("x" ** 130, second);
}
test "version 4 files use 4096-byte sectors" {
const allocator = testing.allocator;
const big = try allocator.alloc(u8, 9_000);
defer allocator.free(big);
@memset(big, 0x5A);
const bytes = try test_writer.buildCfb(allocator, &.{
.{ .name = "Workbook", .data = big },
.{ .name = "Tiny", .data = "tiny" },
}, .{ .major_version = 4 });
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
const got = try file.readStream(allocator, file.find("Workbook").?);
defer allocator.free(got);
try testing.expectEqualSlices(u8, big, got);
const tiny = try file.readStream(allocator, file.find("tiny").?);
defer allocator.free(tiny);
try testing.expectEqualStrings("tiny", tiny);
}
test "FAT sectors beyond the header's 109 are found through DIFAT sectors" {
// 109 FAT sectors address 109 * 128 sectors (~7 MiB). A stream
// larger than that forces the writer to spill into a DIFAT sector.
const allocator = testing.allocator;
const big = try allocator.alloc(u8, 8 * 1024 * 1024);
defer allocator.free(big);
for (big, 0..) |*b, i| b.* = @truncate(i >> 9);
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
defer allocator.free(bytes);
try testing.expect(readU32(bytes, 72) > 0); // the writer really did use DIFAT sectors
var file = try File.init(allocator, bytes);
defer file.deinit();
const got = try file.readStream(allocator, file.find("Workbook").?);
defer allocator.free(got);
try testing.expectEqualSlices(u8, big, got);
}
test "stream lookup is case-insensitive and type-aware" {
const allocator = testing.allocator;
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = "data" }}, .{});
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
try testing.expect(file.find("WORKBOOK") != null);
try testing.expect(file.find("workbook") != null);
try testing.expect(file.find("Workboo") == null);
// The root entry is not a stream, so its name never matches.
try testing.expect(file.find("Root Entry") == null);
}
test "Entry.nameEql rejects non-ASCII code units" {
var e: Entry = .{ .name = @splat(0), .name_len = 1, .kind = .stream, .start_sector = 0, .size = 0 };
e.name[0] = 0x00E9; // e-acute
try testing.expect(!e.nameEql("e"));
}
test "rejects a non-compound file" {
try testing.expectError(error.NotCompoundFile, File.init(testing.allocator, "not an ole file at all"));
}
test "rejects a header shorter than 512 bytes" {
try testing.expectError(error.Truncated, File.init(testing.allocator, &signature));
}
test "rejects bad header fields" {
const allocator = testing.allocator;
const good = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
defer allocator.free(good);
const Patch = struct { off: usize, value: u16 };
const patches = [_]Patch{
.{ .off = 28, .value = 0xFEFF }, // byte order
.{ .off = 26, .value = 5 }, // major version
.{ .off = 30, .value = 12 }, // sector shift does not match version 3
.{ .off = 32, .value = 7 }, // mini sector shift
};
for (patches) |p| {
const bad = try allocator.dupe(u8, good);
defer allocator.free(bad);
std.mem.writeInt(u16, bad[p.off..][0..2], p.value, .little);
try testing.expectError(error.CorruptFile, File.init(allocator, bad));
}
}
test "rejects a FAT sector count the file cannot hold" {
const allocator = testing.allocator;
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
defer allocator.free(bytes);
std.mem.writeInt(u32, bytes[44..48], 0xFFFFFF, .little);
try testing.expectError(error.CorruptFile, File.init(allocator, bytes));
}
test "rejects a version 4 file cut off inside its header sector" {
const allocator = testing.allocator;
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{ .major_version = 4 });
defer allocator.free(bytes);
try testing.expectError(error.Truncated, File.init(allocator, bytes[0..1000]));
}
test "a FAT sector listed past the end of the file is truncation" {
const allocator = testing.allocator;
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
defer allocator.free(bytes);
std.mem.writeInt(u32, bytes[76..80], 5000, .little); // header DIFAT[0]
try testing.expectError(error.Truncated, File.init(allocator, bytes));
}
test "a cyclic FAT chain is reported, not followed" {
const allocator = testing.allocator;
const big = try allocator.alloc(u8, 5000);
defer allocator.free(big);
@memset(big, 1);
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
const entry = file.find("Workbook").?;
// Point the stream's first sector back at itself.
file.fat[entry.start_sector] = entry.start_sector;
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
}
test "a chain shorter than the declared size is corrupt" {
const allocator = testing.allocator;
const big = try allocator.alloc(u8, 5000);
defer allocator.free(big);
@memset(big, 1);
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
const entry = file.find("Workbook").?;
file.fat[entry.start_sector] = end_of_chain;
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
var mini = entry;
mini.size = 10; // below the cutoff, so read through the (empty) mini stream
try testing.expectError(error.CorruptFile, file.readStream(allocator, mini));
}
test "a stream larger than the file is rejected before allocating" {
const allocator = testing.allocator;
const big = try allocator.alloc(u8, 5000);
defer allocator.free(big);
@memset(big, 1);
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
var entry = file.find("Workbook").?;
entry.size = 1 << 40;
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
}
test "a mini chain pointing outside the mini stream is corrupt" {
const allocator = testing.allocator;
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "x" ** 100 }}, .{});
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
const entry = file.find("S").?;
// The mini FAT has a full sector of entries (128) but the mini
// stream only holds two mini sectors, so entry 100 is in the table
// yet outside the stream.
file.mini_fat[entry.start_sector] = 100;
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
}
test "a mini stream that ends mid-chain is corrupt" {
const allocator = testing.allocator;
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "x" ** 100 }}, .{});
defer allocator.free(bytes);
var file = try File.init(allocator, bytes);
defer file.deinit();
var entry = file.find("S").?;
// Start the chain at the second mini sector and pretend the mini
// stream ends 36 bytes into it: the first hop yields a short read
// that cannot be the end of a 100-byte stream.
entry.start_sector = 1;
const full_len = file.mini_stream.len;
file.mini_stream.len = 100;
defer file.mini_stream.len = full_len;
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
}
test "a file cut short mid-stream reports Truncated" {
const allocator = testing.allocator;
const big = try allocator.alloc(u8, 20_000);
defer allocator.free(big);
@memset(big, 3);
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
defer allocator.free(bytes);
// The writer places the stream last, so dropping the tail cuts it.
var file = try File.init(allocator, bytes);
defer file.deinit();
const cut_at = bytes.len - 4096;
file.bytes = bytes[0..cut_at];
try testing.expectError(error.Truncated, file.readStream(allocator, file.find("Workbook").?));
// A cut that lands inside a sector leaves a short sector that is not
// the stream's last: also Truncated.
file.bytes = bytes[0 .. cut_at + 100];
try testing.expectError(error.Truncated, file.readStream(allocator, file.find("Workbook").?));
}
test "an unpadded final sector is accepted when it ends the stream" {
const allocator = testing.allocator;
const big = try allocator.alloc(u8, 5000);
defer allocator.free(big);
@memset(big, 9);
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
defer allocator.free(bytes);
// 5000 bytes = 9 full sectors + 392 bytes; drop the padding.
const unpadded = bytes[0 .. bytes.len - (512 - 392)];
var file = try File.init(allocator, unpadded);
defer file.deinit();
const got = try file.readStream(allocator, file.find("Workbook").?);
defer allocator.free(got);
try testing.expectEqualSlices(u8, big, got);
}
test "a directory entry with an impossible name length is corrupt" {
var raw: [dir_entry_size]u8 = @splat(0);
std.mem.writeInt(u16, raw[64..66], 66, .little);
try testing.expectError(error.CorruptFile, parseEntry(&raw, true));
std.mem.writeInt(u16, raw[64..66], 3, .little);
try testing.expectError(error.CorruptFile, parseEntry(&raw, true));
}
test "version 3 entries ignore the high half of the size" {
var raw: [dir_entry_size]u8 = @splat(0);
std.mem.writeInt(u64, raw[120..128], 0xDEADBEEF_00000010, .little);
try testing.expectEqual(@as(u64, 0x10), (try parseEntry(&raw, true)).size);
try testing.expectEqual(@as(u64, 0xDEADBEEF_00000010), (try parseEntry(&raw, false)).size);
}
test "the root entry must come first" {
const allocator = testing.allocator;
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
defer allocator.free(bytes);
// Root is the first entry of the first directory sector, which the
// writer places right after the FAT. Retype it as a stream.
const dir_off = (@as(usize, readU32(bytes, 48)) + 1) * 512;
bytes[dir_off + 66] = @intFromEnum(EntryType.stream);
try testing.expectError(error.CorruptFile, File.init(allocator, bytes));
}

183
src/root.zig Normal file
View file

@ -0,0 +1,183 @@
//! biff8: a read-only reader for legacy binary Excel workbooks (`.xls`,
//! Excel 97 through 2003).
//!
//! These files are BIFF8 record streams inside a Compound File Binary
//! ("OLE2") container. This library unwraps the container and decodes
//! each worksheet's cell values: text, numbers, booleans, error codes,
//! and the cached results of formulas. It does not decode formatting,
//! evaluate formulas, read `.xlsx` (a different format entirely), or
//! write anything.
//!
//! ```zig
//! var wb = try biff8.Workbook.parse(allocator, bytes);
//! defer wb.deinit();
//! const sheet = wb.sheets[0];
//! switch (sheet.cell(row, col)) {
//! .text => |t| ...,
//! .number => |n| ...,
//! else => {},
//! }
//! ```
//!
//! Numbers are returned exactly as stored, so a date-formatted cell is
//! an Excel serial day number. Converting it requires knowing the cell
//! is a date, which lives in the number format this library skips.
const std = @import("std");
const cfb = @import("cfb.zig");
const biff = @import("biff.zig");
pub const Cell = biff.Cell;
pub const Sheet = biff.Sheet;
pub const ParseError = cfb.Error || biff.Error || error{
/// The container holds no `Workbook` stream, so it is some other
/// kind of compound file (a `.doc`, an `.msg`, ...).
NoWorkbookStream,
};
/// True when `bytes` starts with the compound-file signature. Every
/// `.xls` does, but so do other legacy Office files; use it for cheap
/// content sniffing, not as proof the bytes are a workbook.
pub const isCompoundFile = cfb.isCompoundFile;
pub const Workbook = struct {
arena: std.heap.ArenaAllocator,
/// Worksheets in workbook order. Chart sheets, macro sheets and VB
/// modules are omitted.
sheets: []const Sheet,
/// Parse a workbook. `bytes` is only read during the call; the
/// result owns everything it references.
pub fn parse(allocator: std.mem.Allocator, bytes: []const u8) ParseError!Workbook {
var file = try cfb.File.init(allocator, bytes);
defer file.deinit();
const entry = file.find("Workbook") orelse {
// Excel 5 and 95 (BIFF5) named the stream "Book".
if (file.find("Book") != null) return error.UnsupportedBiffVersion;
return error.NoWorkbookStream;
};
const stream = try file.readStream(allocator, entry);
defer allocator.free(stream);
var arena = std.heap.ArenaAllocator.init(allocator);
errdefer arena.deinit();
const sheets = try biff.parseSheets(arena.allocator(), allocator, stream);
return .{ .arena = arena, .sheets = sheets };
}
pub fn deinit(self: *Workbook) void {
self.arena.deinit();
}
/// The first worksheet named exactly `name`.
pub fn sheet(self: *const Workbook, name: []const u8) ?*const Sheet {
for (self.sheets) |*s| {
if (std.mem.eql(u8, s.name, name)) return s;
}
return null;
}
};
// ---- Tests ----
const testing = std.testing;
const tw = @import("test_writer.zig");
test {
std.testing.refAllDecls(@This());
_ = cfb;
_ = biff;
}
test "parse: end to end through the compound file" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
const sst = try tw.sst(a, &.{ .{ .text = "Symbol" }, .{ .text = "SAMPLE" } }, 8224);
const bytes = try tw.xls(a, .{
.globals = sst,
.sheets = &.{
.{ .name = "Sample_Positions", .records = &.{
try tw.labelSst(a, 0, 0, 0),
try tw.labelSst(a, 1, 0, 1),
try tw.number(a, 1, 1, 100.25),
} },
.{ .name = "Notes" },
},
});
try testing.expect(isCompoundFile(bytes));
var wb = try Workbook.parse(testing.allocator, bytes);
defer wb.deinit();
try testing.expectEqual(@as(usize, 2), wb.sheets.len);
const s = wb.sheet("Sample_Positions").?;
try testing.expectEqualStrings("Symbol", s.cell(0, 0).asText().?);
try testing.expectEqualStrings("SAMPLE", s.cell(1, 0).asText().?);
try testing.expectEqual(@as(f64, 100.25), s.cell(1, 1).asNumber().?);
try testing.expect(wb.sheet("Notes") != null);
try testing.expect(wb.sheet("notes") == null);
}
test "parse: a large workbook stream lives outside the mini stream" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
var records: std.ArrayList(tw.Record) = .empty;
for (0..1000) |i| try records.append(a, try tw.number(a, @intCast(i), 0, @floatFromInt(i)));
const bytes = try tw.xls(a, .{ .sheets = &.{.{ .name = "Big", .records = records.items }} });
var wb = try Workbook.parse(testing.allocator, bytes);
defer wb.deinit();
try testing.expectEqual(@as(usize, 1000), wb.sheets[0].rows.len);
try testing.expectEqual(@as(f64, 999), wb.sheets[0].cell(999, 0).asNumber().?);
}
test "parse: compound files that are not BIFF8 workbooks" {
const allocator = testing.allocator;
const doc = try tw.buildCfb(allocator, &.{.{ .name = "WordDocument", .data = "not a workbook" }}, .{});
defer allocator.free(doc);
try testing.expectError(error.NoWorkbookStream, Workbook.parse(allocator, doc));
const biff5 = try tw.buildCfb(allocator, &.{.{ .name = "Book", .data = "excel 95" }}, .{});
defer allocator.free(biff5);
try testing.expectError(error.UnsupportedBiffVersion, Workbook.parse(allocator, biff5));
try testing.expectError(error.NotCompoundFile, Workbook.parse(allocator, "PK\x03\x04 xlsx is a zip"));
}
test "parse: errors after the container is read release everything" {
// testing.allocator fails the test on any leak along the error path.
const allocator = testing.allocator;
const bytes = try tw.buildCfb(allocator, &.{.{ .name = "Workbook", .data = "\x09\x08\x02\x00" }}, .{});
defer allocator.free(bytes);
try testing.expectError(error.CorruptFile, Workbook.parse(allocator, bytes));
}
fn parseAndRelease(allocator: std.mem.Allocator, bytes: []const u8) !void {
var wb = try Workbook.parse(allocator, bytes);
wb.deinit();
}
test "parse: every allocation failure is clean" {
var fx = std.heap.ArenaAllocator.init(testing.allocator);
defer fx.deinit();
const a = fx.allocator();
// Small (mini stream) and large (regular sectors) workbooks take
// different allocation paths through the container reader.
const sst = try tw.sst(a, &.{ .{ .text = "alpha" }, .{ .text = "\u{3042}" } }, 8224);
var records: std.ArrayList(tw.Record) = .empty;
try records.append(a, try tw.labelSst(a, 0, 0, 0));
try records.append(a, try tw.labelSst(a, 0, 1, 1));
const small = try tw.xls(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} });
for (1..600) |i| try records.append(a, try tw.number(a, @intCast(i), 0, 1));
const large = try tw.xls(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} });
try testing.checkAllAllocationFailures(testing.allocator, parseAndRelease, .{small});
try testing.checkAllAllocationFailures(testing.allocator, parseAndRelease, .{large});
}

508
src/test_writer.zig Normal file
View file

@ -0,0 +1,508 @@
//! Test-only builders for compound files and BIFF8 record streams.
//!
//! The reader's tests describe their fixtures in code with these
//! helpers instead of checking in binary `.xls` files. That keeps every
//! fixture readable and placeholder-only, and lets a test produce
//! layouts that real writers rarely emit (DIFAT sectors, strings split
//! across CONTINUE records at every possible point, unpaired UTF-16
//! surrogates).
//!
//! Nothing here is validated against a second implementation; the
//! reader's tests against it check self-consistency. The real-world
//! check is parsing an actual Excel-produced file, which lives outside
//! the test suite because such files carry personal data.
const std = @import("std");
const Allocator = std.mem.Allocator;
// ---- Compound file ----
pub const Stream = struct {
name: []const u8,
data: []const u8,
};
pub const CfbOptions = struct {
/// 3 (512-byte sectors) or 4 (4096-byte sectors).
major_version: u16 = 3,
};
const free_sect: u32 = 0xFFFFFFFF;
const end_of_chain: u32 = 0xFFFFFFFE;
const fat_sect: u32 = 0xFFFFFFFD;
const difat_sect: u32 = 0xFFFFFFFC;
const no_stream: u32 = 0xFFFFFFFF;
const mini_cutoff = 4096;
const mini_sector_size = 64;
/// Build a compound file holding `streams` in its root storage.
/// Streams under 4096 bytes go in the mini stream, larger ones get
/// their own sector chains. Layout, in sector order: FAT, DIFAT,
/// directory, mini FAT, mini stream, then each large stream (so the
/// last stream's data ends the file).
pub fn buildCfb(allocator: Allocator, streams: []const Stream, opts: CfbOptions) ![]u8 {
const shift: u5 = switch (opts.major_version) {
3 => 9,
4 => 12,
else => unreachable,
};
const sector_size: usize = @as(usize, 1) << shift;
const per_fat = sector_size / 4;
// Mini stream contents and per-stream placement.
var mini: std.ArrayList(u8) = .empty;
defer mini.deinit(allocator);
const starts = try allocator.alloc(u32, streams.len);
defer allocator.free(starts);
var mini_sectors: usize = 0;
var regular_sectors: usize = 0;
for (streams, 0..) |s, i| {
if (s.data.len < mini_cutoff) {
const n = ceilDiv(s.data.len, mini_sector_size);
starts[i] = if (n == 0) end_of_chain else @intCast(mini_sectors);
mini_sectors += n;
try mini.appendSlice(allocator, s.data);
try mini.appendNTimes(allocator, 0, n * mini_sector_size - s.data.len);
} else {
regular_sectors += ceilDiv(s.data.len, sector_size);
}
}
const dir_sectors = ceilDiv((streams.len + 1) * 128, sector_size);
const mini_fat_sectors = ceilDiv(mini_sectors * 4, sector_size);
const mini_stream_sectors = ceilDiv(mini.items.len, sector_size);
const data_sectors = dir_sectors + mini_fat_sectors + mini_stream_sectors + regular_sectors;
// Smallest FAT (plus the DIFAT sectors it needs) that addresses
// every sector including itself.
var fat_sectors: usize = 1;
var difat_sectors: usize = 0;
while (true) : (fat_sectors += 1) {
difat_sectors = if (fat_sectors > 109) ceilDiv(fat_sectors - 109, per_fat - 1) else 0;
if (fat_sectors * per_fat >= data_sectors + fat_sectors + difat_sectors) break;
}
const total_sectors = fat_sectors + difat_sectors + data_sectors;
const out = try allocator.alloc(u8, (total_sectors + 1) * sector_size);
errdefer allocator.free(out);
@memset(out, 0);
const fat = try allocator.alloc(u32, fat_sectors * per_fat);
defer allocator.free(fat);
@memset(fat, free_sect);
var next: usize = 0;
const fat_first = next;
for (0..fat_sectors) |i| fat[fat_first + i] = fat_sect;
next += fat_sectors;
const difat_first = next;
for (0..difat_sectors) |i| fat[difat_first + i] = difat_sect;
next += difat_sectors;
const dir_first = chain(fat, &next, dir_sectors);
const mini_fat_first = chain(fat, &next, mini_fat_sectors);
const mini_stream_first = chain(fat, &next, mini_stream_sectors);
for (streams, 0..) |s, i| {
if (s.data.len >= mini_cutoff) starts[i] = chain(fat, &next, ceilDiv(s.data.len, sector_size));
}
// Header.
const hdr = out[0..512];
@memcpy(hdr[0..8], &[_]u8{ 0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1 });
put16(hdr, 24, 0x003E);
put16(hdr, 26, opts.major_version);
put16(hdr, 28, 0xFFFE);
put16(hdr, 30, shift);
put16(hdr, 32, 6);
put32(hdr, 40, if (opts.major_version == 4) @intCast(dir_sectors) else 0);
put32(hdr, 44, @intCast(fat_sectors));
put32(hdr, 48, dir_first);
put32(hdr, 56, mini_cutoff);
put32(hdr, 60, mini_fat_first);
put32(hdr, 64, @intCast(mini_fat_sectors));
put32(hdr, 68, if (difat_sectors > 0) @intCast(difat_first) else end_of_chain);
put32(hdr, 72, @intCast(difat_sectors));
for (0..109) |i| put32(hdr, 76 + 4 * i, if (i < fat_sectors) @intCast(fat_first + i) else free_sect);
// DIFAT sectors: FAT sector numbers past the first 109, last slot
// links to the next DIFAT sector.
for (0..difat_sectors) |d| {
const s = sectorBytes(out, sector_size, difat_first + d);
for (0..per_fat - 1) |i| {
const n = 109 + d * (per_fat - 1) + i;
put32(s, 4 * i, if (n < fat_sectors) @intCast(fat_first + n) else free_sect);
}
put32(s, sector_size - 4, if (d + 1 < difat_sectors) @intCast(difat_first + d + 1) else end_of_chain);
}
// FAT.
for (fat, 0..) |v, i| put32(out[sector_size..], 4 * i, v);
// Directory: root, then each stream linked as a right-sibling list.
{
const dir = try allocator.alloc(u8, dir_sectors * sector_size);
defer allocator.free(dir);
@memset(dir, 0);
var e: usize = 0;
while (e * 128 < dir.len) : (e += 1) {
const raw = dir[e * 128 ..][0..128];
put32(raw, 68, no_stream);
put32(raw, 72, no_stream);
put32(raw, 76, no_stream);
}
writeEntry(dir[0..128], "Root Entry", 5, if (streams.len > 0) 1 else no_stream, no_stream, if (mini.items.len > 0) mini_stream_first else end_of_chain, mini.items.len);
for (streams, 0..) |s, i| {
const right: u32 = if (i + 1 < streams.len) @intCast(i + 2) else no_stream;
writeEntry(dir[(i + 1) * 128 ..][0..128], s.name, 2, no_stream, right, starts[i], s.data.len);
}
writeChain(out, sector_size, dir_first, dir);
}
// Mini FAT and mini stream.
{
const table = try allocator.alloc(u8, mini_fat_sectors * sector_size);
defer allocator.free(table);
@memset(table, 0xFF);
var idx: usize = 0;
for (streams) |s| {
if (s.data.len >= mini_cutoff) continue;
const n = ceilDiv(s.data.len, mini_sector_size);
for (0..n) |k| put32(table, 4 * (idx + k), if (k + 1 < n) @intCast(idx + k + 1) else end_of_chain);
idx += n;
}
writeChain(out, sector_size, mini_fat_first, table);
writeChain(out, sector_size, mini_stream_first, mini.items);
}
for (streams, 0..) |s, i| {
if (s.data.len >= mini_cutoff) writeChain(out, sector_size, starts[i], s.data);
}
return out;
}
/// Allocate `n` consecutive sectors as one chain. Returns the first
/// sector, or ENDOFCHAIN for an empty chain.
fn chain(fat: []u32, next: *usize, n: usize) u32 {
if (n == 0) return end_of_chain;
const first = next.*;
for (0..n) |i| fat[first + i] = if (i + 1 < n) @intCast(first + i + 1) else end_of_chain;
next.* += n;
return @intCast(first);
}
/// Copy `data` into consecutive sectors starting at `first` (the
/// builder always allocates chains contiguously).
fn writeChain(out: []u8, sector_size: usize, first: u32, data: []const u8) void {
if (data.len == 0) return;
const off = (@as(usize, first) + 1) * sector_size;
@memcpy(out[off..][0..data.len], data);
}
fn sectorBytes(out: []u8, sector_size: usize, index: usize) []u8 {
return out[(index + 1) * sector_size ..][0..sector_size];
}
fn writeEntry(raw: *[128]u8, name: []const u8, kind: u8, child: u32, right: u32, start: u32, size: usize) void {
for (name, 0..) |c, i| put16(raw, 2 * i, c);
put16(raw, 64, @intCast((name.len + 1) * 2));
raw[66] = kind;
raw[67] = 1; // black
put32(raw, 72, right);
put32(raw, 76, child);
put32(raw, 116, start);
std.mem.writeInt(u64, raw[120..128], size, .little);
}
// ---- BIFF8 records ----
pub const Record = struct {
kind: u16,
data: []const u8,
};
pub const rt = struct {
pub const bof: u16 = 0x0809;
pub const eof: u16 = 0x000A;
pub const filepass: u16 = 0x002F;
pub const boundsheet: u16 = 0x0085;
pub const sst: u16 = 0x00FC;
pub const @"continue": u16 = 0x003C;
pub const labelsst: u16 = 0x00FD;
pub const number: u16 = 0x0203;
pub const rk: u16 = 0x027E;
pub const mulrk: u16 = 0x00BD;
pub const label: u16 = 0x0204;
pub const rstring: u16 = 0x00D6;
pub const boolerr: u16 = 0x0205;
pub const formula: u16 = 0x0006;
pub const string: u16 = 0x0207;
pub const blank: u16 = 0x0201;
pub const dimensions: u16 = 0x0200;
};
pub const SheetSpec = struct {
name: []const u8,
records: []const Record = &.{},
/// BOUNDSHEET8 `dt`: 0 worksheet, 2 chart, 6 VB module.
boundsheet_type: u8 = 0,
/// BOF `dt` for the substream; 0x0010 worksheet, 0x0020 chart.
bof_type: u16 = 0x0010,
};
pub const WorkbookSpec = struct {
version: u16 = 0x0600,
/// Records between the globals BOF and the BOUNDSHEET records
/// (SST, FILEPASS, ...).
globals: []const Record = &.{},
sheets: []const SheetSpec = &.{},
/// Leave the final sheet without its EOF record.
omit_last_eof: bool = false,
/// Overrides every BOUNDSHEET offset (for out-of-range tests).
offset_override: ?u32 = null,
};
/// Assemble a Workbook stream: globals BOF, `globals`, one BOUNDSHEET
/// per sheet (offsets patched once the substreams are placed), EOF,
/// then each sheet substream.
pub fn workbook(allocator: Allocator, spec: WorkbookSpec) ![]u8 {
var out: std.ArrayList(u8) = .empty;
errdefer out.deinit(allocator);
try appendRecord(allocator, &out, rt.bof, &bof(spec.version, 0x0005));
for (spec.globals) |r| try appendRecord(allocator, &out, r.kind, r.data);
const offset_slots = try allocator.alloc(usize, spec.sheets.len);
defer allocator.free(offset_slots);
for (spec.sheets, 0..) |s, i| {
var data: std.ArrayList(u8) = .empty;
defer data.deinit(allocator);
try data.appendNTimes(allocator, 0, 4); // lbPlyPos, patched below
try data.append(allocator, 0); // visible
try data.append(allocator, s.boundsheet_type);
try data.append(allocator, @intCast(s.name.len));
try data.append(allocator, 0); // compressed characters
try data.appendSlice(allocator, s.name);
offset_slots[i] = out.items.len + 4;
try appendRecord(allocator, &out, rt.boundsheet, data.items);
}
try appendRecord(allocator, &out, rt.eof, "");
for (spec.sheets, 0..) |s, i| {
const offset: u32 = spec.offset_override orelse @intCast(out.items.len);
put32(out.items, offset_slots[i], offset);
try appendRecord(allocator, &out, rt.bof, &bof(spec.version, s.bof_type));
for (s.records) |r| try appendRecord(allocator, &out, r.kind, r.data);
if (!(spec.omit_last_eof and i + 1 == spec.sheets.len)) try appendRecord(allocator, &out, rt.eof, "");
}
return out.toOwnedSlice(allocator);
}
/// Workbook stream wrapped in a compound file as the `Workbook` stream.
pub fn xls(allocator: Allocator, spec: WorkbookSpec) ![]u8 {
const stream = try workbook(allocator, spec);
defer allocator.free(stream);
return buildCfb(allocator, &.{.{ .name = "Workbook", .data = stream }}, .{});
}
fn appendRecord(allocator: Allocator, out: *std.ArrayList(u8), kind: u16, data: []const u8) !void {
// SAFETY: both halves are written immediately below.
var head: [4]u8 = undefined;
put16(&head, 0, kind);
put16(&head, 2, @intCast(data.len));
try out.appendSlice(allocator, &head);
try out.appendSlice(allocator, data);
}
pub fn bof(version: u16, dt: u16) [16]u8 {
var b: [16]u8 = @splat(0);
put16(&b, 0, version);
put16(&b, 2, dt);
return b;
}
// The single-cell builders allocate their record bodies so the
// returned `Record` stays valid; tests pass an arena.
pub fn labelSst(allocator: Allocator, row: u16, col: u16, index: u32) !Record {
return cellRecord(allocator, rt.labelsst, row, col, u32, index);
}
pub fn number(allocator: Allocator, row: u16, col: u16, value: f64) !Record {
return cellRecord(allocator, rt.number, row, col, u64, @bitCast(value));
}
pub fn rk(allocator: Allocator, row: u16, col: u16, raw: u32) !Record {
return cellRecord(allocator, rt.rk, row, col, u32, raw);
}
pub fn boolErr(allocator: Allocator, row: u16, col: u16, value: u8, is_error: bool) !Record {
return cellRecord(allocator, rt.boolerr, row, col, u16, @as(u16, value) | (@as(u16, @intFromBool(is_error)) << 8));
}
pub fn blank(allocator: Allocator, row: u16, col: u16) !Record {
return cellRecord(allocator, rt.blank, row, col, void, {});
}
/// FORMULA record with an 8-byte cached value (see `formulaNumber`,
/// `formulaSpecial`) and an empty expression.
pub fn formula(allocator: Allocator, row: u16, col: u16, value: [8]u8) !Record {
const b = try allocator.alloc(u8, 6 + 8 + 2 + 4 + 2);
@memset(b, 0);
put16(b, 0, row);
put16(b, 2, col);
@memcpy(b[6..14], &value);
return .{ .kind = rt.formula, .data = b };
}
pub fn formulaNumber(value: f64) [8]u8 {
// SAFETY: writeInt fills all eight bytes.
var v: [8]u8 = undefined;
std.mem.writeInt(u64, &v, @bitCast(value), .little);
return v;
}
/// Non-numeric cached formula result: 0 string (in a following STRING
/// record), 1 boolean, 2 error, 3 empty string.
pub fn formulaSpecial(kind: u8, payload: u8) [8]u8 {
return .{ kind, 0, payload, 0, 0, 0, 0xFF, 0xFF };
}
/// MULRK: consecutive RK cells starting at `first_col`.
pub fn mulrk(allocator: Allocator, row: u16, first_col: u16, raws: []const u32) !Record {
const b = try allocator.alloc(u8, 4 + 6 * raws.len + 2);
put16(b, 0, row);
put16(b, 2, first_col);
for (raws, 0..) |r, i| {
put16(b, 4 + 6 * i, 0);
put32(b, 4 + 6 * i + 2, r);
}
put16(b, b.len - 2, first_col + @as(u16, @intCast(raws.len)) - 1);
return .{ .kind = rt.mulrk, .data = b };
}
/// LABEL / RSTRING with an inline compressed (Latin-1) string.
pub fn label(allocator: Allocator, kind: u16, row: u16, col: u16, text: []const u8) !Record {
const b = try allocator.alloc(u8, 6 + 3 + text.len);
@memset(b, 0);
put16(b, 0, row);
put16(b, 2, col);
put16(b, 6, @intCast(text.len));
@memcpy(b[9..], text);
return .{ .kind = kind, .data = b };
}
/// STRING record (formula string result), compressed characters.
pub fn string(allocator: Allocator, text: []const u8) !Record {
const b = try allocator.alloc(u8, 3 + text.len);
put16(b, 0, @intCast(text.len));
b[2] = 0;
@memcpy(b[3..], text);
return .{ .kind = rt.string, .data = b };
}
fn cellRecord(allocator: Allocator, kind: u16, row: u16, col: u16, comptime T: type, value: T) !Record {
const b = try allocator.alloc(u8, 6 + @sizeOf(T));
@memset(b, 0);
put16(b, 0, row);
put16(b, 2, col);
if (T != void) std.mem.writeInt(T, b[6..][0..@sizeOf(T)], value, .little);
return .{ .kind = kind, .data = b };
}
pub const SstString = struct {
/// UTF-8 text. Ignored when `units` is set.
text: []const u8 = "",
/// Raw UTF-16 code units, for strings UTF-8 cannot express (an
/// unpaired surrogate).
units: ?[]const u16 = null,
/// Rich-text run count; the runs themselves are filler bytes.
runs: u16 = 0,
/// Phonetic (ExtRst) bytes.
ext: []const u8 = "",
};
/// SST plus CONTINUE records, split so no record body exceeds
/// `max_chunk` bytes. Splits land wherever the limit falls: inside
/// a string's header, characters, runs or ExtRst. A split inside the
/// characters starts the next record with a fresh high-byte flag
/// chosen for the remaining characters, as Excel does, so one string
/// can switch between compressed and UTF-16 storage mid-way.
pub fn sst(allocator: Allocator, strings: []const SstString, max_chunk: usize) ![]Record {
var w: ChunkWriter = .{ .allocator = allocator, .max = max_chunk };
defer w.chunks.deinit(allocator);
try w.int(u32, @intCast(strings.len));
try w.int(u32, @intCast(strings.len));
for (strings) |s| {
const owned = if (s.units == null) try std.unicode.utf8ToUtf16LeAlloc(allocator, s.text) else null;
defer if (owned) |o| allocator.free(o);
const units = s.units orelse owned.?;
var flags: u8 = if (anyHigh(units)) 1 else 0;
if (s.ext.len > 0) flags |= 0x04;
if (s.runs > 0) flags |= 0x08;
try w.int(u16, @intCast(units.len));
try w.byte(flags);
if (s.runs > 0) try w.int(u16, s.runs);
if (s.ext.len > 0) try w.int(u32, @intCast(s.ext.len));
var high = flags & 1 != 0;
for (units, 0..) |u, i| {
const width: usize = if (high) 2 else 1;
if (w.current.items.len + width > w.max) {
try w.finish();
high = anyHigh(units[i..]);
try w.current.append(allocator, @intFromBool(high));
}
if (high) {
try w.current.append(allocator, @truncate(u));
try w.current.append(allocator, @truncate(u >> 8));
} else {
try w.current.append(allocator, @intCast(u));
}
}
for (0..@as(usize, s.runs) * 4) |_| try w.byte(0xAB);
for (s.ext) |b| try w.byte(b);
}
try w.finish();
const records = try allocator.alloc(Record, w.chunks.items.len);
for (w.chunks.items, 0..) |c, i| records[i] = .{ .kind = if (i == 0) rt.sst else rt.@"continue", .data = c };
return records;
}
const ChunkWriter = struct {
allocator: Allocator,
max: usize,
current: std.ArrayList(u8) = .empty,
chunks: std.ArrayList([]u8) = .empty,
fn byte(w: *ChunkWriter, b: u8) !void {
if (w.current.items.len == w.max) try w.finish();
try w.current.append(w.allocator, b);
}
fn int(w: *ChunkWriter, comptime T: type, v: T) !void {
for (0..@sizeOf(T)) |i| try w.byte(@truncate(v >> @intCast(8 * i)));
}
fn finish(w: *ChunkWriter) !void {
try w.chunks.append(w.allocator, try w.current.toOwnedSlice(w.allocator));
}
};
fn anyHigh(units: []const u16) bool {
for (units) |u| if (u > 0xFF) return true;
return false;
}
fn ceilDiv(a: usize, b: usize) usize {
return (a + b - 1) / b;
}
fn put16(b: []u8, off: usize, v: u16) void {
std.mem.writeInt(u16, b[off..][0..2], v, .little);
}
fn put32(b: []u8, off: usize, v: u32) void {
std.mem.writeInt(u32, b[off..][0..4], v, .little);
}