initial vibe coded commit
This commit is contained in:
parent
6903039a2a
commit
4f48adaa18
14 changed files with 2880 additions and 0 deletions
5
.gitignore
vendored
Normal file
5
.gitignore
vendored
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
.zig-cache/
|
||||
zig-out/
|
||||
zig-pkg/
|
||||
coverage/
|
||||
.tmp/
|
||||
7
.mise.toml
Normal file
7
.mise.toml
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
[tools]
|
||||
zig = "0.16.0"
|
||||
zls = "0.16.0"
|
||||
"github:j178/prek" = "0.4.1"
|
||||
|
||||
[tools."github:DonIsaac/zlint"]
|
||||
version = "0.9.0"
|
||||
51
.pre-commit-config.yaml
Normal file
51
.pre-commit-config.yaml
Normal file
|
|
@ -0,0 +1,51 @@
|
|||
# See https://pre-commit.com for more information
|
||||
# See https://pre-commit.com/hooks.html for more hooks
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v6.0.0
|
||||
hooks:
|
||||
- id: trailing-whitespace
|
||||
- id: end-of-file-fixer
|
||||
- id: check-yaml
|
||||
- id: check-added-large-files
|
||||
- repo: local
|
||||
hooks:
|
||||
- id: forbid-ai-punctuation
|
||||
name: Forbid smart punctuation (en/figure dash, minus, ellipsis, arrows, smart quotes)
|
||||
language: pygrep
|
||||
entry: '(–|‒|―|−|…|→|⇐|⇒|⇔|“|”|‘|’)'
|
||||
files: '\.(zig|zon|md|txt|toml|ya?ml)$'
|
||||
exclude: '^\.pre-commit-config\.yaml$'
|
||||
- id: forbid-prose-em-dash
|
||||
name: Forbid prose em-dash (use ASCII hyphen)
|
||||
language: pygrep
|
||||
entry: ' — '
|
||||
files: '\.(zig|zon|md|txt|toml|ya?ml)$'
|
||||
exclude: '^\.pre-commit-config\.yaml$'
|
||||
- repo: https://github.com/batmac/pre-commit-zig
|
||||
rev: v0.3.0
|
||||
hooks:
|
||||
- id: zig-fmt
|
||||
- repo: local
|
||||
hooks:
|
||||
- id: zlint
|
||||
name: Run zlint
|
||||
# zlint accepts file paths only via stdin (-S); positional
|
||||
# args are interpreted as directory names and silently
|
||||
# produce no output.
|
||||
entry: bash -c 'printf "%s\n" "$@" | zlint --deny-warnings --fix -S' --
|
||||
language: system
|
||||
types: [zig]
|
||||
- repo: https://github.com/batmac/pre-commit-zig
|
||||
rev: v0.3.0
|
||||
hooks:
|
||||
- id: zig-build
|
||||
- repo: local
|
||||
hooks:
|
||||
- id: test
|
||||
name: Run zig build coverage
|
||||
entry: zig
|
||||
args: ["build", "coverage", "-Dcoverage-threshold=99"]
|
||||
language: system
|
||||
types: [file]
|
||||
pass_filenames: false
|
||||
21
LICENSE
Normal file
21
LICENSE
Normal file
|
|
@ -0,0 +1,21 @@
|
|||
MIT License
|
||||
|
||||
Copyright (c) 2026 Emil Lerch
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
93
README.md
Normal file
93
README.md
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
# biff8
|
||||
|
||||
A read-only Zig reader for legacy binary Excel workbooks: `.xls` files from
|
||||
Excel 97 through 2003 (BIFF8), which some sites still export as their only
|
||||
"spreadsheet" download.
|
||||
|
||||
It unwraps the Compound File Binary ("OLE2") container, decodes the BIFF8
|
||||
record stream, and gives you each worksheet's cell values. That is all it
|
||||
does.
|
||||
|
||||
## Usage
|
||||
|
||||
```zig
|
||||
const biff8 = @import("biff8");
|
||||
|
||||
var wb = try biff8.Workbook.parse(allocator, bytes);
|
||||
defer wb.deinit();
|
||||
|
||||
const sheet = wb.sheet("Positions") orelse return error.NoSuchSheet;
|
||||
for (sheet.rows, 0..) |row, r| {
|
||||
for (row, 0..) |cell, c| switch (cell) {
|
||||
.text => |t| std.debug.print("{d},{d}: {s}\n", .{ r, c, t }),
|
||||
.number => |n| std.debug.print("{d},{d}: {d}\n", .{ r, c, n }),
|
||||
.boolean, .error_code, .empty => {},
|
||||
};
|
||||
}
|
||||
```
|
||||
|
||||
`sheet.cell(row, col)` does bounds-safe random access and returns `.empty`
|
||||
outside the populated area. `Sheet` has public fields, so a consumer can
|
||||
build one as a literal in its own tests instead of shipping binary fixtures.
|
||||
|
||||
`biff8.isCompoundFile(bytes)` is a cheap signature check for content
|
||||
sniffing. Every `.xls` passes it, but so does every other legacy Office
|
||||
file, so it is not proof of a workbook.
|
||||
|
||||
## What is decoded
|
||||
|
||||
| Record | Becomes |
|
||||
|---|---|
|
||||
| LABELSST, LABEL, RSTRING | `.text` (UTF-8) |
|
||||
| NUMBER, RK, MULRK | `.number` |
|
||||
| BOOLERR | `.boolean` or `.error_code` |
|
||||
| FORMULA (+ STRING) | the cached result, as any of the above |
|
||||
|
||||
Shared strings split across CONTINUE records are handled, including the
|
||||
case where the split falls mid-string and the remainder switches between
|
||||
1-byte and 2-byte characters. Unpaired UTF-16 surrogates decode as U+FFFD.
|
||||
|
||||
## What is not
|
||||
|
||||
- **Formatting.** Numbers are returned as stored, so a date-formatted cell
|
||||
is an Excel serial day number; telling dates apart needs the number
|
||||
format, which is not decoded.
|
||||
- **Formulas.** Only their cached results.
|
||||
- **Anything but BIFF8.** Excel 95 and earlier fail with
|
||||
`error.UnsupportedBiffVersion`. `.xlsx` is a different format (zipped
|
||||
XML) and fails with `error.NotCompoundFile`.
|
||||
- **Encrypted workbooks.** `error.Encrypted`.
|
||||
- **Writing.**
|
||||
|
||||
## Errors
|
||||
|
||||
`Workbook.parse` returns `biff8.ParseError`:
|
||||
|
||||
| Error | Meaning |
|
||||
|---|---|
|
||||
| `NotCompoundFile` | Not an OLE2 file at all |
|
||||
| `NoWorkbookStream` | An OLE2 file, but not a workbook (a `.doc`, an `.msg`, ...) |
|
||||
| `UnsupportedBiffVersion` | Excel 95 or older |
|
||||
| `Encrypted` | Password-protected |
|
||||
| `Truncated` | The file ends early, usually an interrupted download |
|
||||
| `CorruptFile` | Internally inconsistent structure |
|
||||
| `OutOfMemory` | |
|
||||
|
||||
Every sector chain walk is bounded and every declared size is checked
|
||||
before allocating, so hostile input produces an error rather than a hang or
|
||||
a huge allocation.
|
||||
|
||||
## Specs
|
||||
|
||||
- [MS-CFB] Compound File Binary File Format
|
||||
- [MS-XLS] Excel Binary File Format (.xls) Structure
|
||||
|
||||
## Development
|
||||
|
||||
```sh
|
||||
zig build test
|
||||
zig build coverage # kcov, Linux x86_64/aarch64
|
||||
```
|
||||
|
||||
Test fixtures are built in code by `src/test_writer.zig` rather than checked
|
||||
in as binary files.
|
||||
47
build.zig
Normal file
47
build.zig
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
const std = @import("std");
|
||||
const Coverage = @import("build/Coverage.zig");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const target = b.standardTargetOptions(.{});
|
||||
const optimize = b.standardOptimizeOption(.{});
|
||||
|
||||
// The public module. Consumers `@import("biff8")`.
|
||||
const mod = b.addModule("biff8", .{
|
||||
.root_source_file = b.path("src/root.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
});
|
||||
|
||||
// Tests: one binary rooted at src/root.zig. `refAllDecls` in
|
||||
// root.zig's test block pulls in every file's tests through the
|
||||
// import graph.
|
||||
const tests = b.addTest(.{ .root_module = mod });
|
||||
const test_step = b.step("test", "Run all tests");
|
||||
test_step.dependOn(&b.addRunArtifact(tests).step);
|
||||
|
||||
const lib = b.addLibrary(.{
|
||||
.name = "biff8",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/root.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
}),
|
||||
});
|
||||
const docs_step = b.step("docs", "Generate documentation");
|
||||
docs_step.dependOn(&b.addInstallDirectory(.{
|
||||
.source_dir = lib.getEmittedDocs(),
|
||||
.install_dir = .prefix,
|
||||
.install_subdir = "docs",
|
||||
}).step);
|
||||
|
||||
// Coverage: `zig build coverage` (kcov, Linux x86_64/aarch64 only)
|
||||
{
|
||||
var cov = Coverage.init(b);
|
||||
const cov_mod = b.createModule(.{
|
||||
.root_source_file = b.path("src/root.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
});
|
||||
_ = cov.addModule(cov_mod, "biff8");
|
||||
}
|
||||
}
|
||||
14
build.zig.zon
Normal file
14
build.zig.zon
Normal file
|
|
@ -0,0 +1,14 @@
|
|||
.{
|
||||
.name = .biff8,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x3401a894ea1f9c5a, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{
|
||||
"build.zig",
|
||||
"build.zig.zon",
|
||||
"src",
|
||||
"LICENSE",
|
||||
"README.md",
|
||||
},
|
||||
}
|
||||
239
build/Coverage.zig
Normal file
239
build/Coverage.zig
Normal file
|
|
@ -0,0 +1,239 @@
|
|||
const builtin = @import("builtin");
|
||||
const std = @import("std");
|
||||
const Build = std.Build;
|
||||
|
||||
const Coverage = @This();
|
||||
|
||||
/// Whether the host platform supports kcov-based coverage.
|
||||
/// Only x86_64 and aarch64 Linux are supported (kcov binary availability).
|
||||
/// On unsupported platforms, the coverage step will fail at runtime with
|
||||
/// a clear error from the kcov download or execution step.
|
||||
// pub const supported = builtin.os.tag == .linux and
|
||||
// (builtin.cpu.arch == .x86_64 or builtin.cpu.arch == .aarch64);
|
||||
|
||||
/// Initialize coverage infrastructure. Creates the "coverage" build step,
|
||||
/// registers build options (-Dcoverage-threshold, -Dcoverage-dir),
|
||||
/// and sets up the kcov download step. The kcov binary is downloaded into the
|
||||
/// zig cache on first use and reused thereafter.
|
||||
///
|
||||
/// Use `zig build coverage --verbose` to see per-file coverage breakdown.
|
||||
///
|
||||
/// Call `addModule()` on the returned value to add the test module to the
|
||||
/// coverage run.
|
||||
///
|
||||
/// Because addModule creates a new test executable from the root module provided,
|
||||
/// if there are any linking steps being done to your test executable, those
|
||||
/// must also be done to the test_exe returned by addModule.
|
||||
pub fn init(b: *Build) Coverage {
|
||||
// Add options
|
||||
const coverage_threshold = b.option(u7, "coverage-threshold", "Minimum coverage percentage required") orelse 0;
|
||||
const coverage_dir = b.option([]const u8, "coverage-dir", "Coverage output directory") orelse
|
||||
b.pathJoin(&.{ b.build_root.path orelse ".", "coverage" });
|
||||
const coverage_step = b.step("coverage", "Generate test coverage report");
|
||||
|
||||
// Set up kcov download.
|
||||
// We can't download directly because we are sandboxed during build, but
|
||||
// we can create a helper program and run it. First we need the destination
|
||||
// directory, keyed by architecture.
|
||||
const arch_name = switch (builtin.cpu.arch) {
|
||||
.x86_64 => "x86_64",
|
||||
.aarch64 => "aarch64",
|
||||
else => @tagName(builtin.cpu.arch),
|
||||
};
|
||||
|
||||
const Algo = std.crypto.hash.sha2.Sha256;
|
||||
var hasher = Algo.init(.{});
|
||||
hasher.update("kcov-");
|
||||
hasher.update(arch_name);
|
||||
var cache_hash: [Algo.digest_length]u8 = undefined;
|
||||
hasher.final(&cache_hash);
|
||||
|
||||
const cache_dir = b.pathJoin(&.{
|
||||
b.cache_root.path.?,
|
||||
"o",
|
||||
b.fmt("{s}", .{std.fmt.bytesToHex(cache_hash, .lower)}),
|
||||
});
|
||||
|
||||
const kcov_path = b.pathJoin(&.{ cache_dir, b.fmt("kcov-{s}", .{arch_name}) });
|
||||
|
||||
// Create the download helper executable
|
||||
const download_exe = b.addExecutable(.{
|
||||
.name = "download-kcov",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("build/download_kcov.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
|
||||
const run_download = b.addRunArtifact(download_exe);
|
||||
run_download.addArg(kcov_path);
|
||||
run_download.addArg(arch_name);
|
||||
|
||||
return .{
|
||||
.b = b,
|
||||
.coverage_step = coverage_step,
|
||||
.coverage_dir = coverage_dir,
|
||||
.coverage_threshold = coverage_threshold,
|
||||
.kcov_path = kcov_path,
|
||||
.run_download = run_download,
|
||||
};
|
||||
}
|
||||
|
||||
/// Add a test module to the coverage run. Runs kcov on the test binary,
|
||||
/// then reads the coverage JSON and prints a summary (with per-file
|
||||
/// breakdown if --verbose). Fails if below -Dcoverage-threshold.
|
||||
///
|
||||
/// Returns the test executable so the caller can add any extra linking steps.
|
||||
pub fn addModule(self: *Coverage, root_module: *Build.Module, name: []const u8) *Build.Step.Compile {
|
||||
const b = self.b;
|
||||
|
||||
// Set up kcov run: filter to src/ only, use custom CSS for HTML report
|
||||
const run_coverage = b.addSystemCommand(&.{self.kcov_path});
|
||||
const include_path = b.pathJoin(&.{ b.build_root.path.?, "src" });
|
||||
run_coverage.addArgs(&.{ "--include-path", include_path });
|
||||
const css_file = b.pathJoin(&.{ b.build_root.path.?, "build", "bcov.css" });
|
||||
run_coverage.addArg(b.fmt("--configure=css-file={s}", .{css_file}));
|
||||
run_coverage.addArg(self.coverage_dir);
|
||||
|
||||
// Create a test executable for this module.
|
||||
// We need to set use_llvm because the self-hosted backend
|
||||
// does not emit the DWARF data that kcov needs.
|
||||
const test_exe = b.addTest(.{
|
||||
.name = name,
|
||||
.root_module = root_module,
|
||||
.use_llvm = true,
|
||||
});
|
||||
run_coverage.addArtifactArg(test_exe);
|
||||
run_coverage.step.dependOn(&test_exe.step);
|
||||
run_coverage.step.dependOn(&self.run_download.step);
|
||||
|
||||
// Wire up the threshold check step after kcov completes
|
||||
const check = b.allocator.create(Check) catch @panic("OOM");
|
||||
check.* = .{
|
||||
.step = Build.Step.init(.{
|
||||
.id = .custom,
|
||||
.name = "check coverage",
|
||||
.owner = b,
|
||||
.makeFn = make,
|
||||
}),
|
||||
.json_path = b.fmt("{s}/{s}/coverage.json", .{ self.coverage_dir, name }),
|
||||
.threshold = self.coverage_threshold,
|
||||
};
|
||||
check.step.dependOn(&run_coverage.step);
|
||||
self.coverage_step.dependOn(&check.step);
|
||||
|
||||
return test_exe;
|
||||
}
|
||||
|
||||
// ── Coverage struct fields ──────────────────────────────────
|
||||
|
||||
// Fields used by init() to configure the shared coverage infrastructure
|
||||
b: *Build,
|
||||
coverage_step: *Build.Step,
|
||||
coverage_dir: []const u8,
|
||||
coverage_threshold: u7,
|
||||
kcov_path: []const u8,
|
||||
run_download: *Build.Step.Run,
|
||||
|
||||
// Per-module threshold-check step. Created in `addModule`; `make`
|
||||
// recovers the instance via `@fieldParentPtr("step", ...)`.
|
||||
const Check = struct {
|
||||
step: Build.Step,
|
||||
json_path: []const u8,
|
||||
threshold: u7,
|
||||
};
|
||||
|
||||
// This must be kept in step with kcov per-binary coverage.json format
|
||||
const CoverageReport = struct {
|
||||
files: []const CoverageFile,
|
||||
};
|
||||
|
||||
const CoverageFile = struct {
|
||||
file: []const u8,
|
||||
covered_lines: usize,
|
||||
total_lines: usize,
|
||||
};
|
||||
|
||||
const File = struct {
|
||||
file: []const u8,
|
||||
percent_covered: f64,
|
||||
covered_lines: usize,
|
||||
total_lines: usize,
|
||||
|
||||
pub fn coverageLessThanDesc(_: void, lhs: File, rhs: File) bool {
|
||||
return lhs.percent_covered > rhs.percent_covered;
|
||||
}
|
||||
};
|
||||
|
||||
/// Build step make function: reads kcov JSON output, prints a summary
|
||||
/// (with per-file breakdown if verbose), and fails if below threshold.
|
||||
fn make(step: *Build.Step, options: Build.Step.MakeOptions) !void {
|
||||
_ = options;
|
||||
const check: *Check = @fieldParentPtr("step", step);
|
||||
const allocator = step.owner.allocator;
|
||||
const io = step.owner.graph.io;
|
||||
|
||||
const file = std.Io.Dir.cwd().openFile(io, check.json_path, .{}) catch |err| {
|
||||
return step.fail("Failed to open coverage report {s}: {}", .{ check.json_path, err });
|
||||
};
|
||||
defer file.close(io);
|
||||
|
||||
var file_reader = file.reader(io, &.{});
|
||||
const content = try file_reader.interface.allocRemaining(allocator, .limited(10 * 1024 * 1024));
|
||||
defer allocator.free(content);
|
||||
|
||||
const json = std.json.parseFromSlice(CoverageReport, allocator, content, .{
|
||||
.ignore_unknown_fields = true,
|
||||
}) catch |err| {
|
||||
return step.fail("Failed to parse coverage JSON: {}", .{err});
|
||||
};
|
||||
defer json.deinit();
|
||||
|
||||
var total_covered: usize = 0;
|
||||
var total_lines: usize = 0;
|
||||
|
||||
var file_list = std.ArrayList(File).empty;
|
||||
defer file_list.deinit(allocator);
|
||||
|
||||
for (json.value.files) |f| {
|
||||
const pct: f64 = if (f.total_lines > 0)
|
||||
@as(f64, @floatFromInt(f.covered_lines)) / @as(f64, @floatFromInt(f.total_lines)) * 100.0
|
||||
else
|
||||
0;
|
||||
try file_list.append(allocator, .{
|
||||
.file = f.file,
|
||||
.covered_lines = f.covered_lines,
|
||||
.total_lines = f.total_lines,
|
||||
.percent_covered = pct,
|
||||
});
|
||||
total_covered += f.covered_lines;
|
||||
total_lines += f.total_lines;
|
||||
}
|
||||
|
||||
std.mem.sort(File, file_list.items, {}, File.coverageLessThanDesc);
|
||||
|
||||
var stdout_buffer: [1024]u8 = undefined;
|
||||
var stdout_writer = std.Io.File.stdout().writer(io, &stdout_buffer);
|
||||
const stdout = &stdout_writer.interface;
|
||||
if (step.owner.verbose) {
|
||||
for (file_list.items) |f| {
|
||||
try stdout.print(
|
||||
"{d: >5.1}% {d: >5}/{d: <5}:{s}\n",
|
||||
.{ f.percent_covered, f.covered_lines, f.total_lines, f.file },
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const total_pct: f64 = if (total_lines > 0)
|
||||
@as(f64, @floatFromInt(total_covered)) / @as(f64, @floatFromInt(total_lines)) * 100.0
|
||||
else
|
||||
0;
|
||||
try stdout.print(
|
||||
"Total test coverage: {d:.2}% ({d}/{d})\n",
|
||||
.{ total_pct, total_covered, total_lines },
|
||||
);
|
||||
try stdout.flush();
|
||||
|
||||
if (@as(u7, @intFromFloat(@floor(total_pct))) < check.threshold)
|
||||
return step.fail("Coverage {d:.2}% is below threshold {d}%", .{ total_pct, check.threshold });
|
||||
}
|
||||
46
build/bcov.css
Normal file
46
build/bcov.css
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
/* Based upon the lcov CSS style, style files can be reused - Dark Theme */
|
||||
body { color: #e0e0e0; background-color: #1e1e1e; }
|
||||
a:link { color: #6b9aff; text-decoration: underline; }
|
||||
a:visited { color: #4dbb7a; text-decoration: underline; }
|
||||
a:active { color: #ff6b8a; text-decoration: underline; }
|
||||
td.title { text-align: center; padding-bottom: 10px; font-size: 20pt; font-weight: bold; }
|
||||
td.ruler { background-color: #4a6ba8; }
|
||||
td.headerItem { text-align: right; padding-right: 6px; font-family: sans-serif; font-weight: bold; }
|
||||
td.headerValue { text-align: left; color: #6b9aff; font-family: sans-serif; font-weight: bold; }
|
||||
td.versionInfo { text-align: center; padding-top: 2px; }
|
||||
th.headerItem { text-align: right; padding-right: 6px; font-family: sans-serif; font-weight: bold; }
|
||||
th.headerValue { text-align: left; color: #6b9aff; font-family: sans-serif; font-weight: bold; }
|
||||
pre.source { font-family: monospace; white-space: pre; overflow: hidden; text-overflow: ellipsis; }
|
||||
span.lineNum { background-color: #5a5a2a; }
|
||||
span.lineNumLegend { background-color: #5a5a2a; width: 96px; font-weight: bold ;}
|
||||
span.lineCov { background-color: #2d5a2d; }
|
||||
span.linePartCov { background-color: #707000; }
|
||||
span.lineNoCov { background-color: #762c2c; }
|
||||
span.orderNum { background-color: #5a4a2a; float: right; width:5em; text-align: left; }
|
||||
span.orderNumLegend { background-color: #5a4a2a; width: 96px; font-weight: bold ;}
|
||||
span.coverHits { background-color: #4a4a2a; padding-left: 3px; padding-right: 1px; text-align: right; list-style-type: none; display: inline-block; width: 5em; }
|
||||
span.coverHitsLegend { background-color: #4a4a2a; width: 96px; font-weight: bold; margin: 0 auto;}
|
||||
td.tableHead { text-align: center; color: #e0e0e0; background-color: #4a6ba8; font-family: sans-serif; font-size: 120%; font-weight: bold; }
|
||||
td.coverFile { text-align: left; padding-left: 10px; padding-right: 20px; color: #6b9aff; font-family: monospace; background-color: #3a3a3a; }
|
||||
td.coverBar { padding-left: 10px; padding-right: 10px; background-color: #3a3a3a; }
|
||||
td.coverBarOutline { background-color: #4a4a4a; }
|
||||
td.coverPer { text-align: left; padding-left: 10px; padding-right: 10px; font-weight: bold; background-color: #3a3a3a; color: #e0e0e0; }
|
||||
td.coverPerLeftMed { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #5a5a00; font-weight: bold; color: #e0e0e0; }
|
||||
td.coverPerLeftLo { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #5a2d2d; font-weight: bold; color: #e0e0e0; }
|
||||
td.coverPerLeftHi { text-align: left; padding-left: 10px; padding-right: 10px; background-color: #2d5a2d; font-weight: bold; color: #e0e0e0; }
|
||||
td.coverNum { text-align: right; padding-left: 10px; padding-right: 10px; background-color: #3a3a3a; color: #e0e0e0; }
|
||||
|
||||
/* Override tablesorter hover styles for dark theme */
|
||||
.tablesorter-blue tbody > tr:hover > td,
|
||||
.tablesorter-blue tbody > tr:hover + tr.tablesorter-childRow > td,
|
||||
.tablesorter-blue tbody > tr:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td,
|
||||
.tablesorter-blue tbody > tr.even:hover > td,
|
||||
.tablesorter-blue tbody > tr.even:hover + tr.tablesorter-childRow > td,
|
||||
.tablesorter-blue tbody > tr.even:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td {
|
||||
background: #4a4a4a;
|
||||
}
|
||||
.tablesorter-blue tbody > tr.odd:hover > td,
|
||||
.tablesorter-blue tbody > tr.odd:hover + tr.tablesorter-childRow > td,
|
||||
.tablesorter-blue tbody > tr.odd:hover + tr.tablesorter-childRow + tr.tablesorter-childRow > td {
|
||||
background: #4a4a4a;
|
||||
}
|
||||
85
build/download_kcov.zig
Normal file
85
build/download_kcov.zig
Normal file
|
|
@ -0,0 +1,85 @@
|
|||
const std = @import("std");
|
||||
|
||||
pub fn main(init: std.process.Init) !void {
|
||||
// Build-time helper: short-lived process that downloads a single
|
||||
// file. Arena lets us skip per-allocation `defer free(...)` and
|
||||
// amortizes the allocation cost across the run via the arena's
|
||||
// exponential block growth. Process exit reclaims everything.
|
||||
const allocator = init.arena.allocator();
|
||||
const io = init.io;
|
||||
|
||||
const args = try init.minimal.args.toSlice(allocator);
|
||||
|
||||
if (args.len != 3) return error.InvalidArgs;
|
||||
|
||||
const kcov_path = args[1];
|
||||
const arch_name = args[2];
|
||||
|
||||
// Check to see if file exists. If it does, we have nothing more to do
|
||||
const stat = std.Io.Dir.cwd().statFile(io, kcov_path, .{}) catch |err| blk: {
|
||||
if (err == error.FileNotFound) break :blk null else return err;
|
||||
};
|
||||
// This might be better checking whether it's executable and >= 7MB, but
|
||||
// for now, we'll do a simple exists check
|
||||
if (stat != null) return;
|
||||
var stdout_buffer: [1024]u8 = undefined;
|
||||
var stdout_writer = std.Io.File.stdout().writer(io, &stdout_buffer);
|
||||
const stdout = &stdout_writer.interface;
|
||||
|
||||
try stdout.writeAll("Determining latest kcov version\n");
|
||||
try stdout.flush();
|
||||
|
||||
var client = std.http.Client{ .allocator = allocator, .io = io };
|
||||
defer client.deinit();
|
||||
|
||||
// Get redirect to find latest version
|
||||
const list_uri = try std.Uri.parse("https://git.lerch.org/lobo/-/packages/generic/kcov/");
|
||||
var req = try client.request(.GET, list_uri, .{ .redirect_behavior = .unhandled });
|
||||
defer req.deinit();
|
||||
|
||||
try req.sendBodiless();
|
||||
var redirect_buf: [1024]u8 = undefined;
|
||||
const response = try req.receiveHead(&redirect_buf);
|
||||
|
||||
if (response.head.status != .see_other) return error.UnexpectedResponse;
|
||||
|
||||
const location = response.head.location orelse return error.NoLocation;
|
||||
const version_start = std.mem.lastIndexOfScalar(u8, location, '/') orelse return error.InvalidLocation;
|
||||
const version = location[version_start + 1 ..];
|
||||
|
||||
try stdout.print(
|
||||
"Downloading kcov version {s} for {s} to {s}...",
|
||||
.{ version, arch_name, kcov_path },
|
||||
);
|
||||
try stdout.flush();
|
||||
|
||||
const binary_url = try std.fmt.allocPrint(
|
||||
allocator,
|
||||
"https://git.lerch.org/api/packages/lobo/generic/kcov/{s}/kcov-{s}",
|
||||
.{ version, arch_name },
|
||||
);
|
||||
|
||||
const cache_dir = std.fs.path.dirname(kcov_path) orelse return error.InvalidPath;
|
||||
std.Io.Dir.cwd().createDir(io, cache_dir, std.Io.File.Permissions.default_dir) catch |e| switch (e) {
|
||||
error.PathAlreadyExists => {},
|
||||
else => return e,
|
||||
};
|
||||
|
||||
const uri = try std.Uri.parse(binary_url);
|
||||
const file = try std.Io.Dir.cwd().createFile(io, kcov_path, .{});
|
||||
defer file.close(io);
|
||||
try file.setPermissions(io, @enumFromInt(0o755));
|
||||
|
||||
var buffer: [8192]u8 = undefined;
|
||||
var writer = file.writer(io, &buffer);
|
||||
const result = try client.fetch(.{
|
||||
.location = .{ .uri = uri },
|
||||
.response_writer = &writer.interface,
|
||||
});
|
||||
|
||||
if (result.status != .ok) return error.DownloadFailed;
|
||||
try writer.interface.flush();
|
||||
|
||||
try stdout.writeAll("done\n");
|
||||
try stdout.flush();
|
||||
}
|
||||
910
src/biff.zig
Normal file
910
src/biff.zig
Normal file
|
|
@ -0,0 +1,910 @@
|
|||
//! BIFF8 decoding: turns the `Workbook` stream of a legacy `.xls` file
|
||||
//! (Excel 97 through 2003) into sheets of cell values.
|
||||
//!
|
||||
//! The stream is a flat sequence of records (`u16` type, `u16` length,
|
||||
//! body). It opens with a "workbook globals" substream (BOF ... EOF)
|
||||
//! holding the shared string table and one BOUNDSHEET8 record per
|
||||
//! sheet; each BOUNDSHEET8 gives the stream offset of that sheet's own
|
||||
//! BOF ... EOF substream, which holds its cell records.
|
||||
//!
|
||||
//! Decoded: LABELSST, LABEL, RSTRING (text), NUMBER, RK, MULRK
|
||||
//! (numbers), BOOLERR (booleans and error codes), and FORMULA cached
|
||||
//! results (with the STRING record that carries a text result).
|
||||
//! Formatting, formulas themselves, merged cells, comments and charts
|
||||
//! are ignored. Numbers are returned raw: a date-formatted cell is an
|
||||
//! Excel serial number, because interpreting it needs the cell's
|
||||
//! number format, which this reader does not decode.
|
||||
//!
|
||||
//! The shared string table (SST) is the one tricky part. A record body
|
||||
//! is capped at 8224 bytes, so a long table spills into CONTINUE
|
||||
//! records, and a string may be split across that boundary. When the
|
||||
//! split falls inside a string's characters, the CONTINUE body begins
|
||||
//! with a fresh "high byte" flag saying whether the rest of the
|
||||
//! characters are 1-byte (Latin-1) or 2-byte (UTF-16), and it can
|
||||
//! differ from the flag the string started with. Splits elsewhere
|
||||
//! (string header, rich-text runs, phonetic data) carry no flag byte.
|
||||
//!
|
||||
//! Spec: [MS-XLS] Excel Binary File Format (.xls) Structure.
|
||||
|
||||
const std = @import("std");
|
||||
const Allocator = std.mem.Allocator;
|
||||
|
||||
pub const Error = error{
|
||||
/// Records are inconsistent: a length past the end of the stream,
|
||||
/// a string index outside the shared string table, a malformed
|
||||
/// record body.
|
||||
CorruptFile,
|
||||
/// The stream ends before a substream's EOF record.
|
||||
Truncated,
|
||||
/// Not BIFF8. Excel 95 and earlier (BIFF5 and below) are out of
|
||||
/// scope.
|
||||
UnsupportedBiffVersion,
|
||||
/// The workbook is password-protected (a FILEPASS record).
|
||||
Encrypted,
|
||||
OutOfMemory,
|
||||
};
|
||||
|
||||
/// One cell's value.
|
||||
pub const Cell = union(enum) {
|
||||
empty,
|
||||
/// UTF-8.
|
||||
text: []const u8,
|
||||
/// Raw double. Date cells are Excel serial numbers (see module doc).
|
||||
number: f64,
|
||||
boolean: bool,
|
||||
/// Excel error code: 0x00 #NULL!, 0x07 #DIV/0!, 0x0F #VALUE!,
|
||||
/// 0x17 #REF!, 0x1D #NAME?, 0x24 #NUM!, 0x2A #N/A.
|
||||
error_code: u8,
|
||||
|
||||
pub fn asText(self: Cell) ?[]const u8 {
|
||||
return switch (self) {
|
||||
.text => |t| t,
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn asNumber(self: Cell) ?f64 {
|
||||
return switch (self) {
|
||||
.number => |n| n,
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// A worksheet's cells, row-major. `rows[r]` is as wide as the
|
||||
/// rightmost populated cell of row `r` (possibly empty), so a sparse
|
||||
/// sheet costs memory proportional to its content. Use `cell` for
|
||||
/// bounds-safe access.
|
||||
///
|
||||
/// The fields are public so callers can build a `Sheet` literal in
|
||||
/// their own tests without producing a binary file.
|
||||
pub const Sheet = struct {
|
||||
name: []const u8,
|
||||
rows: []const []const Cell,
|
||||
|
||||
/// The cell at zero-based (`row`, `col`), or `.empty` when out of
|
||||
/// range.
|
||||
pub fn cell(self: Sheet, row: usize, col: usize) Cell {
|
||||
if (row >= self.rows.len) return .empty;
|
||||
const r = self.rows[row];
|
||||
if (col >= r.len) return .empty;
|
||||
return r[col];
|
||||
}
|
||||
};
|
||||
|
||||
const rt = struct {
|
||||
const bof: u16 = 0x0809;
|
||||
const eof: u16 = 0x000A;
|
||||
const filepass: u16 = 0x002F;
|
||||
const boundsheet: u16 = 0x0085;
|
||||
const sst: u16 = 0x00FC;
|
||||
const @"continue": u16 = 0x003C;
|
||||
const labelsst: u16 = 0x00FD;
|
||||
const number: u16 = 0x0203;
|
||||
const rk: u16 = 0x027E;
|
||||
const mulrk: u16 = 0x00BD;
|
||||
const label: u16 = 0x0204;
|
||||
const rstring: u16 = 0x00D6;
|
||||
const boolerr: u16 = 0x0205;
|
||||
const formula: u16 = 0x0006;
|
||||
const string: u16 = 0x0207;
|
||||
// BOF record numbers of BIFF2, BIFF3 and BIFF4.
|
||||
const bof_biff2: u16 = 0x0009;
|
||||
const bof_biff3: u16 = 0x0209;
|
||||
const bof_biff4: u16 = 0x0409;
|
||||
};
|
||||
|
||||
const biff8_version: u16 = 0x0600;
|
||||
const dt_globals: u16 = 0x0005;
|
||||
const dt_worksheet: u16 = 0x0010;
|
||||
/// BOUNDSHEET8 `dt` for a worksheet (or dialog sheet).
|
||||
const sheet_type_worksheet: u8 = 0;
|
||||
|
||||
/// Decode every worksheet in a BIFF8 `Workbook` stream, in workbook
|
||||
/// order. Charts, macro sheets and VB modules are skipped.
|
||||
///
|
||||
/// Everything returned is allocated from `arena` and never freed
|
||||
/// individually. `scratch` holds temporaries, all freed before return.
|
||||
pub fn parseSheets(arena: Allocator, scratch: Allocator, stream: []const u8) Error![]const Sheet {
|
||||
var recs: Records = .{ .stream = stream };
|
||||
const first = (try recs.next()) orelse return error.Truncated;
|
||||
try expectBof(first, dt_globals);
|
||||
|
||||
var bounds: std.ArrayList(BoundSheet) = .empty;
|
||||
defer bounds.deinit(scratch);
|
||||
var sst: []const []const u8 = &.{};
|
||||
|
||||
while (true) {
|
||||
const rec = (try recs.next()) orelse return error.Truncated;
|
||||
switch (rec.kind) {
|
||||
rt.eof => break,
|
||||
rt.filepass => return error.Encrypted,
|
||||
rt.boundsheet => try bounds.append(scratch, try parseBoundSheet(arena, scratch, rec.data)),
|
||||
rt.sst => sst = try parseSst(arena, scratch, &recs, rec.data),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
var sheets: std.ArrayList(Sheet) = .empty;
|
||||
for (bounds.items) |b| {
|
||||
if (b.kind != sheet_type_worksheet) continue;
|
||||
try sheets.append(arena, try parseSheet(arena, scratch, stream, b, sst));
|
||||
}
|
||||
return sheets.items;
|
||||
}
|
||||
|
||||
const BoundSheet = struct {
|
||||
name: []const u8,
|
||||
offset: u32,
|
||||
kind: u8,
|
||||
};
|
||||
|
||||
fn parseBoundSheet(arena: Allocator, scratch: Allocator, data: []const u8) Error!BoundSheet {
|
||||
if (data.len < 8) return error.CorruptFile;
|
||||
var r: Chunks = .{ .chunks = &.{data[6..]} };
|
||||
// ShortXLUnicodeString: one-byte length, then flags and characters.
|
||||
const cch = try r.byte();
|
||||
return .{
|
||||
.offset = readInt(u32, data, 0),
|
||||
.kind = data[5],
|
||||
.name = try r.characters(arena, scratch, cch, (try r.byte()) & 1 != 0),
|
||||
};
|
||||
}
|
||||
|
||||
/// SST body plus every CONTINUE record that follows it.
|
||||
fn parseSst(arena: Allocator, scratch: Allocator, recs: *Records, first: []const u8) Error![]const []const u8 {
|
||||
var chunks: std.ArrayList([]const u8) = .empty;
|
||||
defer chunks.deinit(scratch);
|
||||
try chunks.append(scratch, first);
|
||||
while (recs.peekKind() == rt.@"continue") try chunks.append(scratch, (try recs.next()).?.data);
|
||||
|
||||
var r: Chunks = .{ .chunks = chunks.items };
|
||||
_ = try r.int(u32); // total references; irrelevant to a reader
|
||||
const unique = try r.int(u32);
|
||||
// Every string costs at least three bytes (length + flags), which
|
||||
// bounds the allocation a corrupt count can request.
|
||||
var total_len: usize = 0;
|
||||
for (chunks.items) |c| total_len += c.len;
|
||||
if (unique > total_len / 3) return error.CorruptFile;
|
||||
|
||||
const strings = try arena.alloc([]const u8, unique);
|
||||
for (strings) |*s| {
|
||||
// XLUnicodeRichExtendedString.
|
||||
const cch = try r.int(u16);
|
||||
const flags = try r.byte();
|
||||
const runs: usize = if (flags & 0x08 != 0) try r.int(u16) else 0;
|
||||
const ext: usize = if (flags & 0x04 != 0) try r.int(u32) else 0;
|
||||
s.* = try r.characters(arena, scratch, cch, flags & 1 != 0);
|
||||
try r.skip(4 * runs);
|
||||
try r.skip(ext);
|
||||
}
|
||||
return strings;
|
||||
}
|
||||
|
||||
const Placed = struct {
|
||||
row: u16,
|
||||
col: u16,
|
||||
cell: Cell,
|
||||
};
|
||||
|
||||
fn parseSheet(arena: Allocator, scratch: Allocator, stream: []const u8, b: BoundSheet, sst: []const []const u8) Error!Sheet {
|
||||
if (b.offset >= stream.len) return error.CorruptFile;
|
||||
var recs: Records = .{ .stream = stream, .pos = b.offset };
|
||||
const first = (try recs.next()) orelse return error.CorruptFile;
|
||||
try expectBof(first, dt_worksheet);
|
||||
|
||||
var cells: std.ArrayList(Placed) = .empty;
|
||||
defer cells.deinit(scratch);
|
||||
|
||||
// Embedded charts are complete BOF ... EOF substreams nested in the
|
||||
// sheet's; their records are not this sheet's cells.
|
||||
var depth: usize = 0;
|
||||
// A FORMULA whose cached result is text is followed by a STRING
|
||||
// record carrying that text.
|
||||
var pending_string: ?Placed = null;
|
||||
|
||||
while (true) {
|
||||
const rec = (try recs.next()) orelse return error.Truncated;
|
||||
const d = rec.data;
|
||||
switch (rec.kind) {
|
||||
rt.bof => {
|
||||
depth += 1;
|
||||
continue;
|
||||
},
|
||||
rt.eof => {
|
||||
if (depth == 0) break;
|
||||
depth -= 1;
|
||||
continue;
|
||||
},
|
||||
else => if (depth > 0) continue,
|
||||
}
|
||||
|
||||
switch (rec.kind) {
|
||||
rt.labelsst => {
|
||||
if (d.len < 10) return error.CorruptFile;
|
||||
const index = readInt(u32, d, 6);
|
||||
if (index >= sst.len) return error.CorruptFile;
|
||||
try put(scratch, &cells, d, .{ .text = sst[index] });
|
||||
},
|
||||
rt.number => {
|
||||
if (d.len < 14) return error.CorruptFile;
|
||||
try put(scratch, &cells, d, .{ .number = @bitCast(readInt(u64, d, 6)) });
|
||||
},
|
||||
rt.rk => {
|
||||
if (d.len < 10) return error.CorruptFile;
|
||||
try put(scratch, &cells, d, .{ .number = decodeRk(readInt(u32, d, 6)) });
|
||||
},
|
||||
rt.mulrk => {
|
||||
// row, first col, N x (xf index, RK), last col.
|
||||
if (d.len < 12 or (d.len - 6) % 6 != 0) return error.CorruptFile;
|
||||
const row = readInt(u16, d, 0);
|
||||
const first_col = readInt(u16, d, 2);
|
||||
const n = (d.len - 6) / 6;
|
||||
if (@as(usize, readInt(u16, d, d.len - 2)) + 1 != @as(usize, first_col) + n) return error.CorruptFile;
|
||||
for (0..n) |i| {
|
||||
try cells.append(scratch, .{
|
||||
.row = row,
|
||||
.col = first_col + @as(u16, @intCast(i)),
|
||||
.cell = .{ .number = decodeRk(readInt(u32, d, 4 + 6 * i + 2)) },
|
||||
});
|
||||
}
|
||||
},
|
||||
// Both carry an XLUnicodeString after the cell header;
|
||||
// RSTRING's trailing formatting runs are ignored.
|
||||
rt.label, rt.rstring => {
|
||||
if (d.len < 9) return error.CorruptFile;
|
||||
var r: Chunks = .{ .chunks = &.{d[6..]} };
|
||||
try put(scratch, &cells, d, .{ .text = try r.unicodeString(arena, scratch) });
|
||||
},
|
||||
rt.boolerr => {
|
||||
if (d.len < 8) return error.CorruptFile;
|
||||
const cell: Cell = if (d[7] != 0) .{ .error_code = d[6] } else .{ .boolean = d[6] != 0 };
|
||||
try put(scratch, &cells, d, cell);
|
||||
},
|
||||
rt.formula => {
|
||||
if (d.len < 20) return error.CorruptFile;
|
||||
pending_string = null;
|
||||
const v = d[6..14];
|
||||
// 0xFFFF in the top two bytes marks a non-numeric result.
|
||||
if (readInt(u16, v, 6) != 0xFFFF) {
|
||||
try put(scratch, &cells, d, .{ .number = @bitCast(readInt(u64, v, 0)) });
|
||||
} else switch (v[0]) {
|
||||
0 => pending_string = .{ .row = readInt(u16, d, 0), .col = readInt(u16, d, 2), .cell = .empty },
|
||||
1 => try put(scratch, &cells, d, .{ .boolean = v[2] != 0 }),
|
||||
2 => try put(scratch, &cells, d, .{ .error_code = v[2] }),
|
||||
3 => try put(scratch, &cells, d, .{ .text = "" }),
|
||||
else => return error.CorruptFile,
|
||||
}
|
||||
},
|
||||
rt.string => if (pending_string) |p| {
|
||||
var chunks: std.ArrayList([]const u8) = .empty;
|
||||
defer chunks.deinit(scratch);
|
||||
try chunks.append(scratch, d);
|
||||
while (recs.peekKind() == rt.@"continue") try chunks.append(scratch, (try recs.next()).?.data);
|
||||
var r: Chunks = .{ .chunks = chunks.items };
|
||||
try cells.append(scratch, .{ .row = p.row, .col = p.col, .cell = .{ .text = try r.unicodeString(arena, scratch) } });
|
||||
pending_string = null;
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
return .{ .name = b.name, .rows = try buildRows(arena, scratch, cells.items) };
|
||||
}
|
||||
|
||||
/// Record a cell whose row and column are the first four bytes of `d`.
|
||||
fn put(scratch: Allocator, cells: *std.ArrayList(Placed), d: []const u8, cell: Cell) Error!void {
|
||||
try cells.append(scratch, .{ .row = readInt(u16, d, 0), .col = readInt(u16, d, 2), .cell = cell });
|
||||
}
|
||||
|
||||
/// Lay placed cells out as rows. A later record for the same cell
|
||||
/// wins, matching how Excel would overwrite it.
|
||||
fn buildRows(arena: Allocator, scratch: Allocator, cells: []const Placed) Error![]const []const Cell {
|
||||
var row_count: usize = 0;
|
||||
for (cells) |c| row_count = @max(row_count, @as(usize, c.row) + 1);
|
||||
|
||||
const widths = try scratch.alloc(usize, row_count);
|
||||
defer scratch.free(widths);
|
||||
@memset(widths, 0);
|
||||
for (cells) |c| widths[c.row] = @max(widths[c.row], @as(usize, c.col) + 1);
|
||||
|
||||
const rows = try arena.alloc([]Cell, row_count);
|
||||
for (rows, widths) |*r, w| {
|
||||
r.* = try arena.alloc(Cell, w);
|
||||
@memset(r.*, .empty);
|
||||
}
|
||||
for (cells) |c| rows[c.row][c.col] = c.cell;
|
||||
return rows;
|
||||
}
|
||||
|
||||
/// RK: a compressed number. Bit 1 selects a 30-bit signed integer over
|
||||
/// the high 30 bits of an IEEE double; bit 0 divides by 100.
|
||||
fn decodeRk(raw: u32) f64 {
|
||||
const value: f64 = if (raw & 2 != 0)
|
||||
@floatFromInt(@as(i32, @bitCast(raw)) >> 2)
|
||||
else
|
||||
@bitCast(@as(u64, raw & 0xFFFFFFFC) << 32);
|
||||
return if (raw & 1 != 0) value / 100 else value;
|
||||
}
|
||||
|
||||
fn expectBof(rec: Record, dt: u16) Error!void {
|
||||
switch (rec.kind) {
|
||||
rt.bof => {},
|
||||
rt.bof_biff2, rt.bof_biff3, rt.bof_biff4 => return error.UnsupportedBiffVersion,
|
||||
else => return error.CorruptFile,
|
||||
}
|
||||
if (rec.data.len < 4) return error.CorruptFile;
|
||||
if (readInt(u16, rec.data, 0) != biff8_version) return error.UnsupportedBiffVersion;
|
||||
if (readInt(u16, rec.data, 2) != dt) return error.CorruptFile;
|
||||
}
|
||||
|
||||
const Record = struct {
|
||||
kind: u16,
|
||||
data: []const u8,
|
||||
};
|
||||
|
||||
const Records = struct {
|
||||
stream: []const u8,
|
||||
pos: usize = 0,
|
||||
|
||||
fn next(self: *Records) Error!?Record {
|
||||
if (self.pos == self.stream.len) return null;
|
||||
if (self.stream.len - self.pos < 4) return error.CorruptFile;
|
||||
const kind = readInt(u16, self.stream, self.pos);
|
||||
const len = readInt(u16, self.stream, self.pos + 2);
|
||||
const start = self.pos + 4;
|
||||
if (len > self.stream.len - start) return error.CorruptFile;
|
||||
self.pos = start + len;
|
||||
return .{ .kind = kind, .data = self.stream[start..][0..len] };
|
||||
}
|
||||
|
||||
fn peekKind(self: Records) ?u16 {
|
||||
if (self.stream.len - self.pos < 4) return null;
|
||||
return readInt(u16, self.stream, self.pos);
|
||||
}
|
||||
};
|
||||
|
||||
/// Reads across a record body and the CONTINUE bodies after it. Plain
|
||||
/// reads (`byte`, `int`, `skip`) step across a boundary transparently;
|
||||
/// `characters` consumes the high-byte flag a boundary inside character
|
||||
/// data introduces.
|
||||
const Chunks = struct {
|
||||
chunks: []const []const u8,
|
||||
index: usize = 0,
|
||||
pos: usize = 0,
|
||||
|
||||
fn current(self: Chunks) []const u8 {
|
||||
return self.chunks[self.index];
|
||||
}
|
||||
|
||||
fn nextChunk(self: *Chunks) Error!void {
|
||||
if (self.index + 1 >= self.chunks.len) return error.CorruptFile;
|
||||
self.index += 1;
|
||||
self.pos = 0;
|
||||
}
|
||||
|
||||
fn byte(self: *Chunks) Error!u8 {
|
||||
while (self.pos == self.current().len) try self.nextChunk();
|
||||
const b = self.current()[self.pos];
|
||||
self.pos += 1;
|
||||
return b;
|
||||
}
|
||||
|
||||
fn int(self: *Chunks, comptime T: type) Error!T {
|
||||
var v: T = 0;
|
||||
for (0..@sizeOf(T)) |i| v |= @as(T, try self.byte()) << @intCast(8 * i);
|
||||
return v;
|
||||
}
|
||||
|
||||
fn skip(self: *Chunks, n: usize) Error!void {
|
||||
var left = n;
|
||||
while (left > 0) {
|
||||
if (self.pos == self.current().len) try self.nextChunk();
|
||||
const k = @min(left, self.current().len - self.pos);
|
||||
self.pos += k;
|
||||
left -= k;
|
||||
}
|
||||
}
|
||||
|
||||
/// XLUnicodeString: `u16` length, flags, characters.
|
||||
fn unicodeString(self: *Chunks, arena: Allocator, scratch: Allocator) Error![]const u8 {
|
||||
const cch = try self.int(u16);
|
||||
const flags = try self.byte();
|
||||
return self.characters(arena, scratch, cch, flags & 1 != 0);
|
||||
}
|
||||
|
||||
/// `count` characters, 1-byte (Latin-1) or 2-byte (UTF-16LE) per
|
||||
/// `high`, re-reading the flag whenever the characters cross into
|
||||
/// the next chunk. Returns UTF-8 allocated from `arena`.
|
||||
fn characters(self: *Chunks, arena: Allocator, scratch: Allocator, count: usize, high_start: bool) Error![]const u8 {
|
||||
var units: std.ArrayList(u16) = .empty;
|
||||
defer units.deinit(scratch);
|
||||
try units.ensureTotalCapacity(scratch, count);
|
||||
|
||||
var high = high_start;
|
||||
var left = count;
|
||||
while (left > 0) {
|
||||
if (self.pos == self.current().len) {
|
||||
try self.nextChunk();
|
||||
if (self.current().len == 0) return error.CorruptFile;
|
||||
high = self.current()[0] & 1 != 0;
|
||||
self.pos = 1;
|
||||
}
|
||||
const avail = self.current()[self.pos..];
|
||||
if (high) {
|
||||
const k = @min(left, avail.len / 2);
|
||||
// A lone trailing byte cannot hold a UTF-16 unit.
|
||||
if (k == 0) return error.CorruptFile;
|
||||
for (0..k) |i| units.appendAssumeCapacity(readInt(u16, avail, 2 * i));
|
||||
self.pos += 2 * k;
|
||||
left -= k;
|
||||
} else {
|
||||
const k = @min(left, avail.len);
|
||||
for (avail[0..k]) |c| units.appendAssumeCapacity(c);
|
||||
self.pos += k;
|
||||
left -= k;
|
||||
}
|
||||
}
|
||||
return utf8FromUtf16(arena, units.items);
|
||||
}
|
||||
};
|
||||
|
||||
/// UTF-16 to UTF-8. Unpaired surrogates become U+FFFD rather than an
|
||||
/// error: a stray half in a cell should not make the workbook
|
||||
/// unreadable.
|
||||
fn utf8FromUtf16(arena: Allocator, units: []const u16) Error![]const u8 {
|
||||
var len: usize = 0;
|
||||
var i: usize = 0;
|
||||
while (i < units.len) len += utf8Len(nextCodepoint(units, &i));
|
||||
|
||||
const out = try arena.alloc(u8, len);
|
||||
var o: usize = 0;
|
||||
i = 0;
|
||||
while (i < units.len) o += encodeUtf8(nextCodepoint(units, &i), out[o..]);
|
||||
return out;
|
||||
}
|
||||
|
||||
fn nextCodepoint(units: []const u16, i: *usize) u21 {
|
||||
const u = units[i.*];
|
||||
i.* += 1;
|
||||
if (u >= 0xD800 and u <= 0xDBFF and i.* < units.len and units[i.*] >= 0xDC00 and units[i.*] <= 0xDFFF) {
|
||||
const lo = units[i.*];
|
||||
i.* += 1;
|
||||
return 0x10000 + ((@as(u21, u) - 0xD800) << 10) + (lo - 0xDC00);
|
||||
}
|
||||
if (u >= 0xD800 and u <= 0xDFFF) return 0xFFFD;
|
||||
return u;
|
||||
}
|
||||
|
||||
fn utf8Len(c: u21) usize {
|
||||
if (c < 0x80) return 1;
|
||||
if (c < 0x800) return 2;
|
||||
if (c < 0x10000) return 3;
|
||||
return 4;
|
||||
}
|
||||
|
||||
fn encodeUtf8(c: u21, out: []u8) usize {
|
||||
switch (utf8Len(c)) {
|
||||
1 => out[0] = @intCast(c),
|
||||
2 => {
|
||||
out[0] = @intCast(0xC0 | (c >> 6));
|
||||
out[1] = @intCast(0x80 | (c & 0x3F));
|
||||
},
|
||||
3 => {
|
||||
out[0] = @intCast(0xE0 | (c >> 12));
|
||||
out[1] = @intCast(0x80 | ((c >> 6) & 0x3F));
|
||||
out[2] = @intCast(0x80 | (c & 0x3F));
|
||||
},
|
||||
else => {
|
||||
out[0] = @intCast(0xF0 | (c >> 18));
|
||||
out[1] = @intCast(0x80 | ((c >> 12) & 0x3F));
|
||||
out[2] = @intCast(0x80 | ((c >> 6) & 0x3F));
|
||||
out[3] = @intCast(0x80 | (c & 0x3F));
|
||||
},
|
||||
}
|
||||
return utf8Len(c);
|
||||
}
|
||||
|
||||
fn readInt(comptime T: type, bytes: []const u8, off: usize) T {
|
||||
return std.mem.readInt(T, bytes[off..][0..@sizeOf(T)], .little);
|
||||
}
|
||||
|
||||
// ---- Tests ----
|
||||
|
||||
const testing = std.testing;
|
||||
const tw = @import("test_writer.zig");
|
||||
|
||||
/// Parse a workbook stream with a leak-checked scratch allocator and an
|
||||
/// arena for the results.
|
||||
const Parsed = struct {
|
||||
arena: std.heap.ArenaAllocator,
|
||||
sheets: []const Sheet,
|
||||
|
||||
fn init(stream: []const u8) Error!Parsed {
|
||||
var arena = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
errdefer arena.deinit();
|
||||
const sheets = try parseSheets(arena.allocator(), testing.allocator, stream);
|
||||
return .{ .arena = arena, .sheets = sheets };
|
||||
}
|
||||
|
||||
fn deinit(self: *Parsed) void {
|
||||
self.arena.deinit();
|
||||
}
|
||||
};
|
||||
|
||||
fn expectParseError(expected: Error, spec: tw.WorkbookSpec) !void {
|
||||
const stream = try tw.workbook(testing.allocator, spec);
|
||||
defer testing.allocator.free(stream);
|
||||
var arena = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer arena.deinit();
|
||||
try testing.expectError(expected, parseSheets(arena.allocator(), testing.allocator, stream));
|
||||
}
|
||||
|
||||
test "decodes the common cell records" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
|
||||
const sst = try tw.sst(a, &.{ .{ .text = "Description" }, .{ .text = "Sample Brokerage" } }, 8224);
|
||||
const stream = try tw.workbook(a, .{
|
||||
.globals = sst,
|
||||
.sheets = &.{.{
|
||||
.name = "Positions",
|
||||
.records = &.{
|
||||
try tw.labelSst(a, 0, 0, 0),
|
||||
try tw.labelSst(a, 1, 0, 1),
|
||||
try tw.number(a, 1, 1, 1234.5678),
|
||||
try tw.rk(a, 1, 2, (25 << 2) | 2), // integer 25
|
||||
try tw.boolErr(a, 2, 0, 1, false),
|
||||
try tw.boolErr(a, 2, 1, 0x07, true),
|
||||
try tw.label(a, tw.rt.label, 3, 0, "inline label"),
|
||||
try tw.label(a, tw.rt.rstring, 3, 1, "rich label"),
|
||||
try tw.blank(a, 4, 3),
|
||||
},
|
||||
}},
|
||||
});
|
||||
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
try testing.expectEqual(@as(usize, 1), p.sheets.len);
|
||||
const s = p.sheets[0];
|
||||
try testing.expectEqualStrings("Positions", s.name);
|
||||
try testing.expectEqualStrings("Description", s.cell(0, 0).asText().?);
|
||||
try testing.expectEqualStrings("Sample Brokerage", s.cell(1, 0).asText().?);
|
||||
try testing.expectEqual(@as(f64, 1234.5678), s.cell(1, 1).asNumber().?);
|
||||
try testing.expectEqual(@as(f64, 25), s.cell(1, 2).asNumber().?);
|
||||
try testing.expectEqual(Cell{ .boolean = true }, s.cell(2, 0));
|
||||
try testing.expectEqual(Cell{ .error_code = 0x07 }, s.cell(2, 1));
|
||||
try testing.expectEqualStrings("inline label", s.cell(3, 0).asText().?);
|
||||
try testing.expectEqualStrings("rich label", s.cell(3, 1).asText().?);
|
||||
// BLANK carries formatting only, so it adds no cell (or row).
|
||||
try testing.expectEqual(@as(usize, 4), s.rows.len);
|
||||
try testing.expectEqual(Cell.empty, s.cell(4, 3));
|
||||
// Out of range in either direction is empty, not a crash.
|
||||
try testing.expectEqual(Cell.empty, s.cell(0, 200));
|
||||
try testing.expectEqual(Cell.empty, s.cell(9999, 0));
|
||||
try testing.expect(s.cell(1, 1).asText() == null);
|
||||
try testing.expect(s.cell(0, 0).asNumber() == null);
|
||||
}
|
||||
|
||||
test "RK encodings" {
|
||||
// Integer, integer / 100, float (high 30 bits of a double), float / 100.
|
||||
try testing.expectEqual(@as(f64, 25), decodeRk((25 << 2) | 2));
|
||||
try testing.expectEqual(@as(f64, -25), decodeRk(@as(u32, @bitCast(@as(i32, -25) << 2)) | 2));
|
||||
try testing.expectEqual(@as(f64, 12.34), decodeRk((1234 << 2) | 3));
|
||||
const one_and_half: u64 = @bitCast(@as(f64, 1.5));
|
||||
const hi: u32 = @intCast(one_and_half >> 32);
|
||||
try testing.expectEqual(@as(f64, 1.5), decodeRk(hi));
|
||||
try testing.expectEqual(@as(f64, 0.015), decodeRk(hi | 1));
|
||||
}
|
||||
|
||||
test "MULRK fans out across columns" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{
|
||||
try tw.mulrk(a, 5, 2, &.{ (1 << 2) | 2, (2 << 2) | 2, (300 << 2) | 3 }),
|
||||
} }} });
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
const s = p.sheets[0];
|
||||
try testing.expectEqual(@as(f64, 1), s.cell(5, 2).asNumber().?);
|
||||
try testing.expectEqual(@as(f64, 2), s.cell(5, 3).asNumber().?);
|
||||
try testing.expectEqual(@as(f64, 3), s.cell(5, 4).asNumber().?);
|
||||
try testing.expectEqual(Cell.empty, s.cell(5, 1));
|
||||
}
|
||||
|
||||
test "MULRK whose last column disagrees with its length is corrupt" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
const rec = try tw.mulrk(a, 0, 0, &.{ 2, 6 });
|
||||
const bad = try a.dupe(u8, rec.data);
|
||||
std.mem.writeInt(u16, bad[bad.len - 2 ..][0..2], 7, .little);
|
||||
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{.{ .kind = tw.rt.mulrk, .data = bad }} }} });
|
||||
}
|
||||
|
||||
test "formula cached results" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
const stream = try tw.workbook(a, .{
|
||||
.sheets = &.{.{
|
||||
.name = "S",
|
||||
.records = &.{
|
||||
try tw.formula(a, 0, 0, tw.formulaNumber(42.5)),
|
||||
try tw.formula(a, 0, 1, tw.formulaSpecial(0, 0)),
|
||||
// Records such as SHRFMLA may sit between FORMULA and STRING.
|
||||
.{ .kind = 0x04BC, .data = "\x00\x00" },
|
||||
try tw.string(a, "computed text"),
|
||||
try tw.formula(a, 0, 2, tw.formulaSpecial(1, 1)),
|
||||
try tw.formula(a, 0, 3, tw.formulaSpecial(2, 0x2A)),
|
||||
try tw.formula(a, 0, 4, tw.formulaSpecial(3, 0)),
|
||||
// A STRING with no text formula pending is ignored.
|
||||
try tw.string(a, "orphan"),
|
||||
// A text formula whose STRING never arrives leaves the cell empty.
|
||||
try tw.formula(a, 0, 5, tw.formulaSpecial(0, 0)),
|
||||
try tw.number(a, 1, 0, 1),
|
||||
},
|
||||
}},
|
||||
});
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
const s = p.sheets[0];
|
||||
try testing.expectEqual(@as(f64, 42.5), s.cell(0, 0).asNumber().?);
|
||||
try testing.expectEqualStrings("computed text", s.cell(0, 1).asText().?);
|
||||
try testing.expectEqual(Cell{ .boolean = true }, s.cell(0, 2));
|
||||
try testing.expectEqual(Cell{ .error_code = 0x2A }, s.cell(0, 3));
|
||||
try testing.expectEqualStrings("", s.cell(0, 4).asText().?);
|
||||
try testing.expectEqual(Cell.empty, s.cell(0, 5));
|
||||
}
|
||||
|
||||
test "formula string result continued across a CONTINUE record" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
// STRING body: cch=6, compressed, "abc" | CONTINUE: flag 1 (UTF-16), "def".
|
||||
const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{
|
||||
try tw.formula(a, 0, 0, tw.formulaSpecial(0, 0)),
|
||||
.{ .kind = tw.rt.string, .data = "\x06\x00\x00abc" },
|
||||
.{ .kind = tw.rt.@"continue", .data = "\x01d\x00e\x00f\x00" },
|
||||
} }} });
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
try testing.expectEqualStrings("abcdef", p.sheets[0].cell(0, 0).asText().?);
|
||||
}
|
||||
|
||||
test "unknown formula result type is corrupt" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{
|
||||
try tw.formula(a, 0, 0, tw.formulaSpecial(9, 0)),
|
||||
} }} });
|
||||
}
|
||||
|
||||
test "shared strings split across CONTINUE records at every offset" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
|
||||
const strings = [_]tw.SstString{
|
||||
.{ .text = "Bank Deposit Sweep" },
|
||||
.{ .text = "caf\u{e9} au lait" }, // Latin-1 but not ASCII: still compressed
|
||||
.{ .text = "\u{201c}buy\u{201d} or \u{201c}sell\u{201d}" }, // needs UTF-16
|
||||
.{ .text = "rich text", .runs = 3 },
|
||||
.{ .text = "phonetic", .ext = "\x01\x02\x03\x04\x05\x06\x07" },
|
||||
.{ .text = "" },
|
||||
.{ .text = "emoji \u{1F600} end" }, // surrogate pair
|
||||
.{ .text = "\u{3042} then plain ascii after the switch" }, // UTF-16 start, compressed tail
|
||||
};
|
||||
|
||||
// Every chunk size from tiny (splits everywhere, including inside
|
||||
// headers) to large (no split) must decode identically.
|
||||
var max_chunk: usize = 3;
|
||||
while (max_chunk <= 200) : (max_chunk += 1) {
|
||||
const sst = try tw.sst(a, &strings, max_chunk);
|
||||
var records: std.ArrayList(tw.Record) = .empty;
|
||||
for (0..strings.len) |i| try records.append(a, try tw.labelSst(a, @intCast(i), 0, @intCast(i)));
|
||||
const stream = try tw.workbook(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} });
|
||||
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
for (strings, 0..) |s, i| {
|
||||
testing.expectEqualStrings(s.text, p.sheets[0].cell(i, 0).asText().?) catch |err| {
|
||||
std.debug.print("max_chunk={d} string={d}\n", .{ max_chunk, i });
|
||||
return err;
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
test "unpaired surrogates decode as U+FFFD" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
const sst = try tw.sst(a, &.{
|
||||
.{ .units = &.{ 'a', 0xD800, 'b' } },
|
||||
.{ .units = &.{ 'a', 0xDC00 } },
|
||||
.{ .units = &.{0xD83D} },
|
||||
}, 8224);
|
||||
const stream = try tw.workbook(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = &.{
|
||||
try tw.labelSst(a, 0, 0, 0),
|
||||
try tw.labelSst(a, 1, 0, 1),
|
||||
try tw.labelSst(a, 2, 0, 2),
|
||||
} }} });
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
try testing.expectEqualStrings("a\u{FFFD}b", p.sheets[0].cell(0, 0).asText().?);
|
||||
try testing.expectEqualStrings("a\u{FFFD}", p.sheets[0].cell(1, 0).asText().?);
|
||||
try testing.expectEqualStrings("\u{FFFD}", p.sheets[0].cell(2, 0).asText().?);
|
||||
}
|
||||
|
||||
test "UTF-8 encoding covers every sequence length" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const units = [_]u16{ 'A', 0x00E9, 0x20AC, 0xD83D, 0xDE00 };
|
||||
try testing.expectEqualStrings("A\u{e9}\u{20ac}\u{1F600}", try utf8FromUtf16(fx.allocator(), &units));
|
||||
}
|
||||
|
||||
test "multiple sheets keep workbook order; non-worksheets are skipped" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
const stream = try tw.workbook(a, .{ .sheets = &.{
|
||||
.{ .name = "First", .records = &.{try tw.number(a, 0, 0, 1)} },
|
||||
.{ .name = "Chart1", .boundsheet_type = 2, .bof_type = 0x0020 },
|
||||
.{ .name = "Second", .records = &.{try tw.number(a, 0, 0, 2)} },
|
||||
} });
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
try testing.expectEqual(@as(usize, 2), p.sheets.len);
|
||||
try testing.expectEqualStrings("First", p.sheets[0].name);
|
||||
try testing.expectEqualStrings("Second", p.sheets[1].name);
|
||||
try testing.expectEqual(@as(f64, 2), p.sheets[1].cell(0, 0).asNumber().?);
|
||||
}
|
||||
|
||||
test "embedded chart substreams inside a sheet are not its cells" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
const chart_bof = tw.bof(0x0600, 0x0020);
|
||||
const stream = try tw.workbook(a, .{
|
||||
.sheets = &.{.{
|
||||
.name = "S",
|
||||
.records = &.{
|
||||
try tw.number(a, 0, 0, 1),
|
||||
.{ .kind = tw.rt.bof, .data = &chart_bof },
|
||||
try tw.number(a, 0, 1, 99), // belongs to the chart
|
||||
.{ .kind = tw.rt.eof, .data = "" },
|
||||
try tw.number(a, 0, 2, 3),
|
||||
},
|
||||
}},
|
||||
});
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
const s = p.sheets[0];
|
||||
try testing.expectEqual(@as(f64, 1), s.cell(0, 0).asNumber().?);
|
||||
try testing.expectEqual(Cell.empty, s.cell(0, 1));
|
||||
try testing.expectEqual(@as(f64, 3), s.cell(0, 2).asNumber().?);
|
||||
}
|
||||
|
||||
test "a repeated cell keeps the last value; sparse rows stay empty" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
const stream = try tw.workbook(a, .{ .sheets = &.{.{ .name = "S", .records = &.{
|
||||
try tw.number(a, 3, 1, 1),
|
||||
try tw.number(a, 3, 1, 2),
|
||||
try tw.number(a, 0, 4, 7),
|
||||
} }} });
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
const s = p.sheets[0];
|
||||
try testing.expectEqual(@as(usize, 4), s.rows.len);
|
||||
try testing.expectEqual(@as(f64, 2), s.cell(3, 1).asNumber().?);
|
||||
try testing.expectEqual(@as(usize, 0), s.rows[1].len);
|
||||
try testing.expectEqual(@as(usize, 5), s.rows[0].len);
|
||||
}
|
||||
|
||||
test "a sheet with no cells has no rows" {
|
||||
const stream = try tw.workbook(testing.allocator, .{ .sheets = &.{.{ .name = "Empty" }} });
|
||||
defer testing.allocator.free(stream);
|
||||
var p = try Parsed.init(stream);
|
||||
defer p.deinit();
|
||||
try testing.expectEqual(@as(usize, 0), p.sheets[0].rows.len);
|
||||
}
|
||||
|
||||
test "older BIFF versions are unsupported" {
|
||||
try expectParseError(error.UnsupportedBiffVersion, .{ .version = 0x0500 });
|
||||
// BIFF2-4 use different BOF record numbers entirely.
|
||||
const biff4_bof = "\x09\x04\x06\x00\x00\x00\x10\x00\x00\x00";
|
||||
var arena = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer arena.deinit();
|
||||
try testing.expectError(error.UnsupportedBiffVersion, parseSheets(arena.allocator(), testing.allocator, biff4_bof));
|
||||
}
|
||||
|
||||
test "password-protected workbooks report Encrypted" {
|
||||
try expectParseError(error.Encrypted, .{ .globals = &.{.{ .kind = tw.rt.filepass, .data = "\x01\x00" }} });
|
||||
}
|
||||
|
||||
test "structural corruption is reported" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
|
||||
// LABELSST pointing past the shared string table.
|
||||
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{try tw.labelSst(a, 0, 0, 5)} }} });
|
||||
// BOUNDSHEET offset past the end of the stream.
|
||||
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S" }}, .offset_override = 0xFFFFFF });
|
||||
// BOUNDSHEET offset that lands on something other than a BOF.
|
||||
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S" }}, .offset_override = 0 });
|
||||
// Short record bodies.
|
||||
const short = [_]u16{ tw.rt.labelsst, tw.rt.number, tw.rt.rk, tw.rt.mulrk, tw.rt.label, tw.rt.boolerr, tw.rt.formula };
|
||||
for (short) |kind| {
|
||||
try expectParseError(error.CorruptFile, .{ .sheets = &.{.{ .name = "S", .records = &.{.{ .kind = kind, .data = "\x00\x00" }} }} });
|
||||
}
|
||||
// An SST claiming more strings than its bytes could hold.
|
||||
try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.sst, .data = "\x00\x00\x00\x00\xFF\xFF\x00\x00" }} });
|
||||
// An SST whose last string runs off the end of its records.
|
||||
try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x05\x00\x00ab" }} });
|
||||
// A UTF-16 string that leaves one odd byte at a record boundary.
|
||||
try expectParseError(error.CorruptFile, .{ .globals = &.{
|
||||
.{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x02\x00\x01a" },
|
||||
.{ .kind = tw.rt.@"continue", .data = "\x01b\x00" },
|
||||
} });
|
||||
// A CONTINUE with no room for the high-byte flag.
|
||||
try expectParseError(error.CorruptFile, .{ .globals = &.{
|
||||
.{ .kind = tw.rt.sst, .data = "\x01\x00\x00\x00\x01\x00\x00\x00\x02\x00\x00a" },
|
||||
.{ .kind = tw.rt.@"continue", .data = "" },
|
||||
} });
|
||||
// A BOUNDSHEET too short to hold its header.
|
||||
try expectParseError(error.CorruptFile, .{ .globals = &.{.{ .kind = tw.rt.boundsheet, .data = "\x00\x00" }} });
|
||||
}
|
||||
|
||||
test "a substream without its EOF is truncated" {
|
||||
try expectParseError(error.Truncated, .{ .sheets = &.{.{ .name = "S" }}, .omit_last_eof = true });
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer arena.deinit();
|
||||
// Globals that never reach EOF, and an empty stream.
|
||||
const globals_bof = comptime tw.bof(0x0600, 0x0005);
|
||||
const no_eof = "\x09\x08\x10\x00" ++ globals_bof;
|
||||
try testing.expectError(error.Truncated, parseSheets(arena.allocator(), testing.allocator, no_eof));
|
||||
try testing.expectError(error.Truncated, parseSheets(arena.allocator(), testing.allocator, ""));
|
||||
}
|
||||
|
||||
test "record framing errors" {
|
||||
var arena = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer arena.deinit();
|
||||
// A record header cut short, and a length past the end.
|
||||
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08"));
|
||||
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\xFF\x00"));
|
||||
// First record is not a BOF at all.
|
||||
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x0A\x00\x00\x00"));
|
||||
// A BOF too short to carry a version.
|
||||
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\x02\x00\x00\x06"));
|
||||
// A BOF whose substream type is wrong for its position.
|
||||
const ws_bof = comptime tw.bof(0x0600, 0x0010);
|
||||
try testing.expectError(error.CorruptFile, parseSheets(arena.allocator(), testing.allocator, "\x09\x08\x10\x00" ++ ws_bof));
|
||||
}
|
||||
671
src/cfb.zig
Normal file
671
src/cfb.zig
Normal file
|
|
@ -0,0 +1,671 @@
|
|||
//! Read-only reader for the Compound File Binary format (MS-CFB), also
|
||||
//! known as "OLE2 structured storage". It is a FAT-style filesystem
|
||||
//! inside one file, and it is the container a legacy `.xls` workbook
|
||||
//! lives in: the spreadsheet itself is the `Workbook` stream inside it.
|
||||
//!
|
||||
//! Only what reading a stream needs is implemented:
|
||||
//!
|
||||
//! - header validation (signature, byte order, sector sizes for
|
||||
//! major versions 3 and 4)
|
||||
//! - the DIFAT -> FAT sector-allocation table, including DIFAT
|
||||
//! sectors beyond the 109 entries the header holds
|
||||
//! - the directory, scanned linearly (the red-black tree is ignored;
|
||||
//! a name lookup over a few dozen entries does not need it)
|
||||
//! - regular streams (FAT chains) and small streams (mini FAT chains
|
||||
//! inside the root entry's mini stream)
|
||||
//!
|
||||
//! Every chain walk is bounded by the size of the table it walks, so a
|
||||
//! cyclic chain is reported as `CorruptFile` instead of looping, and a
|
||||
//! declared size larger than the file is rejected before allocating.
|
||||
//!
|
||||
//! Spec: [MS-CFB] Compound File Binary File Format.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub const Error = error{
|
||||
/// The bytes do not start with the compound-file signature.
|
||||
NotCompoundFile,
|
||||
/// The structure is internally inconsistent: bad header fields,
|
||||
/// a chain that loops or points outside its table, an entry larger
|
||||
/// than the file can hold.
|
||||
CorruptFile,
|
||||
/// The file ends before a sector it references. Usually an
|
||||
/// interrupted download.
|
||||
Truncated,
|
||||
OutOfMemory,
|
||||
};
|
||||
|
||||
/// First eight bytes of every compound file.
|
||||
pub const signature = [8]u8{ 0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1 };
|
||||
|
||||
/// Special sector numbers (MS-CFB 2.1). Any value above `max_reg_sect`
|
||||
/// is not a real sector.
|
||||
const max_reg_sect: u32 = 0xFFFFFFFA;
|
||||
const end_of_chain: u32 = 0xFFFFFFFE;
|
||||
const free_sect: u32 = 0xFFFFFFFF;
|
||||
|
||||
const header_size = 512;
|
||||
const dir_entry_size = 128;
|
||||
const mini_sector_size = 64;
|
||||
/// DIFAT entries stored in the header itself.
|
||||
const header_difat_count = 109;
|
||||
|
||||
/// True when `bytes` starts with the compound-file signature. Cheap
|
||||
/// enough for content sniffing; does not validate anything else.
|
||||
pub fn isCompoundFile(bytes: []const u8) bool {
|
||||
return bytes.len >= signature.len and std.mem.eql(u8, bytes[0..signature.len], &signature);
|
||||
}
|
||||
|
||||
pub const EntryType = enum(u8) {
|
||||
unknown = 0,
|
||||
storage = 1,
|
||||
stream = 2,
|
||||
root = 5,
|
||||
_,
|
||||
};
|
||||
|
||||
/// One directory entry. Only the fields a reader needs.
|
||||
pub const Entry = struct {
|
||||
/// UTF-16 code units of the name, without the terminating NUL.
|
||||
name: [31]u16,
|
||||
name_len: u8,
|
||||
kind: EntryType,
|
||||
start_sector: u32,
|
||||
size: u64,
|
||||
|
||||
/// Case-insensitive comparison against an ASCII name. CFB names
|
||||
/// compare case-insensitively (MS-CFB 2.6.4), and every stream name
|
||||
/// a reader looks up by literal is ASCII.
|
||||
pub fn nameEql(self: Entry, ascii: []const u8) bool {
|
||||
if (ascii.len != self.name_len) return false;
|
||||
for (self.name[0..self.name_len], ascii) |unit, c| {
|
||||
if (unit > 0x7F) return false;
|
||||
if (std.ascii.toLower(@intCast(unit)) != std.ascii.toLower(c)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
pub const File = struct {
|
||||
allocator: std.mem.Allocator,
|
||||
bytes: []const u8,
|
||||
sector_shift: u4,
|
||||
mini_cutoff: u32,
|
||||
fat: []u32,
|
||||
mini_fat: []u32,
|
||||
entries: []Entry,
|
||||
/// Contents of the root entry's stream, which holds every small
|
||||
/// stream. Empty when the file has no small streams.
|
||||
mini_stream: []u8,
|
||||
|
||||
/// Parse the header, allocation tables and directory. `bytes` is
|
||||
/// borrowed and must outlive the `File`.
|
||||
pub fn init(allocator: std.mem.Allocator, bytes: []const u8) Error!File {
|
||||
if (!isCompoundFile(bytes)) return error.NotCompoundFile;
|
||||
if (bytes.len < header_size) return error.Truncated;
|
||||
|
||||
if (readU16(bytes, 28) != 0xFFFE) return error.CorruptFile; // byte order mark
|
||||
const major = readU16(bytes, 26);
|
||||
const sector_shift: u4 = switch (major) {
|
||||
3 => 9,
|
||||
4 => 12,
|
||||
else => return error.CorruptFile,
|
||||
};
|
||||
if (readU16(bytes, 30) != sector_shift) return error.CorruptFile;
|
||||
if (readU16(bytes, 32) != 6) return error.CorruptFile; // 64-byte mini sectors
|
||||
const sector_size = @as(usize, 1) << sector_shift;
|
||||
// Version 4 pads the header out to a full 4096-byte sector.
|
||||
if (bytes.len < sector_size) return error.Truncated;
|
||||
|
||||
const num_fat = readU32(bytes, 44);
|
||||
const first_dir = readU32(bytes, 48);
|
||||
const mini_cutoff = readU32(bytes, 56);
|
||||
const first_mini_fat = readU32(bytes, 60);
|
||||
const first_difat = readU32(bytes, 68);
|
||||
|
||||
// Upper bound on sectors this file can address. Every table
|
||||
// size below is checked against it before allocating, so a
|
||||
// hostile header cannot request a huge allocation.
|
||||
const total_sectors = (bytes.len - sector_size + sector_size - 1) >> sector_shift;
|
||||
if (num_fat > total_sectors) return error.CorruptFile;
|
||||
|
||||
var self: File = .{
|
||||
.allocator = allocator,
|
||||
.bytes = bytes,
|
||||
.sector_shift = sector_shift,
|
||||
.mini_cutoff = mini_cutoff,
|
||||
.fat = &.{},
|
||||
.mini_fat = &.{},
|
||||
.entries = &.{},
|
||||
.mini_stream = &.{},
|
||||
};
|
||||
errdefer self.deinit();
|
||||
|
||||
self.fat = try self.readFat(num_fat, first_difat, total_sectors);
|
||||
self.entries = try self.readDirectory(first_dir);
|
||||
if (self.entries.len == 0 or self.entries[0].kind != .root) return error.CorruptFile;
|
||||
self.mini_fat = try self.readMiniFat(first_mini_fat);
|
||||
|
||||
const root = self.entries[0];
|
||||
if (root.size > 0) {
|
||||
self.mini_stream = try self.readRegular(allocator, root.start_sector, root.size);
|
||||
}
|
||||
return self;
|
||||
}
|
||||
|
||||
pub fn deinit(self: *File) void {
|
||||
self.allocator.free(self.fat);
|
||||
self.allocator.free(self.mini_fat);
|
||||
self.allocator.free(self.entries);
|
||||
self.allocator.free(self.mini_stream);
|
||||
}
|
||||
|
||||
/// First stream entry whose name matches `ascii` case-insensitively.
|
||||
pub fn find(self: File, ascii: []const u8) ?Entry {
|
||||
for (self.entries) |e| {
|
||||
if (e.kind == .stream and e.nameEql(ascii)) return e;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Read a stream's full contents. Caller owns the returned bytes.
|
||||
pub fn readStream(self: File, allocator: std.mem.Allocator, entry: Entry) Error![]u8 {
|
||||
if (entry.size < self.mini_cutoff) return self.readMini(allocator, entry.start_sector, entry.size);
|
||||
return self.readRegular(allocator, entry.start_sector, entry.size);
|
||||
}
|
||||
|
||||
fn sectorSize(self: File) usize {
|
||||
return @as(usize, 1) << self.sector_shift;
|
||||
}
|
||||
|
||||
/// Bytes of sector `index`. The final sector of a file may be short
|
||||
/// (some writers do not pad it), so the slice can be shorter than a
|
||||
/// sector; callers that need a whole sector use `fullSector`.
|
||||
fn sector(self: File, index: u32) Error![]const u8 {
|
||||
if (index > max_reg_sect) return error.CorruptFile;
|
||||
const start = (@as(usize, index) + 1) << self.sector_shift;
|
||||
if (start >= self.bytes.len) return error.Truncated;
|
||||
const end = @min(start + self.sectorSize(), self.bytes.len);
|
||||
return self.bytes[start..end];
|
||||
}
|
||||
|
||||
fn fullSector(self: File, index: u32) Error![]const u8 {
|
||||
const s = try self.sector(index);
|
||||
if (s.len != self.sectorSize()) return error.Truncated;
|
||||
return s;
|
||||
}
|
||||
|
||||
/// Collect the FAT sector numbers (header DIFAT, then the DIFAT
|
||||
/// sector chain) and concatenate those sectors into one table.
|
||||
fn readFat(self: File, num_fat: u32, first_difat: u32, total_sectors: usize) Error![]u32 {
|
||||
const sector_size = self.sectorSize();
|
||||
const per_sector = sector_size / 4;
|
||||
|
||||
const fat_sectors = try self.allocator.alloc(u32, num_fat);
|
||||
defer self.allocator.free(fat_sectors);
|
||||
|
||||
const in_header = @min(num_fat, header_difat_count);
|
||||
for (0..in_header) |i| fat_sectors[i] = readU32(self.bytes, 76 + 4 * i);
|
||||
|
||||
var filled: usize = in_header;
|
||||
var difat = first_difat;
|
||||
var steps: usize = 0;
|
||||
while (filled < num_fat) {
|
||||
steps += 1;
|
||||
if (difat > max_reg_sect or steps > total_sectors) return error.CorruptFile;
|
||||
const s = try self.fullSector(difat);
|
||||
// The last entry of a DIFAT sector links to the next one.
|
||||
const take = @min(num_fat - filled, per_sector - 1);
|
||||
for (0..take) |i| fat_sectors[filled + i] = readU32(s, 4 * i);
|
||||
filled += take;
|
||||
difat = readU32(s, sector_size - 4);
|
||||
}
|
||||
|
||||
const fat = try self.allocator.alloc(u32, @as(usize, num_fat) * per_sector);
|
||||
errdefer self.allocator.free(fat);
|
||||
for (fat_sectors, 0..) |fs, n| {
|
||||
const s = try self.fullSector(fs);
|
||||
for (0..per_sector) |i| fat[n * per_sector + i] = readU32(s, 4 * i);
|
||||
}
|
||||
return fat;
|
||||
}
|
||||
|
||||
fn readDirectory(self: File, first_dir: u32) Error![]Entry {
|
||||
var entries: std.ArrayList(Entry) = .empty;
|
||||
errdefer entries.deinit(self.allocator);
|
||||
|
||||
var it: ChainIterator = .{ .table = self.fat, .next_index = first_dir };
|
||||
while (try it.next()) |index| {
|
||||
const s = try self.fullSector(index);
|
||||
var off: usize = 0;
|
||||
while (off + dir_entry_size <= s.len) : (off += dir_entry_size) {
|
||||
try entries.append(self.allocator, try parseEntry(s[off..][0..dir_entry_size], self.sector_shift == 9));
|
||||
}
|
||||
}
|
||||
return entries.toOwnedSlice(self.allocator);
|
||||
}
|
||||
|
||||
fn readMiniFat(self: File, first_mini_fat: u32) Error![]u32 {
|
||||
var table: std.ArrayList(u32) = .empty;
|
||||
errdefer table.deinit(self.allocator);
|
||||
|
||||
if (first_mini_fat == end_of_chain or first_mini_fat == free_sect) return table.toOwnedSlice(self.allocator);
|
||||
var it: ChainIterator = .{ .table = self.fat, .next_index = first_mini_fat };
|
||||
while (try it.next()) |index| {
|
||||
const s = try self.fullSector(index);
|
||||
var off: usize = 0;
|
||||
while (off + 4 <= s.len) : (off += 4) try table.append(self.allocator, readU32(s, off));
|
||||
}
|
||||
return table.toOwnedSlice(self.allocator);
|
||||
}
|
||||
|
||||
/// Read `size` bytes following the FAT chain from `start`.
|
||||
fn readRegular(self: File, allocator: std.mem.Allocator, start: u32, size: u64) Error![]u8 {
|
||||
// Beyond what the FAT can address is impossible; beyond the end
|
||||
// of the bytes we have means the file was cut short.
|
||||
if (size > @as(u64, self.fat.len) << self.sector_shift) return error.CorruptFile;
|
||||
if (size > self.bytes.len) return error.Truncated;
|
||||
const out = try allocator.alloc(u8, @intCast(size));
|
||||
errdefer allocator.free(out);
|
||||
|
||||
var filled: usize = 0;
|
||||
var it: ChainIterator = .{ .table = self.fat, .next_index = start };
|
||||
while (filled < out.len) {
|
||||
const index = (try it.next()) orelse return error.CorruptFile; // chain shorter than size
|
||||
const s = try self.sector(index);
|
||||
const n = @min(s.len, out.len - filled);
|
||||
// A short final sector is only acceptable when it holds the
|
||||
// rest of the stream.
|
||||
if (n < self.sectorSize() and filled + n < out.len) return error.Truncated;
|
||||
@memcpy(out[filled..][0..n], s[0..n]);
|
||||
filled += n;
|
||||
}
|
||||
try it.finish();
|
||||
return out;
|
||||
}
|
||||
|
||||
/// Read `size` bytes following the mini FAT chain from `start`
|
||||
/// inside the mini stream.
|
||||
fn readMini(self: File, allocator: std.mem.Allocator, start: u32, size: u64) Error![]u8 {
|
||||
if (size > self.mini_stream.len) return error.CorruptFile;
|
||||
const out = try allocator.alloc(u8, @intCast(size));
|
||||
errdefer allocator.free(out);
|
||||
|
||||
var filled: usize = 0;
|
||||
var it: ChainIterator = .{ .table = self.mini_fat, .next_index = start };
|
||||
while (filled < out.len) {
|
||||
const index = (try it.next()) orelse return error.CorruptFile;
|
||||
const off = @as(usize, index) * mini_sector_size;
|
||||
if (off >= self.mini_stream.len) return error.CorruptFile;
|
||||
const n = @min(mini_sector_size, out.len - filled, self.mini_stream.len - off);
|
||||
// The mini stream ran out before the stream did.
|
||||
if (n < mini_sector_size and filled + n < out.len) return error.CorruptFile;
|
||||
@memcpy(out[filled..][0..n], self.mini_stream[off..][0..n]);
|
||||
filled += n;
|
||||
}
|
||||
try it.finish();
|
||||
return out;
|
||||
}
|
||||
};
|
||||
|
||||
/// Walks a sector chain through a FAT or mini FAT. Each step must land
|
||||
/// inside the table, and a chain can visit at most `table.len` sectors,
|
||||
/// so loops and dangling links are reported instead of followed.
|
||||
const ChainIterator = struct {
|
||||
table: []const u32,
|
||||
next_index: u32,
|
||||
steps: usize = 0,
|
||||
|
||||
fn next(it: *ChainIterator) Error!?u32 {
|
||||
if (it.next_index == end_of_chain) return null;
|
||||
if (it.next_index >= it.table.len) return error.CorruptFile;
|
||||
it.steps += 1;
|
||||
if (it.steps > it.table.len) return error.CorruptFile;
|
||||
const current = it.next_index;
|
||||
it.next_index = it.table[current];
|
||||
return current;
|
||||
}
|
||||
|
||||
/// Walk whatever is left of the chain. A reader stops once it has
|
||||
/// a stream's declared size, so a loop late in the chain would
|
||||
/// otherwise go unnoticed and its repeated sectors would be
|
||||
/// returned as data. Extra sectors that do end are tolerated.
|
||||
fn finish(it: *ChainIterator) Error!void {
|
||||
while (try it.next()) |_| {}
|
||||
}
|
||||
};
|
||||
|
||||
fn parseEntry(raw: *const [dir_entry_size]u8, is_v3: bool) Error!Entry {
|
||||
// Name length is in bytes and includes the UTF-16 NUL terminator.
|
||||
const name_bytes = readU16(raw, 64);
|
||||
if (name_bytes > 64 or name_bytes % 2 != 0) return error.CorruptFile;
|
||||
const units: u8 = if (name_bytes == 0) 0 else @intCast(name_bytes / 2 - 1);
|
||||
|
||||
var e: Entry = .{
|
||||
// SAFETY: the first `units` code units are written by the loop
|
||||
// below, and nothing reads past `name_len`.
|
||||
.name = undefined,
|
||||
.name_len = units,
|
||||
.kind = @enumFromInt(raw[66]),
|
||||
.start_sector = readU32(raw, 116),
|
||||
.size = std.mem.readInt(u64, raw[120..128], .little),
|
||||
};
|
||||
// Version 3 files only define the low 32 bits of the size; writers
|
||||
// are allowed to leave junk in the high half (MS-CFB 2.6.3).
|
||||
if (is_v3) e.size &= 0xFFFFFFFF;
|
||||
for (0..units) |i| e.name[i] = readU16(raw, 2 * i);
|
||||
return e;
|
||||
}
|
||||
|
||||
fn readU16(bytes: []const u8, off: usize) u16 {
|
||||
return std.mem.readInt(u16, bytes[off..][0..2], .little);
|
||||
}
|
||||
|
||||
fn readU32(bytes: []const u8, off: usize) u32 {
|
||||
return std.mem.readInt(u32, bytes[off..][0..4], .little);
|
||||
}
|
||||
|
||||
// ---- Tests ----
|
||||
|
||||
const testing = std.testing;
|
||||
const test_writer = @import("test_writer.zig");
|
||||
|
||||
test "isCompoundFile" {
|
||||
try testing.expect(isCompoundFile(&signature));
|
||||
try testing.expect(!isCompoundFile(signature[0..7]));
|
||||
try testing.expect(!isCompoundFile("PK\x03\x04 a zip file, i.e. xlsx"));
|
||||
}
|
||||
|
||||
test "reads a regular stream through the FAT" {
|
||||
const allocator = testing.allocator;
|
||||
const big = try allocator.alloc(u8, 10_000);
|
||||
defer allocator.free(big);
|
||||
for (big, 0..) |*b, i| b.* = @truncate(i *% 7);
|
||||
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
const entry = file.find("Workbook").?;
|
||||
const got = try file.readStream(allocator, entry);
|
||||
defer allocator.free(got);
|
||||
try testing.expectEqualSlices(u8, big, got);
|
||||
}
|
||||
|
||||
test "reads small streams through the mini FAT" {
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{
|
||||
.{ .name = "First", .data = "a small stream, well under the 4096-byte cutoff" },
|
||||
.{ .name = "Second", .data = "x" ** 130 }, // spans three mini sectors
|
||||
}, .{});
|
||||
defer allocator.free(bytes);
|
||||
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
|
||||
const first = try file.readStream(allocator, file.find("First").?);
|
||||
defer allocator.free(first);
|
||||
try testing.expectEqualStrings("a small stream, well under the 4096-byte cutoff", first);
|
||||
|
||||
const second = try file.readStream(allocator, file.find("Second").?);
|
||||
defer allocator.free(second);
|
||||
try testing.expectEqualStrings("x" ** 130, second);
|
||||
}
|
||||
|
||||
test "version 4 files use 4096-byte sectors" {
|
||||
const allocator = testing.allocator;
|
||||
const big = try allocator.alloc(u8, 9_000);
|
||||
defer allocator.free(big);
|
||||
@memset(big, 0x5A);
|
||||
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{
|
||||
.{ .name = "Workbook", .data = big },
|
||||
.{ .name = "Tiny", .data = "tiny" },
|
||||
}, .{ .major_version = 4 });
|
||||
defer allocator.free(bytes);
|
||||
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
const got = try file.readStream(allocator, file.find("Workbook").?);
|
||||
defer allocator.free(got);
|
||||
try testing.expectEqualSlices(u8, big, got);
|
||||
const tiny = try file.readStream(allocator, file.find("tiny").?);
|
||||
defer allocator.free(tiny);
|
||||
try testing.expectEqualStrings("tiny", tiny);
|
||||
}
|
||||
|
||||
test "FAT sectors beyond the header's 109 are found through DIFAT sectors" {
|
||||
// 109 FAT sectors address 109 * 128 sectors (~7 MiB). A stream
|
||||
// larger than that forces the writer to spill into a DIFAT sector.
|
||||
const allocator = testing.allocator;
|
||||
const big = try allocator.alloc(u8, 8 * 1024 * 1024);
|
||||
defer allocator.free(big);
|
||||
for (big, 0..) |*b, i| b.* = @truncate(i >> 9);
|
||||
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
try testing.expect(readU32(bytes, 72) > 0); // the writer really did use DIFAT sectors
|
||||
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
const got = try file.readStream(allocator, file.find("Workbook").?);
|
||||
defer allocator.free(got);
|
||||
try testing.expectEqualSlices(u8, big, got);
|
||||
}
|
||||
|
||||
test "stream lookup is case-insensitive and type-aware" {
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = "data" }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
|
||||
try testing.expect(file.find("WORKBOOK") != null);
|
||||
try testing.expect(file.find("workbook") != null);
|
||||
try testing.expect(file.find("Workboo") == null);
|
||||
// The root entry is not a stream, so its name never matches.
|
||||
try testing.expect(file.find("Root Entry") == null);
|
||||
}
|
||||
|
||||
test "Entry.nameEql rejects non-ASCII code units" {
|
||||
var e: Entry = .{ .name = @splat(0), .name_len = 1, .kind = .stream, .start_sector = 0, .size = 0 };
|
||||
e.name[0] = 0x00E9; // e-acute
|
||||
try testing.expect(!e.nameEql("e"));
|
||||
}
|
||||
|
||||
test "rejects a non-compound file" {
|
||||
try testing.expectError(error.NotCompoundFile, File.init(testing.allocator, "not an ole file at all"));
|
||||
}
|
||||
|
||||
test "rejects a header shorter than 512 bytes" {
|
||||
try testing.expectError(error.Truncated, File.init(testing.allocator, &signature));
|
||||
}
|
||||
|
||||
test "rejects bad header fields" {
|
||||
const allocator = testing.allocator;
|
||||
const good = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
|
||||
defer allocator.free(good);
|
||||
|
||||
const Patch = struct { off: usize, value: u16 };
|
||||
const patches = [_]Patch{
|
||||
.{ .off = 28, .value = 0xFEFF }, // byte order
|
||||
.{ .off = 26, .value = 5 }, // major version
|
||||
.{ .off = 30, .value = 12 }, // sector shift does not match version 3
|
||||
.{ .off = 32, .value = 7 }, // mini sector shift
|
||||
};
|
||||
for (patches) |p| {
|
||||
const bad = try allocator.dupe(u8, good);
|
||||
defer allocator.free(bad);
|
||||
std.mem.writeInt(u16, bad[p.off..][0..2], p.value, .little);
|
||||
try testing.expectError(error.CorruptFile, File.init(allocator, bad));
|
||||
}
|
||||
}
|
||||
|
||||
test "rejects a FAT sector count the file cannot hold" {
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
std.mem.writeInt(u32, bytes[44..48], 0xFFFFFF, .little);
|
||||
try testing.expectError(error.CorruptFile, File.init(allocator, bytes));
|
||||
}
|
||||
|
||||
test "rejects a version 4 file cut off inside its header sector" {
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{ .major_version = 4 });
|
||||
defer allocator.free(bytes);
|
||||
try testing.expectError(error.Truncated, File.init(allocator, bytes[0..1000]));
|
||||
}
|
||||
|
||||
test "a FAT sector listed past the end of the file is truncation" {
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
std.mem.writeInt(u32, bytes[76..80], 5000, .little); // header DIFAT[0]
|
||||
try testing.expectError(error.Truncated, File.init(allocator, bytes));
|
||||
}
|
||||
|
||||
test "a cyclic FAT chain is reported, not followed" {
|
||||
const allocator = testing.allocator;
|
||||
const big = try allocator.alloc(u8, 5000);
|
||||
defer allocator.free(big);
|
||||
@memset(big, 1);
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
const entry = file.find("Workbook").?;
|
||||
// Point the stream's first sector back at itself.
|
||||
file.fat[entry.start_sector] = entry.start_sector;
|
||||
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
||||
}
|
||||
|
||||
test "a chain shorter than the declared size is corrupt" {
|
||||
const allocator = testing.allocator;
|
||||
const big = try allocator.alloc(u8, 5000);
|
||||
defer allocator.free(big);
|
||||
@memset(big, 1);
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
const entry = file.find("Workbook").?;
|
||||
file.fat[entry.start_sector] = end_of_chain;
|
||||
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
||||
|
||||
var mini = entry;
|
||||
mini.size = 10; // below the cutoff, so read through the (empty) mini stream
|
||||
try testing.expectError(error.CorruptFile, file.readStream(allocator, mini));
|
||||
}
|
||||
|
||||
test "a stream larger than the file is rejected before allocating" {
|
||||
const allocator = testing.allocator;
|
||||
const big = try allocator.alloc(u8, 5000);
|
||||
defer allocator.free(big);
|
||||
@memset(big, 1);
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
var entry = file.find("Workbook").?;
|
||||
entry.size = 1 << 40;
|
||||
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
||||
}
|
||||
|
||||
test "a mini chain pointing outside the mini stream is corrupt" {
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "x" ** 100 }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
const entry = file.find("S").?;
|
||||
// The mini FAT has a full sector of entries (128) but the mini
|
||||
// stream only holds two mini sectors, so entry 100 is in the table
|
||||
// yet outside the stream.
|
||||
file.mini_fat[entry.start_sector] = 100;
|
||||
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
||||
}
|
||||
|
||||
test "a mini stream that ends mid-chain is corrupt" {
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "x" ** 100 }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
var entry = file.find("S").?;
|
||||
// Start the chain at the second mini sector and pretend the mini
|
||||
// stream ends 36 bytes into it: the first hop yields a short read
|
||||
// that cannot be the end of a 100-byte stream.
|
||||
entry.start_sector = 1;
|
||||
const full_len = file.mini_stream.len;
|
||||
file.mini_stream.len = 100;
|
||||
defer file.mini_stream.len = full_len;
|
||||
try testing.expectError(error.CorruptFile, file.readStream(allocator, entry));
|
||||
}
|
||||
|
||||
test "a file cut short mid-stream reports Truncated" {
|
||||
const allocator = testing.allocator;
|
||||
const big = try allocator.alloc(u8, 20_000);
|
||||
defer allocator.free(big);
|
||||
@memset(big, 3);
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
|
||||
// The writer places the stream last, so dropping the tail cuts it.
|
||||
var file = try File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
const cut_at = bytes.len - 4096;
|
||||
file.bytes = bytes[0..cut_at];
|
||||
try testing.expectError(error.Truncated, file.readStream(allocator, file.find("Workbook").?));
|
||||
|
||||
// A cut that lands inside a sector leaves a short sector that is not
|
||||
// the stream's last: also Truncated.
|
||||
file.bytes = bytes[0 .. cut_at + 100];
|
||||
try testing.expectError(error.Truncated, file.readStream(allocator, file.find("Workbook").?));
|
||||
}
|
||||
|
||||
test "an unpadded final sector is accepted when it ends the stream" {
|
||||
const allocator = testing.allocator;
|
||||
const big = try allocator.alloc(u8, 5000);
|
||||
defer allocator.free(big);
|
||||
@memset(big, 9);
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "Workbook", .data = big }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
|
||||
// 5000 bytes = 9 full sectors + 392 bytes; drop the padding.
|
||||
const unpadded = bytes[0 .. bytes.len - (512 - 392)];
|
||||
var file = try File.init(allocator, unpadded);
|
||||
defer file.deinit();
|
||||
const got = try file.readStream(allocator, file.find("Workbook").?);
|
||||
defer allocator.free(got);
|
||||
try testing.expectEqualSlices(u8, big, got);
|
||||
}
|
||||
|
||||
test "a directory entry with an impossible name length is corrupt" {
|
||||
var raw: [dir_entry_size]u8 = @splat(0);
|
||||
std.mem.writeInt(u16, raw[64..66], 66, .little);
|
||||
try testing.expectError(error.CorruptFile, parseEntry(&raw, true));
|
||||
std.mem.writeInt(u16, raw[64..66], 3, .little);
|
||||
try testing.expectError(error.CorruptFile, parseEntry(&raw, true));
|
||||
}
|
||||
|
||||
test "version 3 entries ignore the high half of the size" {
|
||||
var raw: [dir_entry_size]u8 = @splat(0);
|
||||
std.mem.writeInt(u64, raw[120..128], 0xDEADBEEF_00000010, .little);
|
||||
try testing.expectEqual(@as(u64, 0x10), (try parseEntry(&raw, true)).size);
|
||||
try testing.expectEqual(@as(u64, 0xDEADBEEF_00000010), (try parseEntry(&raw, false)).size);
|
||||
}
|
||||
|
||||
test "the root entry must come first" {
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try test_writer.buildCfb(allocator, &.{.{ .name = "S", .data = "data" }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
// Root is the first entry of the first directory sector, which the
|
||||
// writer places right after the FAT. Retype it as a stream.
|
||||
const dir_off = (@as(usize, readU32(bytes, 48)) + 1) * 512;
|
||||
bytes[dir_off + 66] = @intFromEnum(EntryType.stream);
|
||||
try testing.expectError(error.CorruptFile, File.init(allocator, bytes));
|
||||
}
|
||||
183
src/root.zig
Normal file
183
src/root.zig
Normal file
|
|
@ -0,0 +1,183 @@
|
|||
//! biff8: a read-only reader for legacy binary Excel workbooks (`.xls`,
|
||||
//! Excel 97 through 2003).
|
||||
//!
|
||||
//! These files are BIFF8 record streams inside a Compound File Binary
|
||||
//! ("OLE2") container. This library unwraps the container and decodes
|
||||
//! each worksheet's cell values: text, numbers, booleans, error codes,
|
||||
//! and the cached results of formulas. It does not decode formatting,
|
||||
//! evaluate formulas, read `.xlsx` (a different format entirely), or
|
||||
//! write anything.
|
||||
//!
|
||||
//! ```zig
|
||||
//! var wb = try biff8.Workbook.parse(allocator, bytes);
|
||||
//! defer wb.deinit();
|
||||
//! const sheet = wb.sheets[0];
|
||||
//! switch (sheet.cell(row, col)) {
|
||||
//! .text => |t| ...,
|
||||
//! .number => |n| ...,
|
||||
//! else => {},
|
||||
//! }
|
||||
//! ```
|
||||
//!
|
||||
//! Numbers are returned exactly as stored, so a date-formatted cell is
|
||||
//! an Excel serial day number. Converting it requires knowing the cell
|
||||
//! is a date, which lives in the number format this library skips.
|
||||
|
||||
const std = @import("std");
|
||||
const cfb = @import("cfb.zig");
|
||||
const biff = @import("biff.zig");
|
||||
|
||||
pub const Cell = biff.Cell;
|
||||
pub const Sheet = biff.Sheet;
|
||||
|
||||
pub const ParseError = cfb.Error || biff.Error || error{
|
||||
/// The container holds no `Workbook` stream, so it is some other
|
||||
/// kind of compound file (a `.doc`, an `.msg`, ...).
|
||||
NoWorkbookStream,
|
||||
};
|
||||
|
||||
/// True when `bytes` starts with the compound-file signature. Every
|
||||
/// `.xls` does, but so do other legacy Office files; use it for cheap
|
||||
/// content sniffing, not as proof the bytes are a workbook.
|
||||
pub const isCompoundFile = cfb.isCompoundFile;
|
||||
|
||||
pub const Workbook = struct {
|
||||
arena: std.heap.ArenaAllocator,
|
||||
/// Worksheets in workbook order. Chart sheets, macro sheets and VB
|
||||
/// modules are omitted.
|
||||
sheets: []const Sheet,
|
||||
|
||||
/// Parse a workbook. `bytes` is only read during the call; the
|
||||
/// result owns everything it references.
|
||||
pub fn parse(allocator: std.mem.Allocator, bytes: []const u8) ParseError!Workbook {
|
||||
var file = try cfb.File.init(allocator, bytes);
|
||||
defer file.deinit();
|
||||
|
||||
const entry = file.find("Workbook") orelse {
|
||||
// Excel 5 and 95 (BIFF5) named the stream "Book".
|
||||
if (file.find("Book") != null) return error.UnsupportedBiffVersion;
|
||||
return error.NoWorkbookStream;
|
||||
};
|
||||
const stream = try file.readStream(allocator, entry);
|
||||
defer allocator.free(stream);
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(allocator);
|
||||
errdefer arena.deinit();
|
||||
const sheets = try biff.parseSheets(arena.allocator(), allocator, stream);
|
||||
return .{ .arena = arena, .sheets = sheets };
|
||||
}
|
||||
|
||||
pub fn deinit(self: *Workbook) void {
|
||||
self.arena.deinit();
|
||||
}
|
||||
|
||||
/// The first worksheet named exactly `name`.
|
||||
pub fn sheet(self: *const Workbook, name: []const u8) ?*const Sheet {
|
||||
for (self.sheets) |*s| {
|
||||
if (std.mem.eql(u8, s.name, name)) return s;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
// ---- Tests ----
|
||||
|
||||
const testing = std.testing;
|
||||
const tw = @import("test_writer.zig");
|
||||
|
||||
test {
|
||||
std.testing.refAllDecls(@This());
|
||||
_ = cfb;
|
||||
_ = biff;
|
||||
}
|
||||
|
||||
test "parse: end to end through the compound file" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
|
||||
const sst = try tw.sst(a, &.{ .{ .text = "Symbol" }, .{ .text = "SAMPLE" } }, 8224);
|
||||
const bytes = try tw.xls(a, .{
|
||||
.globals = sst,
|
||||
.sheets = &.{
|
||||
.{ .name = "Sample_Positions", .records = &.{
|
||||
try tw.labelSst(a, 0, 0, 0),
|
||||
try tw.labelSst(a, 1, 0, 1),
|
||||
try tw.number(a, 1, 1, 100.25),
|
||||
} },
|
||||
.{ .name = "Notes" },
|
||||
},
|
||||
});
|
||||
try testing.expect(isCompoundFile(bytes));
|
||||
|
||||
var wb = try Workbook.parse(testing.allocator, bytes);
|
||||
defer wb.deinit();
|
||||
try testing.expectEqual(@as(usize, 2), wb.sheets.len);
|
||||
const s = wb.sheet("Sample_Positions").?;
|
||||
try testing.expectEqualStrings("Symbol", s.cell(0, 0).asText().?);
|
||||
try testing.expectEqualStrings("SAMPLE", s.cell(1, 0).asText().?);
|
||||
try testing.expectEqual(@as(f64, 100.25), s.cell(1, 1).asNumber().?);
|
||||
try testing.expect(wb.sheet("Notes") != null);
|
||||
try testing.expect(wb.sheet("notes") == null);
|
||||
}
|
||||
|
||||
test "parse: a large workbook stream lives outside the mini stream" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
|
||||
var records: std.ArrayList(tw.Record) = .empty;
|
||||
for (0..1000) |i| try records.append(a, try tw.number(a, @intCast(i), 0, @floatFromInt(i)));
|
||||
const bytes = try tw.xls(a, .{ .sheets = &.{.{ .name = "Big", .records = records.items }} });
|
||||
|
||||
var wb = try Workbook.parse(testing.allocator, bytes);
|
||||
defer wb.deinit();
|
||||
try testing.expectEqual(@as(usize, 1000), wb.sheets[0].rows.len);
|
||||
try testing.expectEqual(@as(f64, 999), wb.sheets[0].cell(999, 0).asNumber().?);
|
||||
}
|
||||
|
||||
test "parse: compound files that are not BIFF8 workbooks" {
|
||||
const allocator = testing.allocator;
|
||||
|
||||
const doc = try tw.buildCfb(allocator, &.{.{ .name = "WordDocument", .data = "not a workbook" }}, .{});
|
||||
defer allocator.free(doc);
|
||||
try testing.expectError(error.NoWorkbookStream, Workbook.parse(allocator, doc));
|
||||
|
||||
const biff5 = try tw.buildCfb(allocator, &.{.{ .name = "Book", .data = "excel 95" }}, .{});
|
||||
defer allocator.free(biff5);
|
||||
try testing.expectError(error.UnsupportedBiffVersion, Workbook.parse(allocator, biff5));
|
||||
|
||||
try testing.expectError(error.NotCompoundFile, Workbook.parse(allocator, "PK\x03\x04 xlsx is a zip"));
|
||||
}
|
||||
|
||||
test "parse: errors after the container is read release everything" {
|
||||
// testing.allocator fails the test on any leak along the error path.
|
||||
const allocator = testing.allocator;
|
||||
const bytes = try tw.buildCfb(allocator, &.{.{ .name = "Workbook", .data = "\x09\x08\x02\x00" }}, .{});
|
||||
defer allocator.free(bytes);
|
||||
try testing.expectError(error.CorruptFile, Workbook.parse(allocator, bytes));
|
||||
}
|
||||
|
||||
fn parseAndRelease(allocator: std.mem.Allocator, bytes: []const u8) !void {
|
||||
var wb = try Workbook.parse(allocator, bytes);
|
||||
wb.deinit();
|
||||
}
|
||||
|
||||
test "parse: every allocation failure is clean" {
|
||||
var fx = std.heap.ArenaAllocator.init(testing.allocator);
|
||||
defer fx.deinit();
|
||||
const a = fx.allocator();
|
||||
|
||||
// Small (mini stream) and large (regular sectors) workbooks take
|
||||
// different allocation paths through the container reader.
|
||||
const sst = try tw.sst(a, &.{ .{ .text = "alpha" }, .{ .text = "\u{3042}" } }, 8224);
|
||||
var records: std.ArrayList(tw.Record) = .empty;
|
||||
try records.append(a, try tw.labelSst(a, 0, 0, 0));
|
||||
try records.append(a, try tw.labelSst(a, 0, 1, 1));
|
||||
const small = try tw.xls(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} });
|
||||
for (1..600) |i| try records.append(a, try tw.number(a, @intCast(i), 0, 1));
|
||||
const large = try tw.xls(a, .{ .globals = sst, .sheets = &.{.{ .name = "S", .records = records.items }} });
|
||||
|
||||
try testing.checkAllAllocationFailures(testing.allocator, parseAndRelease, .{small});
|
||||
try testing.checkAllAllocationFailures(testing.allocator, parseAndRelease, .{large});
|
||||
}
|
||||
508
src/test_writer.zig
Normal file
508
src/test_writer.zig
Normal file
|
|
@ -0,0 +1,508 @@
|
|||
//! Test-only builders for compound files and BIFF8 record streams.
|
||||
//!
|
||||
//! The reader's tests describe their fixtures in code with these
|
||||
//! helpers instead of checking in binary `.xls` files. That keeps every
|
||||
//! fixture readable and placeholder-only, and lets a test produce
|
||||
//! layouts that real writers rarely emit (DIFAT sectors, strings split
|
||||
//! across CONTINUE records at every possible point, unpaired UTF-16
|
||||
//! surrogates).
|
||||
//!
|
||||
//! Nothing here is validated against a second implementation; the
|
||||
//! reader's tests against it check self-consistency. The real-world
|
||||
//! check is parsing an actual Excel-produced file, which lives outside
|
||||
//! the test suite because such files carry personal data.
|
||||
|
||||
const std = @import("std");
|
||||
const Allocator = std.mem.Allocator;
|
||||
|
||||
// ---- Compound file ----
|
||||
|
||||
pub const Stream = struct {
|
||||
name: []const u8,
|
||||
data: []const u8,
|
||||
};
|
||||
|
||||
pub const CfbOptions = struct {
|
||||
/// 3 (512-byte sectors) or 4 (4096-byte sectors).
|
||||
major_version: u16 = 3,
|
||||
};
|
||||
|
||||
const free_sect: u32 = 0xFFFFFFFF;
|
||||
const end_of_chain: u32 = 0xFFFFFFFE;
|
||||
const fat_sect: u32 = 0xFFFFFFFD;
|
||||
const difat_sect: u32 = 0xFFFFFFFC;
|
||||
const no_stream: u32 = 0xFFFFFFFF;
|
||||
const mini_cutoff = 4096;
|
||||
const mini_sector_size = 64;
|
||||
|
||||
/// Build a compound file holding `streams` in its root storage.
|
||||
/// Streams under 4096 bytes go in the mini stream, larger ones get
|
||||
/// their own sector chains. Layout, in sector order: FAT, DIFAT,
|
||||
/// directory, mini FAT, mini stream, then each large stream (so the
|
||||
/// last stream's data ends the file).
|
||||
pub fn buildCfb(allocator: Allocator, streams: []const Stream, opts: CfbOptions) ![]u8 {
|
||||
const shift: u5 = switch (opts.major_version) {
|
||||
3 => 9,
|
||||
4 => 12,
|
||||
else => unreachable,
|
||||
};
|
||||
const sector_size: usize = @as(usize, 1) << shift;
|
||||
const per_fat = sector_size / 4;
|
||||
|
||||
// Mini stream contents and per-stream placement.
|
||||
var mini: std.ArrayList(u8) = .empty;
|
||||
defer mini.deinit(allocator);
|
||||
const starts = try allocator.alloc(u32, streams.len);
|
||||
defer allocator.free(starts);
|
||||
var mini_sectors: usize = 0;
|
||||
var regular_sectors: usize = 0;
|
||||
for (streams, 0..) |s, i| {
|
||||
if (s.data.len < mini_cutoff) {
|
||||
const n = ceilDiv(s.data.len, mini_sector_size);
|
||||
starts[i] = if (n == 0) end_of_chain else @intCast(mini_sectors);
|
||||
mini_sectors += n;
|
||||
try mini.appendSlice(allocator, s.data);
|
||||
try mini.appendNTimes(allocator, 0, n * mini_sector_size - s.data.len);
|
||||
} else {
|
||||
regular_sectors += ceilDiv(s.data.len, sector_size);
|
||||
}
|
||||
}
|
||||
|
||||
const dir_sectors = ceilDiv((streams.len + 1) * 128, sector_size);
|
||||
const mini_fat_sectors = ceilDiv(mini_sectors * 4, sector_size);
|
||||
const mini_stream_sectors = ceilDiv(mini.items.len, sector_size);
|
||||
const data_sectors = dir_sectors + mini_fat_sectors + mini_stream_sectors + regular_sectors;
|
||||
|
||||
// Smallest FAT (plus the DIFAT sectors it needs) that addresses
|
||||
// every sector including itself.
|
||||
var fat_sectors: usize = 1;
|
||||
var difat_sectors: usize = 0;
|
||||
while (true) : (fat_sectors += 1) {
|
||||
difat_sectors = if (fat_sectors > 109) ceilDiv(fat_sectors - 109, per_fat - 1) else 0;
|
||||
if (fat_sectors * per_fat >= data_sectors + fat_sectors + difat_sectors) break;
|
||||
}
|
||||
|
||||
const total_sectors = fat_sectors + difat_sectors + data_sectors;
|
||||
const out = try allocator.alloc(u8, (total_sectors + 1) * sector_size);
|
||||
errdefer allocator.free(out);
|
||||
@memset(out, 0);
|
||||
|
||||
const fat = try allocator.alloc(u32, fat_sectors * per_fat);
|
||||
defer allocator.free(fat);
|
||||
@memset(fat, free_sect);
|
||||
|
||||
var next: usize = 0;
|
||||
const fat_first = next;
|
||||
for (0..fat_sectors) |i| fat[fat_first + i] = fat_sect;
|
||||
next += fat_sectors;
|
||||
const difat_first = next;
|
||||
for (0..difat_sectors) |i| fat[difat_first + i] = difat_sect;
|
||||
next += difat_sectors;
|
||||
const dir_first = chain(fat, &next, dir_sectors);
|
||||
const mini_fat_first = chain(fat, &next, mini_fat_sectors);
|
||||
const mini_stream_first = chain(fat, &next, mini_stream_sectors);
|
||||
for (streams, 0..) |s, i| {
|
||||
if (s.data.len >= mini_cutoff) starts[i] = chain(fat, &next, ceilDiv(s.data.len, sector_size));
|
||||
}
|
||||
|
||||
// Header.
|
||||
const hdr = out[0..512];
|
||||
@memcpy(hdr[0..8], &[_]u8{ 0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1 });
|
||||
put16(hdr, 24, 0x003E);
|
||||
put16(hdr, 26, opts.major_version);
|
||||
put16(hdr, 28, 0xFFFE);
|
||||
put16(hdr, 30, shift);
|
||||
put16(hdr, 32, 6);
|
||||
put32(hdr, 40, if (opts.major_version == 4) @intCast(dir_sectors) else 0);
|
||||
put32(hdr, 44, @intCast(fat_sectors));
|
||||
put32(hdr, 48, dir_first);
|
||||
put32(hdr, 56, mini_cutoff);
|
||||
put32(hdr, 60, mini_fat_first);
|
||||
put32(hdr, 64, @intCast(mini_fat_sectors));
|
||||
put32(hdr, 68, if (difat_sectors > 0) @intCast(difat_first) else end_of_chain);
|
||||
put32(hdr, 72, @intCast(difat_sectors));
|
||||
for (0..109) |i| put32(hdr, 76 + 4 * i, if (i < fat_sectors) @intCast(fat_first + i) else free_sect);
|
||||
|
||||
// DIFAT sectors: FAT sector numbers past the first 109, last slot
|
||||
// links to the next DIFAT sector.
|
||||
for (0..difat_sectors) |d| {
|
||||
const s = sectorBytes(out, sector_size, difat_first + d);
|
||||
for (0..per_fat - 1) |i| {
|
||||
const n = 109 + d * (per_fat - 1) + i;
|
||||
put32(s, 4 * i, if (n < fat_sectors) @intCast(fat_first + n) else free_sect);
|
||||
}
|
||||
put32(s, sector_size - 4, if (d + 1 < difat_sectors) @intCast(difat_first + d + 1) else end_of_chain);
|
||||
}
|
||||
|
||||
// FAT.
|
||||
for (fat, 0..) |v, i| put32(out[sector_size..], 4 * i, v);
|
||||
|
||||
// Directory: root, then each stream linked as a right-sibling list.
|
||||
{
|
||||
const dir = try allocator.alloc(u8, dir_sectors * sector_size);
|
||||
defer allocator.free(dir);
|
||||
@memset(dir, 0);
|
||||
var e: usize = 0;
|
||||
while (e * 128 < dir.len) : (e += 1) {
|
||||
const raw = dir[e * 128 ..][0..128];
|
||||
put32(raw, 68, no_stream);
|
||||
put32(raw, 72, no_stream);
|
||||
put32(raw, 76, no_stream);
|
||||
}
|
||||
writeEntry(dir[0..128], "Root Entry", 5, if (streams.len > 0) 1 else no_stream, no_stream, if (mini.items.len > 0) mini_stream_first else end_of_chain, mini.items.len);
|
||||
for (streams, 0..) |s, i| {
|
||||
const right: u32 = if (i + 1 < streams.len) @intCast(i + 2) else no_stream;
|
||||
writeEntry(dir[(i + 1) * 128 ..][0..128], s.name, 2, no_stream, right, starts[i], s.data.len);
|
||||
}
|
||||
writeChain(out, sector_size, dir_first, dir);
|
||||
}
|
||||
|
||||
// Mini FAT and mini stream.
|
||||
{
|
||||
const table = try allocator.alloc(u8, mini_fat_sectors * sector_size);
|
||||
defer allocator.free(table);
|
||||
@memset(table, 0xFF);
|
||||
var idx: usize = 0;
|
||||
for (streams) |s| {
|
||||
if (s.data.len >= mini_cutoff) continue;
|
||||
const n = ceilDiv(s.data.len, mini_sector_size);
|
||||
for (0..n) |k| put32(table, 4 * (idx + k), if (k + 1 < n) @intCast(idx + k + 1) else end_of_chain);
|
||||
idx += n;
|
||||
}
|
||||
writeChain(out, sector_size, mini_fat_first, table);
|
||||
writeChain(out, sector_size, mini_stream_first, mini.items);
|
||||
}
|
||||
|
||||
for (streams, 0..) |s, i| {
|
||||
if (s.data.len >= mini_cutoff) writeChain(out, sector_size, starts[i], s.data);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/// Allocate `n` consecutive sectors as one chain. Returns the first
|
||||
/// sector, or ENDOFCHAIN for an empty chain.
|
||||
fn chain(fat: []u32, next: *usize, n: usize) u32 {
|
||||
if (n == 0) return end_of_chain;
|
||||
const first = next.*;
|
||||
for (0..n) |i| fat[first + i] = if (i + 1 < n) @intCast(first + i + 1) else end_of_chain;
|
||||
next.* += n;
|
||||
return @intCast(first);
|
||||
}
|
||||
|
||||
/// Copy `data` into consecutive sectors starting at `first` (the
|
||||
/// builder always allocates chains contiguously).
|
||||
fn writeChain(out: []u8, sector_size: usize, first: u32, data: []const u8) void {
|
||||
if (data.len == 0) return;
|
||||
const off = (@as(usize, first) + 1) * sector_size;
|
||||
@memcpy(out[off..][0..data.len], data);
|
||||
}
|
||||
|
||||
fn sectorBytes(out: []u8, sector_size: usize, index: usize) []u8 {
|
||||
return out[(index + 1) * sector_size ..][0..sector_size];
|
||||
}
|
||||
|
||||
fn writeEntry(raw: *[128]u8, name: []const u8, kind: u8, child: u32, right: u32, start: u32, size: usize) void {
|
||||
for (name, 0..) |c, i| put16(raw, 2 * i, c);
|
||||
put16(raw, 64, @intCast((name.len + 1) * 2));
|
||||
raw[66] = kind;
|
||||
raw[67] = 1; // black
|
||||
put32(raw, 72, right);
|
||||
put32(raw, 76, child);
|
||||
put32(raw, 116, start);
|
||||
std.mem.writeInt(u64, raw[120..128], size, .little);
|
||||
}
|
||||
|
||||
// ---- BIFF8 records ----
|
||||
|
||||
pub const Record = struct {
|
||||
kind: u16,
|
||||
data: []const u8,
|
||||
};
|
||||
|
||||
pub const rt = struct {
|
||||
pub const bof: u16 = 0x0809;
|
||||
pub const eof: u16 = 0x000A;
|
||||
pub const filepass: u16 = 0x002F;
|
||||
pub const boundsheet: u16 = 0x0085;
|
||||
pub const sst: u16 = 0x00FC;
|
||||
pub const @"continue": u16 = 0x003C;
|
||||
pub const labelsst: u16 = 0x00FD;
|
||||
pub const number: u16 = 0x0203;
|
||||
pub const rk: u16 = 0x027E;
|
||||
pub const mulrk: u16 = 0x00BD;
|
||||
pub const label: u16 = 0x0204;
|
||||
pub const rstring: u16 = 0x00D6;
|
||||
pub const boolerr: u16 = 0x0205;
|
||||
pub const formula: u16 = 0x0006;
|
||||
pub const string: u16 = 0x0207;
|
||||
pub const blank: u16 = 0x0201;
|
||||
pub const dimensions: u16 = 0x0200;
|
||||
};
|
||||
|
||||
pub const SheetSpec = struct {
|
||||
name: []const u8,
|
||||
records: []const Record = &.{},
|
||||
/// BOUNDSHEET8 `dt`: 0 worksheet, 2 chart, 6 VB module.
|
||||
boundsheet_type: u8 = 0,
|
||||
/// BOF `dt` for the substream; 0x0010 worksheet, 0x0020 chart.
|
||||
bof_type: u16 = 0x0010,
|
||||
};
|
||||
|
||||
pub const WorkbookSpec = struct {
|
||||
version: u16 = 0x0600,
|
||||
/// Records between the globals BOF and the BOUNDSHEET records
|
||||
/// (SST, FILEPASS, ...).
|
||||
globals: []const Record = &.{},
|
||||
sheets: []const SheetSpec = &.{},
|
||||
/// Leave the final sheet without its EOF record.
|
||||
omit_last_eof: bool = false,
|
||||
/// Overrides every BOUNDSHEET offset (for out-of-range tests).
|
||||
offset_override: ?u32 = null,
|
||||
};
|
||||
|
||||
/// Assemble a Workbook stream: globals BOF, `globals`, one BOUNDSHEET
|
||||
/// per sheet (offsets patched once the substreams are placed), EOF,
|
||||
/// then each sheet substream.
|
||||
pub fn workbook(allocator: Allocator, spec: WorkbookSpec) ![]u8 {
|
||||
var out: std.ArrayList(u8) = .empty;
|
||||
errdefer out.deinit(allocator);
|
||||
|
||||
try appendRecord(allocator, &out, rt.bof, &bof(spec.version, 0x0005));
|
||||
for (spec.globals) |r| try appendRecord(allocator, &out, r.kind, r.data);
|
||||
|
||||
const offset_slots = try allocator.alloc(usize, spec.sheets.len);
|
||||
defer allocator.free(offset_slots);
|
||||
for (spec.sheets, 0..) |s, i| {
|
||||
var data: std.ArrayList(u8) = .empty;
|
||||
defer data.deinit(allocator);
|
||||
try data.appendNTimes(allocator, 0, 4); // lbPlyPos, patched below
|
||||
try data.append(allocator, 0); // visible
|
||||
try data.append(allocator, s.boundsheet_type);
|
||||
try data.append(allocator, @intCast(s.name.len));
|
||||
try data.append(allocator, 0); // compressed characters
|
||||
try data.appendSlice(allocator, s.name);
|
||||
offset_slots[i] = out.items.len + 4;
|
||||
try appendRecord(allocator, &out, rt.boundsheet, data.items);
|
||||
}
|
||||
try appendRecord(allocator, &out, rt.eof, "");
|
||||
|
||||
for (spec.sheets, 0..) |s, i| {
|
||||
const offset: u32 = spec.offset_override orelse @intCast(out.items.len);
|
||||
put32(out.items, offset_slots[i], offset);
|
||||
try appendRecord(allocator, &out, rt.bof, &bof(spec.version, s.bof_type));
|
||||
for (s.records) |r| try appendRecord(allocator, &out, r.kind, r.data);
|
||||
if (!(spec.omit_last_eof and i + 1 == spec.sheets.len)) try appendRecord(allocator, &out, rt.eof, "");
|
||||
}
|
||||
return out.toOwnedSlice(allocator);
|
||||
}
|
||||
|
||||
/// Workbook stream wrapped in a compound file as the `Workbook` stream.
|
||||
pub fn xls(allocator: Allocator, spec: WorkbookSpec) ![]u8 {
|
||||
const stream = try workbook(allocator, spec);
|
||||
defer allocator.free(stream);
|
||||
return buildCfb(allocator, &.{.{ .name = "Workbook", .data = stream }}, .{});
|
||||
}
|
||||
|
||||
fn appendRecord(allocator: Allocator, out: *std.ArrayList(u8), kind: u16, data: []const u8) !void {
|
||||
// SAFETY: both halves are written immediately below.
|
||||
var head: [4]u8 = undefined;
|
||||
put16(&head, 0, kind);
|
||||
put16(&head, 2, @intCast(data.len));
|
||||
try out.appendSlice(allocator, &head);
|
||||
try out.appendSlice(allocator, data);
|
||||
}
|
||||
|
||||
pub fn bof(version: u16, dt: u16) [16]u8 {
|
||||
var b: [16]u8 = @splat(0);
|
||||
put16(&b, 0, version);
|
||||
put16(&b, 2, dt);
|
||||
return b;
|
||||
}
|
||||
|
||||
// The single-cell builders allocate their record bodies so the
|
||||
// returned `Record` stays valid; tests pass an arena.
|
||||
|
||||
pub fn labelSst(allocator: Allocator, row: u16, col: u16, index: u32) !Record {
|
||||
return cellRecord(allocator, rt.labelsst, row, col, u32, index);
|
||||
}
|
||||
|
||||
pub fn number(allocator: Allocator, row: u16, col: u16, value: f64) !Record {
|
||||
return cellRecord(allocator, rt.number, row, col, u64, @bitCast(value));
|
||||
}
|
||||
|
||||
pub fn rk(allocator: Allocator, row: u16, col: u16, raw: u32) !Record {
|
||||
return cellRecord(allocator, rt.rk, row, col, u32, raw);
|
||||
}
|
||||
|
||||
pub fn boolErr(allocator: Allocator, row: u16, col: u16, value: u8, is_error: bool) !Record {
|
||||
return cellRecord(allocator, rt.boolerr, row, col, u16, @as(u16, value) | (@as(u16, @intFromBool(is_error)) << 8));
|
||||
}
|
||||
|
||||
pub fn blank(allocator: Allocator, row: u16, col: u16) !Record {
|
||||
return cellRecord(allocator, rt.blank, row, col, void, {});
|
||||
}
|
||||
|
||||
/// FORMULA record with an 8-byte cached value (see `formulaNumber`,
|
||||
/// `formulaSpecial`) and an empty expression.
|
||||
pub fn formula(allocator: Allocator, row: u16, col: u16, value: [8]u8) !Record {
|
||||
const b = try allocator.alloc(u8, 6 + 8 + 2 + 4 + 2);
|
||||
@memset(b, 0);
|
||||
put16(b, 0, row);
|
||||
put16(b, 2, col);
|
||||
@memcpy(b[6..14], &value);
|
||||
return .{ .kind = rt.formula, .data = b };
|
||||
}
|
||||
|
||||
pub fn formulaNumber(value: f64) [8]u8 {
|
||||
// SAFETY: writeInt fills all eight bytes.
|
||||
var v: [8]u8 = undefined;
|
||||
std.mem.writeInt(u64, &v, @bitCast(value), .little);
|
||||
return v;
|
||||
}
|
||||
|
||||
/// Non-numeric cached formula result: 0 string (in a following STRING
|
||||
/// record), 1 boolean, 2 error, 3 empty string.
|
||||
pub fn formulaSpecial(kind: u8, payload: u8) [8]u8 {
|
||||
return .{ kind, 0, payload, 0, 0, 0, 0xFF, 0xFF };
|
||||
}
|
||||
|
||||
/// MULRK: consecutive RK cells starting at `first_col`.
|
||||
pub fn mulrk(allocator: Allocator, row: u16, first_col: u16, raws: []const u32) !Record {
|
||||
const b = try allocator.alloc(u8, 4 + 6 * raws.len + 2);
|
||||
put16(b, 0, row);
|
||||
put16(b, 2, first_col);
|
||||
for (raws, 0..) |r, i| {
|
||||
put16(b, 4 + 6 * i, 0);
|
||||
put32(b, 4 + 6 * i + 2, r);
|
||||
}
|
||||
put16(b, b.len - 2, first_col + @as(u16, @intCast(raws.len)) - 1);
|
||||
return .{ .kind = rt.mulrk, .data = b };
|
||||
}
|
||||
|
||||
/// LABEL / RSTRING with an inline compressed (Latin-1) string.
|
||||
pub fn label(allocator: Allocator, kind: u16, row: u16, col: u16, text: []const u8) !Record {
|
||||
const b = try allocator.alloc(u8, 6 + 3 + text.len);
|
||||
@memset(b, 0);
|
||||
put16(b, 0, row);
|
||||
put16(b, 2, col);
|
||||
put16(b, 6, @intCast(text.len));
|
||||
@memcpy(b[9..], text);
|
||||
return .{ .kind = kind, .data = b };
|
||||
}
|
||||
|
||||
/// STRING record (formula string result), compressed characters.
|
||||
pub fn string(allocator: Allocator, text: []const u8) !Record {
|
||||
const b = try allocator.alloc(u8, 3 + text.len);
|
||||
put16(b, 0, @intCast(text.len));
|
||||
b[2] = 0;
|
||||
@memcpy(b[3..], text);
|
||||
return .{ .kind = rt.string, .data = b };
|
||||
}
|
||||
|
||||
fn cellRecord(allocator: Allocator, kind: u16, row: u16, col: u16, comptime T: type, value: T) !Record {
|
||||
const b = try allocator.alloc(u8, 6 + @sizeOf(T));
|
||||
@memset(b, 0);
|
||||
put16(b, 0, row);
|
||||
put16(b, 2, col);
|
||||
if (T != void) std.mem.writeInt(T, b[6..][0..@sizeOf(T)], value, .little);
|
||||
return .{ .kind = kind, .data = b };
|
||||
}
|
||||
|
||||
pub const SstString = struct {
|
||||
/// UTF-8 text. Ignored when `units` is set.
|
||||
text: []const u8 = "",
|
||||
/// Raw UTF-16 code units, for strings UTF-8 cannot express (an
|
||||
/// unpaired surrogate).
|
||||
units: ?[]const u16 = null,
|
||||
/// Rich-text run count; the runs themselves are filler bytes.
|
||||
runs: u16 = 0,
|
||||
/// Phonetic (ExtRst) bytes.
|
||||
ext: []const u8 = "",
|
||||
};
|
||||
|
||||
/// SST plus CONTINUE records, split so no record body exceeds
|
||||
/// `max_chunk` bytes. Splits land wherever the limit falls: inside
|
||||
/// a string's header, characters, runs or ExtRst. A split inside the
|
||||
/// characters starts the next record with a fresh high-byte flag
|
||||
/// chosen for the remaining characters, as Excel does, so one string
|
||||
/// can switch between compressed and UTF-16 storage mid-way.
|
||||
pub fn sst(allocator: Allocator, strings: []const SstString, max_chunk: usize) ![]Record {
|
||||
var w: ChunkWriter = .{ .allocator = allocator, .max = max_chunk };
|
||||
defer w.chunks.deinit(allocator);
|
||||
|
||||
try w.int(u32, @intCast(strings.len));
|
||||
try w.int(u32, @intCast(strings.len));
|
||||
for (strings) |s| {
|
||||
const owned = if (s.units == null) try std.unicode.utf8ToUtf16LeAlloc(allocator, s.text) else null;
|
||||
defer if (owned) |o| allocator.free(o);
|
||||
const units = s.units orelse owned.?;
|
||||
|
||||
var flags: u8 = if (anyHigh(units)) 1 else 0;
|
||||
if (s.ext.len > 0) flags |= 0x04;
|
||||
if (s.runs > 0) flags |= 0x08;
|
||||
try w.int(u16, @intCast(units.len));
|
||||
try w.byte(flags);
|
||||
if (s.runs > 0) try w.int(u16, s.runs);
|
||||
if (s.ext.len > 0) try w.int(u32, @intCast(s.ext.len));
|
||||
|
||||
var high = flags & 1 != 0;
|
||||
for (units, 0..) |u, i| {
|
||||
const width: usize = if (high) 2 else 1;
|
||||
if (w.current.items.len + width > w.max) {
|
||||
try w.finish();
|
||||
high = anyHigh(units[i..]);
|
||||
try w.current.append(allocator, @intFromBool(high));
|
||||
}
|
||||
if (high) {
|
||||
try w.current.append(allocator, @truncate(u));
|
||||
try w.current.append(allocator, @truncate(u >> 8));
|
||||
} else {
|
||||
try w.current.append(allocator, @intCast(u));
|
||||
}
|
||||
}
|
||||
for (0..@as(usize, s.runs) * 4) |_| try w.byte(0xAB);
|
||||
for (s.ext) |b| try w.byte(b);
|
||||
}
|
||||
try w.finish();
|
||||
|
||||
const records = try allocator.alloc(Record, w.chunks.items.len);
|
||||
for (w.chunks.items, 0..) |c, i| records[i] = .{ .kind = if (i == 0) rt.sst else rt.@"continue", .data = c };
|
||||
return records;
|
||||
}
|
||||
|
||||
const ChunkWriter = struct {
|
||||
allocator: Allocator,
|
||||
max: usize,
|
||||
current: std.ArrayList(u8) = .empty,
|
||||
chunks: std.ArrayList([]u8) = .empty,
|
||||
|
||||
fn byte(w: *ChunkWriter, b: u8) !void {
|
||||
if (w.current.items.len == w.max) try w.finish();
|
||||
try w.current.append(w.allocator, b);
|
||||
}
|
||||
|
||||
fn int(w: *ChunkWriter, comptime T: type, v: T) !void {
|
||||
for (0..@sizeOf(T)) |i| try w.byte(@truncate(v >> @intCast(8 * i)));
|
||||
}
|
||||
|
||||
fn finish(w: *ChunkWriter) !void {
|
||||
try w.chunks.append(w.allocator, try w.current.toOwnedSlice(w.allocator));
|
||||
}
|
||||
};
|
||||
|
||||
fn anyHigh(units: []const u16) bool {
|
||||
for (units) |u| if (u > 0xFF) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
fn ceilDiv(a: usize, b: usize) usize {
|
||||
return (a + b - 1) / b;
|
||||
}
|
||||
|
||||
fn put16(b: []u8, off: usize, v: u16) void {
|
||||
std.mem.writeInt(u16, b[off..][0..2], v, .little);
|
||||
}
|
||||
|
||||
fn put32(b: []u8, off: usize, v: u32) void {
|
||||
std.mem.writeInt(u32, b[off..][0..4], v, .little);
|
||||
}
|
||||
Loading…
Add table
Reference in a new issue