zfin/src/srf_lint.zig
Emil Lerch 40869cb3a4
lint srf files in audit / doctor commands
This also provides ensurance through zig build test that all
fields are documented in markdown so they do not get out of sync
2026-09-24 06:39:23 -07:00

1802 lines
70 KiB
Zig

//! SRF schema lint - catch hand-edit typos in user-authored SRF files.
//!
//! ## The problem
//!
//! SRF's `fields.to(T)` silently discards any record key that doesn't
//! name a field of `T`. The relevant loop is `srf.zig`'s:
//!
//! ```zig
//! while (try self.next()) |f| {
//! var field_match = false; // declared here...
//! inline for (std.meta.fields(T)) |type_field| {
//! if (!field_match and std.mem.eql(u8, type_field.name, f.key) ...) {
//! field_match = true; // ...set here...
//! }
//! }
//! } // ...and never read.
//! ```
//!
//! `field_match` is write-only, so a typo'd key is byte-for-byte
//! equivalent to omitting the field. There is no strict mode, no
//! unknown-field callback, and no `error.UnknownField` anywhere in the
//! library. A field WITHOUT a default errors as
//! `FieldNotFoundOnFieldWithoutDefaultValue` (which names the missing
//! field, not the wrong key); a field WITH a default is silently
//! skipped. Most fields on zfin's user-authored models have defaults:
//! 19 of `Lot`'s 22, 14 of `AccountTaxEntry`'s 16, and all 17 of
//! `SrfConfig`'s - the last behind an infallible parser, so a typo
//! there changes projected retirement math with zero output.
//!
//! ## Division of responsibility
//!
//! This module is an AGGREGATION POINT, not a rule book. It owns the
//! raw-record walker, the name matchers, rollup, and the comptime
//! contract validator. It knows nothing about any particular model.
//!
//! Each model module owns its own schema and declares it as
//! `pub const srf_schema` (see `validateSchema` for the contract, and
//! `models/portfolio.zig` for the reference implementation). That
//! placement is deliberate: rules that live away from the fields they
//! describe drift from them, which is exactly how
//! `docs/reference/config/portfolio-srf.md` came to document 14 of
//! `Lot`'s 22 fields. Three mechanisms keep a schema honest:
//!
//! 1. The valid-name set is DERIVED (`std.meta.fields(Record)`),
//! never declared, so it cannot drift.
//! 2. `field_rules` is checked for exhaustiveness over
//! `std.meta.fields(Record)` in BOTH directions at comptime - a
//! new field fails the build until classified, and a renamed
//! field fails the build too.
//! 3. `doc_path` + `undocumented` drive a doc-sync test, so adding
//! a field without documenting it fails `zig build test`.
//!
//! ## Why a separate pass
//!
//! SRF's `RecordIterator`/`FieldIterator` are single-pass and
//! consuming - no peek, no rewind - and `to()` drains the field
//! iterator. Raw inspection and `to()` are therefore mutually
//! exclusive on the same record, so this is a second walk over the
//! same bytes rather than a change to any existing parse path. It runs
//! only from `zfin doctor` and `zfin audit`, never on the hot path.
//! With `.parse_allocator = .none` the keys borrow from the input
//! buffer, so a clean file allocates nothing but the result arena.
const std = @import("std");
const srf = @import("srf");
const comptime_validator = @import("comptime_validator.zig");
const Date = @import("Date.zig");
/// Backticked identifiers found in each `docs/reference/config/*.md`
/// page, extracted at build time by `build/gen_config_docs.zig`.
/// `@embedFile` cannot reach outside `src/`, hence the generator.
const config_docs = @import("config_docs");
/// Longest record this lints in full. Every zfin model is far smaller
/// (`Lot`, the widest, has 22 fields), so the cap only bites on a
/// malformed file; records beyond it are still name-checked, they just
/// stop contributing conditional-applicability findings (which need
/// the whole record buffered to locate the discriminator).
const max_fields_per_record = 64;
/// Upper bound on DISTINCT findings, after rollup. A file that trips
/// more than this is misconfigured in a way a longer list won't
/// clarify, and the renderers have to print whatever we return.
const max_findings = 200;
/// Shortest name length eligible for prefix matching. Below this the
/// suggestion is noise: `pc` would "match" `pct`.
///
/// Three is safe for every model in the registry - no valid field name
/// is a 3-character prefix of another name in the same record's set
/// (`tax_type` and `tax_mix_*` share `tax_` but neither prefixes the
/// other). A test pins that property so a future field addition can't
/// quietly break it.
const min_prefix_len = 3;
// ── Findings ──────────────────────────────────────────────────
pub const Kind = enum {
/// Key matches no field, and no matcher found a near relative.
unknown,
/// Key differs from a real field only by case. SRF matches field
/// names with `std.mem.eql`, so this silently does nothing.
case_mismatch,
/// Key is a prefix of a real field or vice versa - the shape of a
/// dropped or doubled character.
near_miss,
/// Key appears twice in one record. SRF keeps the FIRST and
/// silently drops the rest, so editing the second does nothing.
duplicate,
/// Real field, but never read for this record's discriminator
/// value (e.g. `rate` on a `security_type::stock` lot).
inapplicable,
/// Real field that zfin derives and never reads from a
/// hand-edited file (e.g. `Lot.split_factor`).
derived,
/// Model-specific rule, reported by the schema's `semanticCheck`.
semantic,
};
/// One rolled-up problem. Identity is `(kind, key)`; repeats across
/// records bump `count` and record up to two more line numbers rather
/// than emitting another entry - one typo in a copy-pasted template
/// must not produce 200 lines of output.
pub const Finding = struct {
kind: Kind,
/// The offending field name. Borrows from the `data` passed to
/// `check`, so `data` must outlive the `Result`.
key: []const u8,
/// The real field name this probably meant. Set for
/// `case_mismatch` and `near_miss`. Static (a comptime field
/// name), never owned.
suggestion: ?[]const u8 = null,
/// Free-text explanation. Static or owned by the `Result` arena.
detail: []const u8 = "",
first_line: u32,
count: u32 = 1,
/// Second and third line this appeared on, for a "lines 12, 19,
/// 26, ..." tail. Only `extra_len` entries are meaningful.
extra_lines: [2]u32 = .{ 0, 0 },
extra_len: u8 = 0,
fn noteRepeat(self: *Finding, line: u32) void {
self.count += 1;
if (self.extra_len < self.extra_lines.len) {
self.extra_lines[self.extra_len] = line;
self.extra_len += 1;
}
}
/// Longest string `describe` can produce. Field names are bounded
/// by Zig identifier length in practice, and `detail` by
/// `describeOnly`'s output over one discriminator enum.
pub const describe_max = 320;
/// One-line human description, written into `buf` and returned as a
/// slice of it. Shared by `zfin doctor` and `zfin audit` so the two
/// surfaces cannot word the same finding differently.
///
/// `buf` should be at least `describe_max` bytes; a shorter buffer
/// truncates rather than failing, because a clipped diagnostic is
/// still more useful than none.
pub fn describe(self: Finding, buf: []u8) []const u8 {
var w = std.Io.Writer.fixed(buf);
self.write(&w) catch return buf[0..w.end];
return buf[0..w.end];
}
fn write(self: Finding, w: *std.Io.Writer) !void {
switch (self.kind) {
.unknown => try w.print("unrecognized field '{s}'", .{self.key}),
.case_mismatch => try w.print(
"field '{s}' differs only by case from '{s}' - SRF field names are case-sensitive",
.{ self.key, self.suggestion orelse "" },
),
.near_miss => try w.print(
"unrecognized field '{s}' - did you mean '{s}'?",
.{ self.key, self.suggestion orelse "" },
),
.duplicate => try w.print("field '{s}' appears twice in one record; {s}", .{ self.key, self.detail }),
.inapplicable => try w.print("field '{s}' {s}", .{ self.key, self.detail }),
.derived => try w.print("field '{s}' is {s}", .{ self.key, self.detail }),
.semantic => try w.writeAll(self.detail),
}
if (self.count > 1) {
try w.print(" ({d} records: lines {d}", .{ self.count, self.first_line });
for (self.extra_lines[0..self.extra_len]) |l| try w.print(", {d}", .{l});
try w.writeAll(if (self.count > 1 + self.extra_len) ", ...)" else ")");
}
}
};
/// Findings for one file, plus the shape they were checked against so
/// a renderer can print the valid-name footer.
///
/// Owns an arena for any allocated `detail` strings. `Finding.key`
/// borrows from the `data` slice passed to `check`.
pub const Result = struct {
findings: []const Finding,
shape: Shape,
/// True when `max_findings` was hit and some were dropped.
truncated: bool = false,
arena: std.heap.ArenaAllocator,
pub fn deinit(self: *Result) void {
self.arena.deinit();
}
pub fn isClean(self: Result) bool {
return self.findings.len == 0;
}
/// True when some finding is about a field NAME (unknown, wrong
/// case, near miss) - the cases where printing the valid-name list
/// helps. A report of only lifecycle or duplicate findings names
/// real fields already, and the list would just be noise.
pub fn hasNameFindings(self: Result) bool {
for (self.findings) |f| {
switch (f.kind) {
.unknown, .case_mismatch, .near_miss => return true,
.duplicate, .inapplicable, .derived, .semantic => {},
}
}
return false;
}
/// Order findings by the line they were first seen on, then by key.
///
/// Needed because `checkSemantic` appends a whole second pass after
/// `check`'s, so an unsorted report interleaves line 4 before line
/// 2 and reads like a bug. Call after the last pass that appends.
pub fn sort(self: *Result) void {
// `findings` is const to callers but owned by our arena, so the
// cast is sound - nothing else can alias it.
const items = @constCast(self.findings);
std.mem.sort(Finding, items, {}, struct {
fn lessThan(_: void, a: Finding, b: Finding) bool {
if (a.first_line != b.first_line) return a.first_line < b.first_line;
return std.mem.lessThan(u8, a.key, b.key);
}
}.lessThan);
}
};
/// Inputs a schema's `semanticCheck` may need beyond the record itself.
///
/// Passed by value alongside the record rather than stored on `Sink`,
/// which collects findings and should not also carry inputs.
pub const Context = struct {
/// The current calendar day - not an `as_of`. Rules like "a
/// `close_date` in the future" are nonsensical against a
/// back-dated reference: they would flag every real close made
/// after it. Captured once at the unit-of-work entry point and
/// threaded down, per the `today` rule in AGENTS.md.
today: Date,
};
/// Collects findings during a walk. Passed to a schema's
/// `semanticCheck` so model-owned rules report through the same
/// channel as the generic ones.
///
/// Deliberately has no format-string method: `addOwned` takes a string
/// the caller built with `std.fmt.allocPrint(sink.allocator, ...)` and
/// assumes ownership of it. That keeps `anytype` out of the contract
/// while leaving ownership explicit at the call site.
pub const Sink = struct {
/// The `Result` arena. Anything allocated here lives as long as
/// the `Result`.
allocator: std.mem.Allocator,
findings: std.ArrayList(Finding),
truncated: bool = false,
/// Line of the record being walked. Set by `check`; read by
/// `semanticCheck` implementations via `add*`.
line: u32 = 0,
/// Add a finding whose `detail` is a static string.
pub fn addStatic(self: *Sink, kind: Kind, key: []const u8, detail: []const u8) !void {
try self.push(.{ .kind = kind, .key = key, .detail = detail, .first_line = self.line });
}
/// Add a finding whose `detail` was allocated from
/// `self.allocator`. The `Result` arena frees it.
pub fn addOwned(self: *Sink, kind: Kind, key: []const u8, detail: []const u8) !void {
try self.push(.{ .kind = kind, .key = key, .detail = detail, .first_line = self.line });
}
fn addSuggestion(self: *Sink, kind: Kind, key: []const u8, suggestion: []const u8) !void {
try self.push(.{ .kind = kind, .key = key, .suggestion = suggestion, .first_line = self.line });
}
/// Roll `f` into an existing matching finding, or append it.
///
/// Identity is `(kind, key, detail)` - `detail` included on purpose.
/// The generic kinds carry a static per-kind detail, so they still
/// collapse (one typo across 200 records stays one line). A
/// `semantic` finding's detail names the specific record ("account
/// 'Sample IRA': ..."), so including it keeps two accounts with the
/// same broken field from merging into one line that names only the
/// first.
fn push(self: *Sink, f: Finding) !void {
for (self.findings.items) |*existing| {
if (existing.kind == f.kind and
std.mem.eql(u8, existing.key, f.key) and
std.mem.eql(u8, existing.detail, f.detail))
{
existing.noteRepeat(f.first_line);
return;
}
}
if (self.findings.items.len >= max_findings) {
self.truncated = true;
return;
}
try self.findings.append(self.allocator, f);
}
};
// ── Shape: what a record is allowed to contain ────────────────
/// A conditional-applicability rule for one field.
pub const Rule = struct {
name: []const u8,
/// Discriminator values this field is actually read for. `null`
/// means "read for all of them".
only: ?[]const []const u8 = null,
/// Derived by zfin, never read from a hand-edited file. Presence
/// in a user file is itself a finding.
derived: bool = false,
};
/// Conditional-applicability rules for a flat record, keyed off one
/// enum field's value.
pub const Scopes = struct {
/// Field whose value selects applicability (e.g. `security_type`).
discriminator: []const u8,
/// Value used when the discriminator is absent from a record.
/// DERIVED from the field's declared default by `shapeOfSchema`,
/// never hand-written, so it cannot drift from the struct.
default_value: []const u8,
rules: []const Rule,
fn ruleFor(self: Scopes, name: []const u8) ?Rule {
for (self.rules) |r| {
if (std.mem.eql(u8, r.name, name)) return r;
}
return null;
}
};
/// What field names a record may carry.
pub const Shape = union(enum) {
/// A plain struct: one fixed name set for every record.
flat: Flat,
/// A tagged union: the tag field's value selects the name set.
tagged: Tagged,
pub const Flat = struct {
names: []const []const u8,
scopes: ?Scopes = null,
};
pub const Tagged = struct {
tag_key: []const u8,
variants: []const Variant,
fn variantFor(self: Tagged, tag_value: []const u8) ?Variant {
for (self.variants) |v| {
if (std.mem.eql(u8, v.tag_value, tag_value)) return v;
}
return null;
}
};
pub const Variant = struct {
tag_value: []const u8,
names: []const []const u8,
};
};
/// Field names of `T`, as a comptime slice. The single source of the
/// valid-name set - derived, so it cannot drift from the struct.
///
/// The data is held as a container-level `const` so it has static
/// storage and the returned slice stays valid when this is called from
/// a runtime context.
pub fn namesOf(comptime T: type) []const []const u8 {
const Holder = struct {
const names = blk: {
const fields = std.meta.fields(T);
var n: [fields.len][]const u8 = undefined;
for (fields, 0..) |f, i| n[i] = f.name;
break :blk n;
};
};
return &Holder.names;
}
/// Build the `Shape` for a schema module, validating its contract.
///
/// Handles struct and tagged-union records. For a union, the tag key
/// is `Record.srf_tag_field` when declared and `"type"` otherwise,
/// matching SRF's own rule, and the tag key is always accepted even
/// when the variant struct does not redeclare it - `to()` consumes the
/// tag before recursing into the variant, so `SrfConfig` (which
/// redeclares `type`) and `Journal.Acknowledgment` (which does not)
/// must both lint clean.
pub fn shapeOfSchema(comptime S: type) Shape {
const Holder = struct {
const shape = blk: {
validateSchema(S);
const Record = S.Record;
break :blk switch (@typeInfo(Record)) {
.@"struct" => Shape{ .flat = .{
.names = namesOf(Record),
.scopes = if (@hasDecl(S, "field_rules")) scopesOfSchema(S) else null,
} },
.@"union" => tagged: {
const tag_key = if (@hasDecl(Record, "srf_tag_field")) Record.srf_tag_field else "type";
const vfields = std.meta.fields(Record);
var variants: [vfields.len]Shape.Variant = undefined;
for (vfields, 0..) |vf, i| {
// The tag key is valid for every variant
// whether or not the variant redeclares it.
const inner = namesOf(vf.type);
var names: [inner.len + 1][]const u8 = undefined;
names[0] = tag_key;
var n: usize = 1;
for (inner) |name| {
if (!std.mem.eql(u8, name, tag_key)) {
names[n] = name;
n += 1;
}
}
const frozen = names;
variants[i] = .{ .tag_value = vf.name, .names = frozen[0..n] };
}
const frozen_variants = variants;
break :tagged Shape{ .tagged = .{ .tag_key = tag_key, .variants = &frozen_variants } };
},
else => @compileError("srf_schema `" ++ @typeName(Record) ++
"`: Record must be a struct or tagged union"),
};
};
};
return Holder.shape;
}
/// Assemble `Scopes` from a schema's `field_rules`, deriving
/// `default_value` from the discriminator field's declared default so
/// the two cannot disagree.
fn scopesOfSchema(comptime S: type) Scopes {
comptime {
const Record = S.Record;
const disc = S.scope_discriminator;
const D = @FieldType(Record, disc);
const default_ptr = for (std.meta.fields(Record)) |f| {
if (std.mem.eql(u8, f.name, disc)) break f.default_value_ptr;
} else unreachable;
if (default_ptr == null) {
@compileError("srf_schema `" ++ @typeName(Record) ++ "`: discriminator `" ++ disc ++
"` must have a default value (it selects applicability for records that omit it)");
}
const default_tag: D = @as(*const D, @ptrCast(@alignCast(default_ptr.?))).*;
return .{
.discriminator = disc,
.default_value = @tagName(default_tag),
.rules = &S.field_rules,
};
}
}
// ── Comptime contract validation ──────────────────────────────
/// Assert a model's `srf_schema` conforms to the contract, with a
/// copy-pasteable `@compileError` when it does not. Mirrors
/// `tui/tab_framework.zig`'s `validateTabModule`.
///
/// Required:
/// pub const Record = <struct or tagged union>;
/// pub const file_label = "portfolio.srf";
/// pub const doc_path = "reference/config/portfolio-srf.md";
///
/// Optional:
/// pub const undocumented = [_][]const u8{ ... };
/// pub const scope_discriminator = "security_type"; // with field_rules
/// pub const field_rules = [_]srf_lint.Rule{ ... }; // with scope_discriminator
/// pub fn semanticCheck(rec: Record, ctx: srf_lint.Context, sink: *srf_lint.Sink) !void
pub fn validateSchema(comptime S: type) void {
comptime {
const kind = "SRF schema";
const name = if (@hasDecl(S, "file_label")) S.file_label else @typeName(S);
if (!@hasDecl(S, "Record")) {
@compileError(kind ++ " `" ++ name ++ "` is missing `pub const Record = <type>;`");
}
comptime_validator.expectDeclWithType(
kind,
name,
S,
"file_label",
[]const u8,
"pub const file_label: []const u8 = \"portfolio.srf\";",
);
comptime_validator.expectDeclWithType(
kind,
name,
S,
"doc_path",
[]const u8,
"pub const doc_path: []const u8 = \"reference/config/portfolio-srf.md\";",
);
const has_disc = @hasDecl(S, "scope_discriminator");
const has_rules = @hasDecl(S, "field_rules");
if (has_disc != has_rules) {
@compileError(kind ++ " `" ++ name ++ "`: `scope_discriminator` and `field_rules` " ++
"must be declared together (one selects applicability, the other lists it)");
}
if (has_rules) validateRules(S, kind, name);
if (@hasDecl(S, "semanticCheck")) {
comptime_validator.expectFnInferredError(
kind,
name,
S,
"semanticCheck",
&.{ S.Record, Context, *Sink },
void,
"pub fn semanticCheck(rec: Record, ctx: srf_lint.Context, sink: *srf_lint.Sink) !void",
);
}
}
}
/// Exhaustiveness check over `field_rules`, in both directions.
///
/// This is the mechanism that stops the rules from drifting from the
/// struct: adding a field to `Record` fails the build until it is
/// classified, and renaming one fails the build here too. `only`
/// values are checked against the discriminator enum's tag names, so a
/// typo in the rules themselves is also a compile error.
fn validateRules(comptime S: type, comptime kind: []const u8, comptime name: []const u8) void {
comptime {
// The exhaustiveness check is O(fields x rules) string
// comparisons - 22 x 22 for `Lot` - which overruns the default
// branch budget on its own.
@setEvalBranchQuota(100_000);
const Record = S.Record;
const disc = S.scope_discriminator;
const fields = std.meta.fields(Record);
if (!@hasField(Record, disc)) {
@compileError(kind ++ " `" ++ name ++ "`: scope_discriminator `" ++ disc ++
"` is not a field of " ++ @typeName(Record));
}
const D = @FieldType(Record, disc);
if (@typeInfo(D) != .@"enum") {
@compileError(kind ++ " `" ++ name ++ "`: scope_discriminator `" ++ disc ++
"` must be an enum field, got " ++ @typeName(D));
}
// Every field classified exactly once.
for (fields) |f| {
var seen = 0;
for (S.field_rules) |r| {
if (std.mem.eql(u8, r.name, f.name)) seen += 1;
}
if (seen == 0) {
@compileError(kind ++ " `" ++ name ++ "`: field `" ++ f.name ++
"` is not classified in `field_rules`. Add one of:\n" ++
" .{ .name = \"" ++ f.name ++ "\" }, // read for every " ++ disc ++ "\n" ++
" .{ .name = \"" ++ f.name ++ "\", .only = &.{.some_value} }, // read only for those\n" ++
" .{ .name = \"" ++ f.name ++ "\", .derived = true }, // zfin derives it; never hand-edited");
}
if (seen > 1) {
@compileError(kind ++ " `" ++ name ++ "`: field `" ++ f.name ++
"` is classified more than once in `field_rules`");
}
}
// Every rule names a real field, and every `only` value a real tag.
for (S.field_rules) |r| {
if (!@hasField(Record, r.name)) {
@compileError(kind ++ " `" ++ name ++ "`: `field_rules` entry `" ++ r.name ++
"` is not a field of " ++ @typeName(Record) ++ " (renamed or removed?)");
}
if (r.derived and r.only != null) {
@compileError(kind ++ " `" ++ name ++ "`: `field_rules` entry `" ++ r.name ++
"` sets both `derived` and `only`; a derived field is never hand-edited for any " ++ disc);
}
if (r.only) |vals| {
if (vals.len == 0) {
@compileError(kind ++ " `" ++ name ++ "`: `field_rules` entry `" ++ r.name ++
"` has an empty `only` list; omit `only` for \"read everywhere\" or set `derived`");
}
for (vals) |v| {
if (!@hasField(D, v)) {
@compileError(kind ++ " `" ++ name ++ "`: `field_rules` entry `" ++ r.name ++
"` lists `only` value `" ++ v ++ "`, which is not a tag of " ++ @typeName(D));
}
}
}
}
}
}
// ── Name matching ─────────────────────────────────────────────
fn indexOfName(names: []const []const u8, key: []const u8) ?usize {
for (names, 0..) |n, i| {
if (std.mem.eql(u8, n, key)) return i;
}
return null;
}
/// A real field name differing from `key` only by case. SRF matches
/// with `std.mem.eql`, so these silently do nothing.
fn caseMatch(names: []const []const u8, key: []const u8) ?[]const u8 {
for (names) |n| {
if (std.ascii.eqlIgnoreCase(n, key)) return n;
}
return null;
}
/// A real field name in a prefix relationship with `key`, in either
/// direction - the shape of a dropped or doubled trailing character
/// (`price_dat`, `price_datee`) or a truncation.
///
/// Deliberately exact rather than an edit-distance score: no threshold
/// to tune, and it cannot produce a wrong suggestion. It catches
/// strictly fewer typos than Levenshtein would, and the renderers make
/// up the difference: `audit` prints the file's full valid-name set
/// whenever a finding is about a name (`Result.hasNameFindings`), and
/// `doctor` points at the reference page, which the doc-sync test keeps
/// complete.
///
/// Picks the candidate whose LENGTH is closest to `key`'s, because one
/// real field can prefix another: `Lot` has both `price` and
/// `price_date`, so `price_dat` prefix-matches both and first-hit-wins
/// would answer `price`. A regression test walks every registered
/// model's fields and asserts each one's truncated and doubled forms
/// suggest it back.
fn prefixMatch(names: []const []const u8, key: []const u8) ?[]const u8 {
if (key.len < min_prefix_len) return null;
var best: ?[]const u8 = null;
var best_delta: usize = std.math.maxInt(usize);
for (names) |n| {
if (n.len < min_prefix_len) continue;
if (!std.mem.startsWith(u8, n, key) and !std.mem.startsWith(u8, key, n)) continue;
const delta = if (n.len > key.len) n.len - key.len else key.len - n.len;
if (delta < best_delta) {
best_delta = delta;
best = n;
}
}
return best;
}
// ── The walk ──────────────────────────────────────────────────
/// One record's raw keys, buffered so the discriminator can be located
/// before applicability is judged (it may appear after the field it
/// governs).
const RecordBuf = struct {
// SAFETY: only `keys[0..len]` is ever read, and `push` writes each
// slot before incrementing `len`.
keys: [max_fields_per_record][]const u8 = undefined,
len: usize = 0,
overflowed: bool = false,
/// Raw string value of the discriminator/tag field, when present.
disc_value: ?[]const u8 = null,
fn push(self: *RecordBuf, key: []const u8) void {
if (self.len == self.keys.len) {
self.overflowed = true;
return;
}
self.keys[self.len] = key;
self.len += 1;
}
};
/// Walk `data` as SRF and report every key that `shape` does not
/// explain. `data` must outlive the returned `Result`.
pub fn check(allocator: std.mem.Allocator, data: []const u8, shape: Shape) !Result {
var arena = std.heap.ArenaAllocator.init(allocator);
errdefer arena.deinit();
var sink: Sink = .{ .allocator = arena.allocator(), .findings = .empty };
var reader = std.Io.Reader.fixed(data);
// A file that isn't SRF at all is `doctor`'s existing parse-check's
// problem, not ours - report no findings rather than a confusing
// wall of "unknown field".
var it = srf.iterator(&reader, arena.allocator(), .{ .parse_allocator = .none }) catch {
return .{ .findings = &.{}, .shape = shape, .arena = arena };
};
defer it.deinit();
while (it.next() catch null) |fields| {
// Matches `cache/store.zig`'s diagnostics: the record's first
// line, captured before the field walk advances it.
const line: u32 = @intCast(it.state.line);
sink.line = line;
var buf: RecordBuf = .{};
const disc_name: ?[]const u8 = switch (shape) {
.flat => |f| if (f.scopes) |s| s.discriminator else null,
.tagged => |t| t.tag_key,
};
while (fields.next() catch null) |f| {
buf.push(f.key);
if (disc_name) |dn| {
if (buf.disc_value == null and std.mem.eql(u8, f.key, dn)) {
if (f.value) |v| {
if (v == .string) buf.disc_value = v.string;
}
}
}
}
try checkRecord(&sink, shape, buf);
}
const findings = try sink.findings.toOwnedSlice(arena.allocator());
return .{
.findings = findings,
.shape = shape,
.truncated = sink.truncated,
.arena = arena,
};
}
fn checkRecord(sink: *Sink, shape: Shape, buf: RecordBuf) !void {
const names: []const []const u8, const scopes: ?Scopes = switch (shape) {
.flat => |f| .{ f.names, f.scopes },
.tagged => |t| blk: {
const tag_value = buf.disc_value orelse return; // untagged record: `to()` errors on it
const variant = t.variantFor(tag_value) orelse return; // unknown tag: ditto
break :blk .{ variant.names, null };
},
};
// Which valid names have been consumed, for duplicate detection.
var seen = [_]bool{false} ** max_fields_per_record;
const disc_value: []const u8 = if (scopes) |s| (buf.disc_value orelse s.default_value) else "";
for (buf.keys[0..buf.len]) |key| {
if (indexOfName(names, key)) |idx| {
if (idx < seen.len) {
if (seen[idx]) {
try sink.addStatic(.duplicate, key, "SRF keeps the first occurrence and ignores the rest");
continue;
}
seen[idx] = true;
}
// Known field. Is it read for this record?
if (scopes) |s| {
if (buf.overflowed) continue;
const rule = s.ruleFor(key) orelse continue;
if (rule.derived) {
try sink.addStatic(.derived, key, "derived by zfin; remove it from your file");
} else if (rule.only) |vals| {
if (indexOfName(vals, disc_value) == null) {
const detail = try describeOnly(sink.allocator, s.discriminator, vals, disc_value);
try sink.addOwned(.inapplicable, key, detail);
}
}
}
continue;
}
if (caseMatch(names, key)) |n| {
try sink.addSuggestion(.case_mismatch, key, n);
} else if (prefixMatch(names, key)) |n| {
try sink.addSuggestion(.near_miss, key, n);
} else {
try sink.addStatic(.unknown, key, "");
}
}
}
/// "ignored for security_type stock; read only for cd".
fn describeOnly(
allocator: std.mem.Allocator,
discriminator: []const u8,
only: []const []const u8,
actual: []const u8,
) ![]const u8 {
var aw: std.Io.Writer.Allocating = .init(allocator);
errdefer aw.deinit();
try aw.writer.print("ignored for {s} {s}; read only for ", .{ discriminator, actual });
for (only, 0..) |v, i| {
if (i > 0) try aw.writer.writeAll(if (i + 1 == only.len) " and " else ", ");
try aw.writer.writeAll(v);
}
return aw.toOwnedSlice();
}
/// Run a schema's model-owned `semanticCheck` over `data`, appending
/// to `result`.
///
/// A SECOND typed pass, separate from `check`'s raw one, because SRF's
/// iterators are single-pass: `to()` drains the fields that the raw
/// walk needs. Records that fail to coerce are skipped silently - the
/// typed parser's own diagnostics (and `doctor`'s parse-check) already
/// report those, and duplicating them here would double every message.
pub fn checkSemantic(comptime S: type, result: *Result, data: []const u8, ctx: Context) !void {
if (!@hasDecl(S, "semanticCheck")) return;
const allocator = result.arena.allocator();
var sink: Sink = .{ .allocator = allocator, .findings = .empty };
try sink.findings.appendSlice(allocator, result.findings);
sink.truncated = result.truncated;
var reader = std.Io.Reader.fixed(data);
var it = srf.iterator(&reader, allocator, .{ .parse_allocator = .none }) catch return;
defer it.deinit();
while (it.next() catch null) |fields| {
sink.line = @intCast(it.state.line);
const rec = fields.to(S.Record, @import("srf_opts.zig").user_edited) catch continue;
try S.semanticCheck(rec, ctx, &sink);
}
result.findings = try sink.findings.toOwnedSlice(allocator);
result.truncated = sink.truncated;
}
/// Write the valid field names for `shape` to `w`, wrapped to `width`
/// columns and indented by `indent` spaces.
///
/// Renderers call this whenever `Result.hasNameFindings`. It is
/// what makes the deliberately-conservative matchers sufficient: even
/// when no suggestion can be offered, the user gets the authoritative
/// list - derived from the struct, so unlike the reference docs it
/// cannot be out of date.
pub fn writeValidNames(w: *std.Io.Writer, shape: Shape, indent: usize, width: usize) !void {
switch (shape) {
.flat => |f| try writeNameList(w, "", f.names, indent, width),
.tagged => |t| {
for (t.variants) |v| {
var label_buf: [64]u8 = undefined;
const label = std.fmt.bufPrint(&label_buf, "{s}::{s} ", .{ t.tag_key, v.tag_value }) catch "";
try writeNameList(w, label, v.names, indent, width);
}
},
}
}
fn writeNameList(
w: *std.Io.Writer,
label: []const u8,
names: []const []const u8,
indent: usize,
width: usize,
) !void {
try w.splatByteAll(' ', indent);
try w.writeAll(label);
var col = indent + label.len;
for (names) |n| {
// +1 for the separating space. Wrap before overflowing so a
// narrow terminal does not ragged-wrap mid-name.
if (col > indent and col + n.len + 1 > width) {
try w.writeAll("\n");
try w.splatByteAll(' ', indent + 2);
col = indent + 2;
}
try w.writeAll(n);
try w.writeAll(" ");
col += n.len + 1;
}
try w.writeAll("\n");
}
// ── Registry ──────────────────────────────────────────────────
/// Every user-authored SRF file zfin reads, paired with the model that
/// owns its schema. Nine one-liners, mirroring `tui.zig`'s
/// `tab_modules`.
///
/// **Adding a tenth user-authored file means adding it here.** That is
/// the one drift this design does not close at comptime - there is no
/// way to ask Zig "who references `srf_opts.user_edited`" - but it is
/// the cheap kind: a new file goes unchecked, nothing becomes wrong.
/// `grep -rn srf_opts.user_edited src/` is the authoritative index, and
/// that constant's doc comment carries the same reminder.
///
/// `history/imported_values.srf` is deliberately ABSENT. It is
/// generated by `tools/import_values.zig` from a spreadsheet export and
/// hand-editing it is explicitly disallowed (see the module doc on
/// `data/imported_values.zig`), so there are no hand-typed field names
/// to get wrong. It is also the only `user_edited` parse site with no
/// `docs/reference/config/*-srf.md` page, which independently confirms
/// the classification.
pub const schemas = .{
@import("models/portfolio.zig").srf_schema,
@import("analytics/analysis.zig").srf_schema,
@import("models/classification.zig").srf_schema,
@import("models/transaction_log.zig").srf_schema,
@import("analytics/projections.zig").srf_schema,
@import("data/Journal.zig").srf_schema,
@import("tui/keybinds.zig").srf_schema,
@import("tui/theme.zig").srf_schema,
@import("commands/common.zig").srf_schema,
};
/// Validate every registered schema at build time. Mirrors
/// `tui.zig`'s comptime sweep over `tab_modules`.
pub const validated_schemas = blk: {
for (schemas) |S| _ = shapeOfSchema(S);
break :blk true;
};
/// Number of registered schemas.
pub const schema_count = schemas.len;
// ── Tests ─────────────────────────────────────────────────────
const testing = std.testing;
// Every field of every registered model must appear in that model's
// reference page, or be listed in the schema's `undocumented`.
//
// This is the mechanism that keeps the docs from drifting the way
// `portfolio-srf.md` already had: it documented 14 of `Lot`'s 22
// fields in its table, and adding a field had no consequence. Now it
// does - this test fails until the field is documented or explicitly
// exempted, and the exemption is a visible decision in the schema.
//
// A field counts as documented if it appears anywhere outside a fenced
// code block, in either spelling the docs use: bare (`` `symbol` ``,
// reference tables) or on-wire (`` `symbol::` ``, prose). Code fences
// are excluded on purpose - an example that happens to mention a field
// is not a description of it.
test "doc sync: every model field appears in its reference page" {
// Referencing this forces the comptime sweep over `schemas`, so a
// schema with a contract violation or a non-exhaustive
// `field_rules` fails the build rather than going unvalidated.
try testing.expect(validated_schemas);
var aw: std.Io.Writer.Allocating = .init(testing.allocator);
defer aw.deinit();
var missing: usize = 0;
inline for (schemas) |S| {
const doc_file = comptime std.fs.path.basename(S.doc_path);
const doc = config_docs.find(doc_file) orelse {
std.debug.print("srf_schema '{s}': no reference page named '{s}'\n", .{ S.file_label, doc_file });
return error.MissingReferencePage;
};
switch (comptime shapeOfSchema(S)) {
.flat => |f| missing += try reportUndocumented(&aw.writer, S, doc, f.names, ""),
.tagged => |t| {
for (t.variants) |v| {
missing += try reportUndocumented(&aw.writer, S, doc, v.names, t.tag_key);
}
},
}
}
if (missing > 0) {
std.debug.print(
\\
\\{d} model field(s) are not described in their reference page:
\\{s}
\\Document each one, or add it to that schema's `undocumented`
\\list with a comment saying why it is not user-facing.
\\
, .{ missing, aw.written() });
}
try testing.expectEqual(@as(usize, 0), missing);
}
/// Write a line to `w` for each field of `S` absent from its reference
/// page, and return how many there were.
///
/// Reports through a writer rather than printing: zlint's `no-print`
/// rule exempts `test` blocks but not the helpers they call, and
/// funnelling the text back to the one caller is both cleaner and
/// keeps the diagnostic in a single flush.
fn reportUndocumented(
w: *std.Io.Writer,
comptime S: type,
doc: config_docs.Doc,
names: []const []const u8,
tag_key: []const u8,
) !usize {
const exempt: []const []const u8 = if (@hasDecl(S, "undocumented")) &S.undocumented else &.{};
var n: usize = 0;
for (names) |name| {
// The union tag key is SRF machinery, not a model field.
if (tag_key.len > 0 and std.mem.eql(u8, name, tag_key)) continue;
if (indexOfName(doc.names, name) != null) continue;
if (indexOfName(exempt, name) != null) continue;
try w.print(" {s}: field '{s}' is undocumented in {s}\n", .{ S.file_label, name, doc.file });
n += 1;
}
return n;
}
test "registry: every schema has a distinct file label and doc page" {
try testing.expect(schema_count == 9);
inline for (schemas, 0..) |A, i| {
inline for (schemas, 0..) |B, j| {
if (comptime i >= j) continue;
try testing.expect(!std.mem.eql(u8, A.file_label, B.file_label));
try testing.expect(!std.mem.eql(u8, A.doc_path, B.doc_path));
}
}
}
test "registry: every schema builds a usable shape and lints a clean empty file" {
inline for (schemas) |S| {
var r = try check(testing.allocator, "#!srfv1\n", comptime shapeOfSchema(S));
defer r.deinit();
try testing.expect(r.isClean());
}
}
test "registry: a one-character typo of any real field suggests that field back" {
// The property that matters to a user, checked against the REAL
// models rather than a fixture: drop the last character of a field
// name, or double it, and the lint must point at the field you
// meant. `Lot` alone has `price`, `price_date` and `price_ratio`,
// so this is where a naive first-hit-wins matcher goes wrong.
var aw: std.Io.Writer.Allocating = .init(testing.allocator);
defer aw.deinit();
var bad: usize = 0;
inline for (schemas) |S| {
switch (comptime shapeOfSchema(S)) {
.flat => |f| bad += try reportBadSuggestions(&aw.writer, S.file_label, f.names),
.tagged => |t| {
for (t.variants) |v| bad += try reportBadSuggestions(&aw.writer, S.file_label, v.names);
},
}
}
if (bad > 0) std.debug.print("\n{s}", .{aw.written()});
try testing.expectEqual(@as(usize, 0), bad);
}
/// Write a line to `w` for each field whose one-character typo forms
/// resolve to the wrong suggestion, and return how many there were.
fn reportBadSuggestions(w: *std.Io.Writer, label: []const u8, names: []const []const u8) !usize {
var buf: [128]u8 = undefined;
var bad: usize = 0;
for (names) |name| {
if (name.len < min_prefix_len + 1) continue;
// Dropped trailing character. If the truncation IS another real
// field, an exact match wins and no suggestion is wanted.
const truncated = name[0 .. name.len - 1];
if (indexOfName(names, truncated) == null) {
const got = prefixMatch(names, truncated);
if (got == null or !std.mem.eql(u8, got.?, name)) {
try w.print(" {s}: '{s}' (from '{s}') suggested {?s}, want '{s}'\n", .{ label, truncated, name, got, name });
bad += 1;
}
}
// Doubled trailing character.
@memcpy(buf[0..name.len], name);
buf[name.len] = name[name.len - 1];
const doubled = buf[0 .. name.len + 1];
if (indexOfName(names, doubled) == null) {
const got = prefixMatch(names, doubled);
if (got == null or !std.mem.eql(u8, got.?, name)) {
try w.print(" {s}: '{s}' (from '{s}') suggested {?s}, want '{s}'\n", .{ label, doubled, name, got, name });
bad += 1;
}
}
}
return bad;
}
// ── Fixtures ──────────────────────────────────────────────────
const TestLotType = enum { stock, option, cd, cash };
const TestLot = struct {
symbol: []const u8 = "",
shares: f64,
account: ?[]const u8 = null,
security_type: TestLotType = .stock,
rate: ?f64 = null,
strike: ?f64 = null,
split_factor: f64 = 1.0,
};
const test_lot_schema = struct {
pub const Record = TestLot;
pub const file_label: []const u8 = "test_lot.srf";
pub const doc_path: []const u8 = "reference/config/test-lot-srf.md";
pub const scope_discriminator: []const u8 = "security_type";
pub const field_rules = [_]Rule{
.{ .name = "symbol" },
.{ .name = "shares" },
.{ .name = "account" },
.{ .name = "security_type" },
.{ .name = "rate", .only = &.{"cd"} },
.{ .name = "strike", .only = &.{"option"} },
.{ .name = "split_factor", .derived = true },
};
};
const TestUnion = union(enum) {
config: struct { type: []const u8 = "", horizon: u16 = 0 },
birthdate: struct { date: []const u8 = "", person: u8 = 1 },
};
const test_union_schema = struct {
pub const Record = TestUnion;
pub const file_label: []const u8 = "test_union.srf";
pub const doc_path: []const u8 = "reference/config/test-union-srf.md";
};
fn lintLot(data: []const u8) !Result {
return check(testing.allocator, data, shapeOfSchema(test_lot_schema));
}
fn findingFor(r: Result, key: []const u8) ?Finding {
for (r.findings) |f| {
if (std.mem.eql(u8, f.key, key)) return f;
}
return null;
}
// ── check: the raw walk ───────────────────────────────────────
test "check: clean file produces no findings" {
const data =
\\#!srfv1
\\symbol::VTI,shares:num:100,account::Sample Brokerage
\\symbol::SPY,shares:num:50,account::Sample IRA
\\
;
var r = try lintLot(data);
defer r.deinit();
try testing.expect(r.isClean());
}
test "check: unknown key with no relative" {
const data =
\\#!srfv1
\\symbol::VTI,shares:num:100,cost_basis:num:1000
\\
;
var r = try lintLot(data);
defer r.deinit();
try testing.expectEqual(@as(usize, 1), r.findings.len);
try testing.expectEqual(Kind.unknown, r.findings[0].kind);
try testing.expectEqualStrings("cost_basis", r.findings[0].key);
try testing.expectEqual(@as(?[]const u8, null), r.findings[0].suggestion);
}
test "check: case-only mismatch names the real field" {
const data =
\\#!srfv1
\\Symbol::VTI,shares:num:100
\\
;
var r = try lintLot(data);
defer r.deinit();
const f = findingFor(r, "Symbol") orelse return error.MissingFinding;
try testing.expectEqual(Kind.case_mismatch, f.kind);
try testing.expectEqualStrings("symbol", f.suggestion.?);
}
test "check: near miss on a dropped trailing character" {
const data =
\\#!srfv1
\\symbol::VTI,shares:num:100,accoun::Sample IRA
\\
;
var r = try lintLot(data);
defer r.deinit();
const f = findingFor(r, "accoun") orelse return error.MissingFinding;
try testing.expectEqual(Kind.near_miss, f.kind);
try testing.expectEqualStrings("account", f.suggestion.?);
}
test "check: near miss on a doubled trailing character" {
const data =
\\#!srfv1
\\symbol::VTI,shares:num:100,accountt::Sample IRA
\\
;
var r = try lintLot(data);
defer r.deinit();
const f = findingFor(r, "accountt") orelse return error.MissingFinding;
try testing.expectEqual(Kind.near_miss, f.kind);
try testing.expectEqualStrings("account", f.suggestion.?);
}
test "check: prefix matching ignores keys below the length floor" {
// `sy` is a prefix of `symbol` but too short to suggest against.
const data =
\\#!srfv1
\\shares:num:100,sy::VTI
\\
;
var r = try lintLot(data);
defer r.deinit();
const f = findingFor(r, "sy") orelse return error.MissingFinding;
try testing.expectEqual(Kind.unknown, f.kind);
}
test "check: duplicate key in one record" {
const data =
\\#!srfv1
\\symbol::VTI,shares:num:100,shares:num:200
\\
;
var r = try lintLot(data);
defer r.deinit();
const f = findingFor(r, "shares") orelse return error.MissingFinding;
try testing.expectEqual(Kind.duplicate, f.kind);
}
test "check: inapplicable field for the record's discriminator" {
const data =
\\#!srfv1
\\symbol::VTI,shares:num:100,rate:num:5.25
\\
;
var r = try lintLot(data);
defer r.deinit();
const f = findingFor(r, "rate") orelse return error.MissingFinding;
try testing.expectEqual(Kind.inapplicable, f.kind);
// Default discriminator value is derived from the struct, so an
// absent `security_type` still reports as `stock`.
try testing.expect(std.mem.indexOf(u8, f.detail, "security_type stock") != null);
try testing.expect(std.mem.indexOf(u8, f.detail, "cd") != null);
}
test "check: applicable field for the right discriminator is clean" {
const data =
\\#!srfv1
\\symbol::CD1,shares:num:1000,security_type::cd,rate:num:5.25
\\
;
var r = try lintLot(data);
defer r.deinit();
try testing.expect(r.isClean());
}
test "check: discriminator is honored even when it appears after the field" {
// `strike` precedes `security_type`, so the record must be
// buffered before applicability is judged.
const data =
\\#!srfv1
\\symbol::AMZN,shares:num:1,strike:num:200,security_type::option
\\
;
var r = try lintLot(data);
defer r.deinit();
try testing.expect(r.isClean());
}
test "check: derived field in a hand-edited file" {
const data =
\\#!srfv1
\\symbol::VTI,shares:num:100,split_factor:num:4
\\
;
var r = try lintLot(data);
defer r.deinit();
const f = findingFor(r, "split_factor") orelse return error.MissingFinding;
try testing.expectEqual(Kind.derived, f.kind);
}
test "check: repeated typo rolls up instead of repeating" {
const data =
\\#!srfv1
\\symbol::A,shares:num:1,accoun::X
\\symbol::B,shares:num:2,accoun::X
\\symbol::C,shares:num:3,accoun::X
\\symbol::D,shares:num:4,accoun::X
\\
;
var r = try lintLot(data);
defer r.deinit();
try testing.expectEqual(@as(usize, 1), r.findings.len);
const f = r.findings[0];
try testing.expectEqual(@as(u32, 4), f.count);
try testing.expectEqual(@as(u32, 2), f.first_line);
// Two more lines retained for the "lines 2, 3, 4, ..." tail.
try testing.expectEqual(@as(u8, 2), f.extra_len);
try testing.expectEqual(@as(u32, 3), f.extra_lines[0]);
try testing.expectEqual(@as(u32, 4), f.extra_lines[1]);
}
test "check: line numbers point at the offending record" {
const data =
\\#!srfv1
\\symbol::A,shares:num:1
\\symbol::B,shares:num:2
\\symbol::C,shares:num:3,bogus::x
\\
;
var r = try lintLot(data);
defer r.deinit();
const f = findingFor(r, "bogus") orelse return error.MissingFinding;
try testing.expectEqual(@as(u32, 4), f.first_line);
}
test "check: non-SRF input reports nothing rather than a wall of unknowns" {
var r = try lintLot("this is not an srf file at all\n");
defer r.deinit();
try testing.expect(r.isClean());
}
test "check: empty file is clean" {
var r = try lintLot("#!srfv1\n");
defer r.deinit();
try testing.expect(r.isClean());
}
// ── Tagged unions ─────────────────────────────────────────────
test "shapeOfSchema: tagged union dispatches on the tag value" {
const shape = shapeOfSchema(test_union_schema);
try testing.expectEqualStrings("type", shape.tagged.tag_key);
try testing.expectEqual(@as(usize, 2), shape.tagged.variants.len);
const data =
\\#!srfv1
\\type::config,horizon:num:30
\\type::birthdate,date::1980-01-01,person:num:1
\\
;
var r = try check(testing.allocator, data, shape);
defer r.deinit();
try testing.expect(r.isClean());
}
test "shapeOfSchema: tag key is valid whether or not the variant redeclares it" {
// `config` redeclares `type`; `birthdate` does not. Both must
// accept `type::` without reporting it as unknown.
const shape = shapeOfSchema(test_union_schema);
for (shape.tagged.variants) |v| {
try testing.expect(indexOfName(v.names, "type") != null);
}
// And it appears exactly once, not twice, for the redeclaring one.
const cfg = shape.tagged.variantFor("config").?;
var type_count: usize = 0;
for (cfg.names) |n| {
if (std.mem.eql(u8, n, "type")) type_count += 1;
}
try testing.expectEqual(@as(usize, 1), type_count);
}
test "check: typo inside a union variant is caught against that variant" {
const data =
\\#!srfv1
\\type::config,horizonn:num:30
\\
;
var r = try check(testing.allocator, data, shapeOfSchema(test_union_schema));
defer r.deinit();
const f = findingFor(r, "horizonn") orelse return error.MissingFinding;
try testing.expectEqual(Kind.near_miss, f.kind);
try testing.expectEqualStrings("horizon", f.suggestion.?);
}
test "check: a field valid on another variant is not valid on this one" {
const data =
\\#!srfv1
\\type::config,person:num:2
\\
;
var r = try check(testing.allocator, data, shapeOfSchema(test_union_schema));
defer r.deinit();
const f = findingFor(r, "person") orelse return error.MissingFinding;
try testing.expectEqual(Kind.unknown, f.kind);
}
test "check: unknown tag value is left to the typed parser" {
const data =
\\#!srfv1
\\type::nonsense,whatever::x
\\
;
var r = try check(testing.allocator, data, shapeOfSchema(test_union_schema));
defer r.deinit();
try testing.expect(r.isClean());
}
// ── Derivation ────────────────────────────────────────────────
test "namesOf: derives the full field set" {
const names = namesOf(TestLot);
try testing.expectEqual(@as(usize, 7), names.len);
try testing.expect(indexOfName(names, "symbol") != null);
try testing.expect(indexOfName(names, "split_factor") != null);
try testing.expect(indexOfName(names, "nope") == null);
}
test "scopesOfSchema: default_value is derived from the struct default" {
const shape = shapeOfSchema(test_lot_schema);
try testing.expectEqualStrings("stock", shape.flat.scopes.?.default_value);
try testing.expectEqualStrings("security_type", shape.flat.scopes.?.discriminator);
}
test "prefixMatch: picks the closest-length candidate, not the first" {
// `Lot`'s real shape: a short field that prefixes two longer ones.
const names: []const []const u8 = &.{ "price", "price_date", "price_ratio" };
try testing.expectEqualStrings("price_date", prefixMatch(names, "price_dat").?);
try testing.expectEqualStrings("price_date", prefixMatch(names, "price_datee").?);
try testing.expectEqualStrings("price_ratio", prefixMatch(names, "price_rati").?);
try testing.expectEqualStrings("price", prefixMatch(names, "pricee").?);
// Below the floor, no guess at all.
try testing.expectEqual(@as(?[]const u8, null), prefixMatch(names, "pr"));
}
test "check: findings are capped so a garbage file cannot flood the report" {
var aw: std.Io.Writer.Allocating = .init(testing.allocator);
defer aw.deinit();
try aw.writer.writeAll("#!srfv1\n");
// Each record carries a DISTINCT bogus key, so rollup cannot
// collapse them and the cap is what bounds the output.
for (0..max_findings + 50) |i| {
try aw.writer.print("shares:num:1,zz{d}::x\n", .{i});
}
var r = try lintLot(aw.writer.buffered());
defer r.deinit();
try testing.expectEqual(@as(usize, max_findings), r.findings.len);
try testing.expect(r.truncated);
}
test "Sink.addOwned detail is freed with the result" {
// Exercises the arena-ownership contract: `describeOnly`
// allocates, and `std.testing.allocator` fails the test if
// `Result.deinit` does not release it.
const data =
\\#!srfv1
\\symbol::VTI,shares:num:100,rate:num:1,strike:num:2
\\
;
var r = try lintLot(data);
defer r.deinit();
try testing.expectEqual(@as(usize, 2), r.findings.len);
}
// ── describe / writeValidNames ────────────────────────────────
fn describeOne(data: []const u8, buf: []u8) ![]const u8 {
var r = try lintLot(data);
defer r.deinit();
if (r.findings.len == 0) return error.NoFinding;
// `describe` writes into `buf`, which outlives `r`, but `key`
// borrows from `data` - so this only holds while `data` is alive.
return r.findings[0].describe(buf);
}
test "describe: unknown field" {
var buf: [Finding.describe_max]u8 = undefined;
const msg = try describeOne("#!srfv1\nshares:num:1,cost_basis:num:5\n", &buf);
try testing.expectEqualStrings("unrecognized field 'cost_basis'", msg);
}
test "describe: near miss names the field it meant" {
var buf: [Finding.describe_max]u8 = undefined;
const msg = try describeOne("#!srfv1\nshares:num:1,accoun::X\n", &buf);
try testing.expectEqualStrings("unrecognized field 'accoun' - did you mean 'account'?", msg);
}
test "describe: case mismatch explains why it silently did nothing" {
var buf: [Finding.describe_max]u8 = undefined;
const msg = try describeOne("#!srfv1\nshares:num:1,Account::X\n", &buf);
try testing.expect(std.mem.indexOf(u8, msg, "differs only by case from 'account'") != null);
try testing.expect(std.mem.indexOf(u8, msg, "case-sensitive") != null);
}
test "describe: duplicate explains that the first wins" {
var buf: [Finding.describe_max]u8 = undefined;
const msg = try describeOne("#!srfv1\nshares:num:1,shares:num:2\n", &buf);
try testing.expect(std.mem.indexOf(u8, msg, "appears twice") != null);
try testing.expect(std.mem.indexOf(u8, msg, "keeps the first") != null);
}
test "describe: inapplicable names the discriminator and the types that read it" {
var buf: [Finding.describe_max]u8 = undefined;
const msg = try describeOne("#!srfv1\nshares:num:1,rate:num:5\n", &buf);
try testing.expectEqualStrings(
"field 'rate' ignored for security_type stock; read only for cd",
msg,
);
}
test "describe: derived field says to remove it" {
var buf: [Finding.describe_max]u8 = undefined;
const msg = try describeOne("#!srfv1\nshares:num:1,split_factor:num:4\n", &buf);
try testing.expectEqualStrings("field 'split_factor' is derived by zfin; remove it from your file", msg);
}
test "describe: rolled-up repeat reports the count and the first lines" {
const data =
\\#!srfv1
\\shares:num:1,accoun::X
\\shares:num:2,accoun::X
\\shares:num:3,accoun::X
\\shares:num:4,accoun::X
\\
;
var buf: [Finding.describe_max]u8 = undefined;
const msg = try describeOne(data, &buf);
try testing.expect(std.mem.indexOf(u8, msg, "(4 records: lines 2, 3, 4, ...)") != null);
}
test "describe: exactly three occurrences omits the ellipsis" {
const data =
\\#!srfv1
\\shares:num:1,accoun::X
\\shares:num:2,accoun::X
\\shares:num:3,accoun::X
\\
;
var buf: [Finding.describe_max]u8 = undefined;
const msg = try describeOne(data, &buf);
try testing.expect(std.mem.indexOf(u8, msg, "(3 records: lines 2, 3, 4)") != null);
try testing.expect(std.mem.indexOf(u8, msg, "...") == null);
}
test "describe: a short buffer truncates rather than failing" {
var r = try lintLot("#!srfv1\nshares:num:1,cost_basis:num:5\n");
defer r.deinit();
var tiny: [8]u8 = undefined;
const msg = r.findings[0].describe(&tiny);
try testing.expect(msg.len <= tiny.len);
try testing.expectEqualStrings("unrecogn", msg);
}
test "writeValidNames: flat shape lists every field, wrapped" {
var aw: std.Io.Writer.Allocating = .init(testing.allocator);
defer aw.deinit();
try writeValidNames(&aw.writer, shapeOfSchema(test_lot_schema), 4, 40);
const out = aw.written();
// Every field present, including the derived one - the list is
// "what SRF will match", not "what you should write".
for (namesOf(TestLot)) |n| {
try testing.expect(std.mem.indexOf(u8, out, n) != null);
}
// Wrapped: more than one line, none wildly over the limit.
var it = std.mem.splitScalar(u8, std.mem.trimEnd(u8, out, "\n"), '\n');
var lines: usize = 0;
while (it.next()) |line| {
lines += 1;
try testing.expect(line.len <= 44);
}
try testing.expect(lines > 1);
}
test "writeValidNames: tagged shape lists each variant separately" {
var aw: std.Io.Writer.Allocating = .init(testing.allocator);
defer aw.deinit();
try writeValidNames(&aw.writer, shapeOfSchema(test_union_schema), 2, 100);
const out = aw.written();
try testing.expect(std.mem.indexOf(u8, out, "type::config") != null);
try testing.expect(std.mem.indexOf(u8, out, "type::birthdate") != null);
try testing.expect(std.mem.indexOf(u8, out, "horizon") != null);
try testing.expect(std.mem.indexOf(u8, out, "person") != null);
}
// ── checkSemantic / Result.sort ────────────────────────────────
/// A schema whose `semanticCheck` reports a model-specific rule, used to
/// exercise the typed second pass without depending on any real model's
/// current rule set.
const SemRecord = struct {
name: []const u8 = "",
pct: f64 = 0,
};
/// Fixed `today` for tests, so a date rule's verdict cannot change as
/// the calendar moves.
const test_ctx: Context = .{ .today = Date.fromYmd(2026, 1, 1) };
const sem_schema = struct {
pub const Record = SemRecord;
pub const file_label: []const u8 = "sem.srf";
pub const doc_path: []const u8 = "reference/config/sem-srf.md";
pub fn semanticCheck(rec: Record, ctx: Context, sink: *Sink) !void {
_ = ctx;
if (rec.pct > 100) {
try sink.addOwned(.semantic, "pct", try std.fmt.allocPrint(
sink.allocator,
"'{s}': pct must be <= 100 (got {d})",
.{ rec.name, rec.pct },
));
}
}
};
test "checkSemantic: model-owned rules reach the findings list" {
const data =
\\#!srfv1
\\name::ok,pct:num:50
\\name::bad,pct:num:150
\\
;
var r = try check(testing.allocator, data, shapeOfSchema(sem_schema));
defer r.deinit();
// Field names are all valid, so the raw pass finds nothing.
try testing.expect(r.isClean());
try checkSemantic(sem_schema, &r, data, test_ctx);
try testing.expectEqual(@as(usize, 1), r.findings.len);
try testing.expectEqual(Kind.semantic, r.findings[0].kind);
try testing.expectEqual(@as(u32, 3), r.findings[0].first_line);
var buf: [Finding.describe_max]u8 = undefined;
try testing.expectEqualStrings("'bad': pct must be <= 100 (got 150)", r.findings[0].describe(&buf));
}
test "checkSemantic: a schema without the hook is a no-op" {
const data = "#!srfv1\nsymbol::VTI,shares:num:1,account::Sample Brokerage\n";
var r = try check(testing.allocator, data, shapeOfSchema(test_lot_schema));
defer r.deinit();
try checkSemantic(test_lot_schema, &r, data, test_ctx);
try testing.expect(r.isClean());
}
test "checkSemantic: records that fail to coerce are skipped, not reported twice" {
// `pct` is a string where a number is declared; under
// `user_edited` coercion that is tolerated, so the record still
// reaches `semanticCheck`. A record missing a required field would
// be skipped - the typed parser's own diagnostics already cover it.
const data =
\\#!srfv1
\\name::bad,pct::150
\\
;
var r = try check(testing.allocator, data, shapeOfSchema(sem_schema));
defer r.deinit();
try checkSemantic(sem_schema, &r, data, test_ctx);
try testing.expectEqual(@as(usize, 1), r.findings.len);
}
test "Result.sort: orders by line so the two passes interleave correctly" {
// Raw-pass finding on line 4, semantic-pass findings on 2 and 3 -
// the order they are produced in is not the order to read them in.
const data =
\\#!srfv1
\\name::a,pct:num:150
\\name::b,pct:num:200
\\name::c,pct:num:1,bogus::x
\\
;
var r = try check(testing.allocator, data, shapeOfSchema(sem_schema));
defer r.deinit();
try checkSemantic(sem_schema, &r, data, test_ctx);
try testing.expectEqual(@as(usize, 3), r.findings.len);
// Unsorted, the raw finding comes first.
try testing.expectEqual(@as(u32, 4), r.findings[0].first_line);
r.sort();
try testing.expectEqual(@as(u32, 2), r.findings[0].first_line);
try testing.expectEqual(@as(u32, 3), r.findings[1].first_line);
try testing.expectEqual(@as(u32, 4), r.findings[2].first_line);
}
test "rollup: semantic findings on different records stay separate" {
// Regression guard. Identity used to be `(kind, key)` alone, which
// merged these two into one line naming only 'a' - silently hiding
// that 'b' was broken too. Both records trip the same field with a
// different detail, so both must survive.
const data =
\\#!srfv1
\\name::a,pct:num:150
\\name::b,pct:num:200
\\
;
var r = try check(testing.allocator, data, shapeOfSchema(sem_schema));
defer r.deinit();
try checkSemantic(sem_schema, &r, data, test_ctx);
try testing.expectEqual(@as(usize, 2), r.findings.len);
var buf: [Finding.describe_max]u8 = undefined;
r.sort();
try testing.expect(std.mem.indexOf(u8, r.findings[0].describe(&buf), "'a'") != null);
try testing.expect(std.mem.indexOf(u8, r.findings[1].describe(&buf), "'b'") != null);
}
test "rollup: identical generic findings still collapse to one line" {
// The other half of the same property: a static per-kind detail
// means the same typo across many records is ONE finding, so a
// copy-pasted mistake does not produce a wall of output.
const data =
\\#!srfv1
\\shares:num:1,accoun::X
\\shares:num:2,accoun::X
\\shares:num:3,accoun::X
\\
;
var r = try lintLot(data);
defer r.deinit();
try testing.expectEqual(@as(usize, 1), r.findings.len);
try testing.expectEqual(@as(u32, 3), r.findings[0].count);
}
test "check: a record wider than the buffer is still name-checked" {
// The `max_fields_per_record` guard. Applicability needs the whole
// record buffered (the discriminator may come last), so an
// over-wide record stops contributing `inapplicable` findings - but
// it must still get its names checked, and must not crash.
var aw: std.Io.Writer.Allocating = .init(testing.allocator);
defer aw.deinit();
try aw.writer.writeAll("#!srfv1\nshares:num:1");
for (0..max_fields_per_record + 10) |i| {
try aw.writer.print(",zz{d}::x", .{i});
}
try aw.writer.writeAll("\n");
var r = try lintLot(aw.writer.buffered());
defer r.deinit();
// Bounded by the buffer, so not every bogus key is reported - but
// the ones that fit are, and nothing panicked.
try testing.expect(r.findings.len > 0);
try testing.expect(r.findings.len <= max_fields_per_record);
}
test "Result.sort: two findings on one line are ordered by key" {
// The tiebreaker. Without it the order of same-line findings
// depends on field declaration order, which makes report diffs
// noisy for no reason.
const data =
\\#!srfv1
\\shares:num:1,zebra::x,alpha::y
\\
;
var r = try lintLot(data);
defer r.deinit();
try testing.expectEqual(@as(usize, 2), r.findings.len);
r.sort();
try testing.expectEqualStrings("alpha", r.findings[0].key);
try testing.expectEqualStrings("zebra", r.findings[1].key);
}
// ── Context ───────────────────────────────────────────────────
const DatedRecord = struct {
name: []const u8 = "",
on: ?Date = null,
};
/// Flags any `on` date after `ctx.today`. Exercises the one thing the
/// real date rules depend on: that `checkSemantic` delivers the caller's
/// `today`, not some other clock.
const dated_schema = struct {
pub const Record = DatedRecord;
pub const file_label: []const u8 = "dated.srf";
pub const doc_path: []const u8 = "reference/config/dated-srf.md";
pub fn semanticCheck(rec: Record, ctx: Context, sink: *Sink) !void {
const on = rec.on orelse return;
if (ctx.today.lessThan(on)) try sink.addStatic(.semantic, "on", "in the future");
}
};
test "checkSemantic: ctx.today reaches the hook" {
const data =
\\#!srfv1
\\name::a,on::2026-06-01
\\
;
// Same record, two different `today`s, opposite verdicts - so the
// hook must be reading the context rather than a clock of its own.
{
var r = try check(testing.allocator, data, shapeOfSchema(dated_schema));
defer r.deinit();
try checkSemantic(dated_schema, &r, data, .{ .today = Date.fromYmd(2026, 1, 1) });
try testing.expectEqual(@as(usize, 1), r.findings.len);
}
{
var r = try check(testing.allocator, data, shapeOfSchema(dated_schema));
defer r.deinit();
try checkSemantic(dated_schema, &r, data, .{ .today = Date.fromYmd(2027, 1, 1) });
try testing.expect(r.isClean());
}
}
test "Result.hasNameFindings: only name problems warrant the valid-field list" {
{
var r = try lintLot("#!srfv1\nshares:num:1,accoun::X\n");
defer r.deinit();
try testing.expect(r.hasNameFindings());
}
{
var r = try lintLot("#!srfv1\nshares:num:1,Account::X\n");
defer r.deinit();
try testing.expect(r.hasNameFindings());
}
{
// Real fields, wrong use - the list would tell the user nothing.
var r = try lintLot("#!srfv1\nshares:num:1,shares:num:2,rate:num:1,split_factor:num:2\n");
defer r.deinit();
try testing.expect(!r.isClean());
try testing.expect(!r.hasNameFindings());
}
}