//! SRF schema lint - catch hand-edit typos in user-authored SRF files. //! //! ## The problem //! //! SRF's `fields.to(T)` silently discards any record key that doesn't //! name a field of `T`. The relevant loop is `srf.zig`'s: //! //! ```zig //! while (try self.next()) |f| { //! var field_match = false; // declared here... //! inline for (std.meta.fields(T)) |type_field| { //! if (!field_match and std.mem.eql(u8, type_field.name, f.key) ...) { //! field_match = true; // ...set here... //! } //! } //! } // ...and never read. //! ``` //! //! `field_match` is write-only, so a typo'd key is byte-for-byte //! equivalent to omitting the field. There is no strict mode, no //! unknown-field callback, and no `error.UnknownField` anywhere in the //! library. A field WITHOUT a default errors as //! `FieldNotFoundOnFieldWithoutDefaultValue` (which names the missing //! field, not the wrong key); a field WITH a default is silently //! skipped. Most fields on zfin's user-authored models have defaults: //! 19 of `Lot`'s 22, 14 of `AccountTaxEntry`'s 16, and all 17 of //! `SrfConfig`'s - the last behind an infallible parser, so a typo //! there changes projected retirement math with zero output. //! //! ## Division of responsibility //! //! This module is an AGGREGATION POINT, not a rule book. It owns the //! raw-record walker, the name matchers, rollup, and the comptime //! contract validator. It knows nothing about any particular model. //! //! Each model module owns its own schema and declares it as //! `pub const srf_schema` (see `validateSchema` for the contract, and //! `models/portfolio.zig` for the reference implementation). That //! placement is deliberate: rules that live away from the fields they //! describe drift from them, which is exactly how //! `docs/reference/config/portfolio-srf.md` came to document 14 of //! `Lot`'s 22 fields. Three mechanisms keep a schema honest: //! //! 1. The valid-name set is DERIVED (`std.meta.fields(Record)`), //! never declared, so it cannot drift. //! 2. `field_rules` is checked for exhaustiveness over //! `std.meta.fields(Record)` in BOTH directions at comptime - a //! new field fails the build until classified, and a renamed //! field fails the build too. //! 3. `doc_path` + `undocumented` drive a doc-sync test, so adding //! a field without documenting it fails `zig build test`. //! //! ## Why a separate pass //! //! SRF's `RecordIterator`/`FieldIterator` are single-pass and //! consuming - no peek, no rewind - and `to()` drains the field //! iterator. Raw inspection and `to()` are therefore mutually //! exclusive on the same record, so this is a second walk over the //! same bytes rather than a change to any existing parse path. It runs //! only from `zfin doctor` and `zfin audit`, never on the hot path. //! With `.parse_allocator = .none` the keys borrow from the input //! buffer, so a clean file allocates nothing but the result arena. const std = @import("std"); const srf = @import("srf"); const comptime_validator = @import("comptime_validator.zig"); const Date = @import("Date.zig"); /// Backticked identifiers found in each `docs/reference/config/*.md` /// page, extracted at build time by `build/gen_config_docs.zig`. /// `@embedFile` cannot reach outside `src/`, hence the generator. const config_docs = @import("config_docs"); /// Longest record this lints in full. Every zfin model is far smaller /// (`Lot`, the widest, has 22 fields), so the cap only bites on a /// malformed file; records beyond it are still name-checked, they just /// stop contributing conditional-applicability findings (which need /// the whole record buffered to locate the discriminator). const max_fields_per_record = 64; /// Upper bound on DISTINCT findings, after rollup. A file that trips /// more than this is misconfigured in a way a longer list won't /// clarify, and the renderers have to print whatever we return. const max_findings = 200; /// Shortest name length eligible for prefix matching. Below this the /// suggestion is noise: `pc` would "match" `pct`. /// /// Three is safe for every model in the registry - no valid field name /// is a 3-character prefix of another name in the same record's set /// (`tax_type` and `tax_mix_*` share `tax_` but neither prefixes the /// other). A test pins that property so a future field addition can't /// quietly break it. const min_prefix_len = 3; // ── Findings ────────────────────────────────────────────────── pub const Kind = enum { /// Key matches no field, and no matcher found a near relative. unknown, /// Key differs from a real field only by case. SRF matches field /// names with `std.mem.eql`, so this silently does nothing. case_mismatch, /// Key is a prefix of a real field or vice versa - the shape of a /// dropped or doubled character. near_miss, /// Key appears twice in one record. SRF keeps the FIRST and /// silently drops the rest, so editing the second does nothing. duplicate, /// Real field, but never read for this record's discriminator /// value (e.g. `rate` on a `security_type::stock` lot). inapplicable, /// Real field that zfin derives and never reads from a /// hand-edited file (e.g. `Lot.split_factor`). derived, /// Model-specific rule, reported by the schema's `semanticCheck`. semantic, }; /// One rolled-up problem. Identity is `(kind, key)`; repeats across /// records bump `count` and record up to two more line numbers rather /// than emitting another entry - one typo in a copy-pasted template /// must not produce 200 lines of output. pub const Finding = struct { kind: Kind, /// The offending field name. Borrows from the `data` passed to /// `check`, so `data` must outlive the `Result`. key: []const u8, /// The real field name this probably meant. Set for /// `case_mismatch` and `near_miss`. Static (a comptime field /// name), never owned. suggestion: ?[]const u8 = null, /// Free-text explanation. Static or owned by the `Result` arena. detail: []const u8 = "", first_line: u32, count: u32 = 1, /// Second and third line this appeared on, for a "lines 12, 19, /// 26, ..." tail. Only `extra_len` entries are meaningful. extra_lines: [2]u32 = .{ 0, 0 }, extra_len: u8 = 0, fn noteRepeat(self: *Finding, line: u32) void { self.count += 1; if (self.extra_len < self.extra_lines.len) { self.extra_lines[self.extra_len] = line; self.extra_len += 1; } } /// Longest string `describe` can produce. Field names are bounded /// by Zig identifier length in practice, and `detail` by /// `describeOnly`'s output over one discriminator enum. pub const describe_max = 320; /// One-line human description, written into `buf` and returned as a /// slice of it. Shared by `zfin doctor` and `zfin audit` so the two /// surfaces cannot word the same finding differently. /// /// `buf` should be at least `describe_max` bytes; a shorter buffer /// truncates rather than failing, because a clipped diagnostic is /// still more useful than none. pub fn describe(self: Finding, buf: []u8) []const u8 { var w = std.Io.Writer.fixed(buf); self.write(&w) catch return buf[0..w.end]; return buf[0..w.end]; } fn write(self: Finding, w: *std.Io.Writer) !void { switch (self.kind) { .unknown => try w.print("unrecognized field '{s}'", .{self.key}), .case_mismatch => try w.print( "field '{s}' differs only by case from '{s}' - SRF field names are case-sensitive", .{ self.key, self.suggestion orelse "" }, ), .near_miss => try w.print( "unrecognized field '{s}' - did you mean '{s}'?", .{ self.key, self.suggestion orelse "" }, ), .duplicate => try w.print("field '{s}' appears twice in one record; {s}", .{ self.key, self.detail }), .inapplicable => try w.print("field '{s}' {s}", .{ self.key, self.detail }), .derived => try w.print("field '{s}' is {s}", .{ self.key, self.detail }), .semantic => try w.writeAll(self.detail), } if (self.count > 1) { try w.print(" ({d} records: lines {d}", .{ self.count, self.first_line }); for (self.extra_lines[0..self.extra_len]) |l| try w.print(", {d}", .{l}); try w.writeAll(if (self.count > 1 + self.extra_len) ", ...)" else ")"); } } }; /// Findings for one file, plus the shape they were checked against so /// a renderer can print the valid-name footer. /// /// Owns an arena for any allocated `detail` strings. `Finding.key` /// borrows from the `data` slice passed to `check`. pub const Result = struct { findings: []const Finding, shape: Shape, /// True when `max_findings` was hit and some were dropped. truncated: bool = false, arena: std.heap.ArenaAllocator, pub fn deinit(self: *Result) void { self.arena.deinit(); } pub fn isClean(self: Result) bool { return self.findings.len == 0; } /// True when some finding is about a field NAME (unknown, wrong /// case, near miss) - the cases where printing the valid-name list /// helps. A report of only lifecycle or duplicate findings names /// real fields already, and the list would just be noise. pub fn hasNameFindings(self: Result) bool { for (self.findings) |f| { switch (f.kind) { .unknown, .case_mismatch, .near_miss => return true, .duplicate, .inapplicable, .derived, .semantic => {}, } } return false; } /// Order findings by the line they were first seen on, then by key. /// /// Needed because `checkSemantic` appends a whole second pass after /// `check`'s, so an unsorted report interleaves line 4 before line /// 2 and reads like a bug. Call after the last pass that appends. pub fn sort(self: *Result) void { // `findings` is const to callers but owned by our arena, so the // cast is sound - nothing else can alias it. const items = @constCast(self.findings); std.mem.sort(Finding, items, {}, struct { fn lessThan(_: void, a: Finding, b: Finding) bool { if (a.first_line != b.first_line) return a.first_line < b.first_line; return std.mem.lessThan(u8, a.key, b.key); } }.lessThan); } }; /// Inputs a schema's `semanticCheck` may need beyond the record itself. /// /// Passed by value alongside the record rather than stored on `Sink`, /// which collects findings and should not also carry inputs. pub const Context = struct { /// The current calendar day - not an `as_of`. Rules like "a /// `close_date` in the future" are nonsensical against a /// back-dated reference: they would flag every real close made /// after it. Captured once at the unit-of-work entry point and /// threaded down, per the `today` rule in AGENTS.md. today: Date, }; /// Collects findings during a walk. Passed to a schema's /// `semanticCheck` so model-owned rules report through the same /// channel as the generic ones. /// /// Deliberately has no format-string method: `addOwned` takes a string /// the caller built with `std.fmt.allocPrint(sink.allocator, ...)` and /// assumes ownership of it. That keeps `anytype` out of the contract /// while leaving ownership explicit at the call site. pub const Sink = struct { /// The `Result` arena. Anything allocated here lives as long as /// the `Result`. allocator: std.mem.Allocator, findings: std.ArrayList(Finding), truncated: bool = false, /// Line of the record being walked, stamped on every finding added. /// Set by `check` and `checkSemantic` before each record; a /// schema's `fileCheck` walks its own records and must set it. line: u32 = 0, /// Add a finding whose `detail` is a static string. pub fn addStatic(self: *Sink, kind: Kind, key: []const u8, detail: []const u8) !void { try self.push(.{ .kind = kind, .key = key, .detail = detail, .first_line = self.line }); } /// Add a finding whose `detail` was allocated from /// `self.allocator`. The `Result` arena frees it. pub fn addOwned(self: *Sink, kind: Kind, key: []const u8, detail: []const u8) !void { try self.push(.{ .kind = kind, .key = key, .detail = detail, .first_line = self.line }); } fn addSuggestion(self: *Sink, kind: Kind, key: []const u8, suggestion: []const u8) !void { try self.push(.{ .kind = kind, .key = key, .suggestion = suggestion, .first_line = self.line }); } /// Roll `f` into an existing matching finding, or append it. /// /// Identity is `(kind, key, detail)` - `detail` included on purpose. /// The generic kinds carry a static per-kind detail, so they still /// collapse (one typo across 200 records stays one line). A /// `semantic` finding's detail names the specific record ("account /// 'Sample IRA': ..."), so including it keeps two accounts with the /// same broken field from merging into one line that names only the /// first. fn push(self: *Sink, f: Finding) !void { for (self.findings.items) |*existing| { if (existing.kind == f.kind and std.mem.eql(u8, existing.key, f.key) and std.mem.eql(u8, existing.detail, f.detail)) { existing.noteRepeat(f.first_line); return; } } if (self.findings.items.len >= max_findings) { self.truncated = true; return; } try self.findings.append(self.allocator, f); } }; // ── Shape: what a record is allowed to contain ──────────────── /// A conditional-applicability rule for one field. pub const Rule = struct { name: []const u8, /// Discriminator values this field is actually read for. `null` /// means "read for all of them". only: ?[]const []const u8 = null, /// Derived by zfin, never read from a hand-edited file. Presence /// in a user file is itself a finding. derived: bool = false, }; /// Conditional-applicability rules for a flat record, keyed off one /// enum field's value. pub const Scopes = struct { /// Field whose value selects applicability (e.g. `security_type`). discriminator: []const u8, /// Value used when the discriminator is absent from a record. /// DERIVED from the field's declared default by `shapeOfSchema`, /// never hand-written, so it cannot drift from the struct. default_value: []const u8, rules: []const Rule, fn ruleFor(self: Scopes, name: []const u8) ?Rule { for (self.rules) |r| { if (std.mem.eql(u8, r.name, name)) return r; } return null; } }; /// What field names a record may carry. pub const Shape = union(enum) { /// A plain struct: one fixed name set for every record. flat: Flat, /// A tagged union: the tag field's value selects the name set. tagged: Tagged, pub const Flat = struct { names: []const []const u8, scopes: ?Scopes = null, }; pub const Tagged = struct { tag_key: []const u8, variants: []const Variant, fn variantFor(self: Tagged, tag_value: []const u8) ?Variant { for (self.variants) |v| { if (std.mem.eql(u8, v.tag_value, tag_value)) return v; } return null; } }; pub const Variant = struct { tag_value: []const u8, names: []const []const u8, }; }; /// Field names of `T`, as a comptime slice. The single source of the /// valid-name set - derived, so it cannot drift from the struct. /// /// The data is held as a container-level `const` so it has static /// storage and the returned slice stays valid when this is called from /// a runtime context. pub fn namesOf(comptime T: type) []const []const u8 { const Holder = struct { const names = blk: { const fields = std.meta.fields(T); var n: [fields.len][]const u8 = undefined; for (fields, 0..) |f, i| n[i] = f.name; break :blk n; }; }; return &Holder.names; } /// Build the `Shape` for a schema module, validating its contract. /// /// Handles struct and tagged-union records. For a union, the tag key /// is `Record.srf_tag_field` when declared and `"type"` otherwise, /// matching SRF's own rule, and the tag key is always accepted even /// when the variant struct does not redeclare it - `to()` consumes the /// tag before recursing into the variant, so `SrfConfig` (which /// redeclares `type`) and `Journal.Acknowledgment` (which does not) /// must both lint clean. pub fn shapeOfSchema(comptime S: type) Shape { const Holder = struct { const shape = blk: { validateSchema(S); const Record = S.Record; break :blk switch (@typeInfo(Record)) { .@"struct" => Shape{ .flat = .{ .names = namesOf(Record), .scopes = if (@hasDecl(S, "field_rules")) scopesOfSchema(S) else null, } }, .@"union" => tagged: { const tag_key = if (@hasDecl(Record, "srf_tag_field")) Record.srf_tag_field else "type"; const vfields = std.meta.fields(Record); var variants: [vfields.len]Shape.Variant = undefined; for (vfields, 0..) |vf, i| { // The tag key is valid for every variant // whether or not the variant redeclares it. const inner = namesOf(vf.type); var names: [inner.len + 1][]const u8 = undefined; names[0] = tag_key; var n: usize = 1; for (inner) |name| { if (!std.mem.eql(u8, name, tag_key)) { names[n] = name; n += 1; } } const frozen = names; variants[i] = .{ .tag_value = vf.name, .names = frozen[0..n] }; } const frozen_variants = variants; break :tagged Shape{ .tagged = .{ .tag_key = tag_key, .variants = &frozen_variants } }; }, else => @compileError("srf_schema `" ++ @typeName(Record) ++ "`: Record must be a struct or tagged union"), }; }; }; return Holder.shape; } /// Assemble `Scopes` from a schema's `field_rules`, deriving /// `default_value` from the discriminator field's declared default so /// the two cannot disagree. fn scopesOfSchema(comptime S: type) Scopes { comptime { const Record = S.Record; const disc = S.scope_discriminator; const D = @FieldType(Record, disc); const default_ptr = for (std.meta.fields(Record)) |f| { if (std.mem.eql(u8, f.name, disc)) break f.default_value_ptr; } else unreachable; if (default_ptr == null) { @compileError("srf_schema `" ++ @typeName(Record) ++ "`: discriminator `" ++ disc ++ "` must have a default value (it selects applicability for records that omit it)"); } const default_tag: D = @as(*const D, @ptrCast(@alignCast(default_ptr.?))).*; return .{ .discriminator = disc, .default_value = @tagName(default_tag), .rules = &S.field_rules, }; } } // ── Comptime contract validation ────────────────────────────── /// Assert a model's `srf_schema` conforms to the contract, with a /// copy-pasteable `@compileError` when it does not. Mirrors /// `tui/tab_framework.zig`'s `validateTabModule`. /// /// Required: /// pub const Record = ; /// pub const file_label = "portfolio.srf"; /// pub const doc_path = "reference/config/portfolio-srf.md"; /// /// Optional: /// pub const undocumented = [_][]const u8{ ... }; /// pub const scope_discriminator = "security_type"; // with field_rules /// pub const field_rules = [_]srf_lint.Rule{ ... }; // with scope_discriminator /// pub fn semanticCheck(rec: Record, ctx: srf_lint.Context, sink: *srf_lint.Sink) !void /// pub fn fileCheck(data: []const u8, ctx: srf_lint.Context, sink: *srf_lint.Sink) !void /// /// `semanticCheck` sees one record at a time and suits rules that are /// about that record alone. `fileCheck` gets the whole file, for rules /// that depend on state ACROSS records - limits, "at most one of", and /// ordering. It walks the records itself and sets `sink.line` as it /// goes. Its intended use is a model whose parser already enforces /// such rules: pass the sink to the parser, so the rules have one /// definition rather than a parser copy and a lint copy that can drift. pub fn validateSchema(comptime S: type) void { comptime { const kind = "SRF schema"; const name = if (@hasDecl(S, "file_label")) S.file_label else @typeName(S); if (!@hasDecl(S, "Record")) { @compileError(kind ++ " `" ++ name ++ "` is missing `pub const Record = ;`"); } comptime_validator.expectDeclWithType( kind, name, S, "file_label", []const u8, "pub const file_label: []const u8 = \"portfolio.srf\";", ); comptime_validator.expectDeclWithType( kind, name, S, "doc_path", []const u8, "pub const doc_path: []const u8 = \"reference/config/portfolio-srf.md\";", ); const has_disc = @hasDecl(S, "scope_discriminator"); const has_rules = @hasDecl(S, "field_rules"); if (has_disc != has_rules) { @compileError(kind ++ " `" ++ name ++ "`: `scope_discriminator` and `field_rules` " ++ "must be declared together (one selects applicability, the other lists it)"); } if (has_rules) validateRules(S, kind, name); if (@hasDecl(S, "semanticCheck")) { comptime_validator.expectFnInferredError( kind, name, S, "semanticCheck", &.{ S.Record, Context, *Sink }, void, "pub fn semanticCheck(rec: Record, ctx: srf_lint.Context, sink: *srf_lint.Sink) !void", ); } if (@hasDecl(S, "fileCheck")) { comptime_validator.expectFnInferredError( kind, name, S, "fileCheck", &.{ []const u8, Context, *Sink }, void, "pub fn fileCheck(data: []const u8, ctx: srf_lint.Context, sink: *srf_lint.Sink) !void", ); } } } /// Exhaustiveness check over `field_rules`, in both directions. /// /// This is the mechanism that stops the rules from drifting from the /// struct: adding a field to `Record` fails the build until it is /// classified, and renaming one fails the build here too. `only` /// values are checked against the discriminator enum's tag names, so a /// typo in the rules themselves is also a compile error. fn validateRules(comptime S: type, comptime kind: []const u8, comptime name: []const u8) void { comptime { // The exhaustiveness check is O(fields x rules) string // comparisons - 22 x 22 for `Lot` - which overruns the default // branch budget on its own. @setEvalBranchQuota(100_000); const Record = S.Record; const disc = S.scope_discriminator; const fields = std.meta.fields(Record); if (!@hasField(Record, disc)) { @compileError(kind ++ " `" ++ name ++ "`: scope_discriminator `" ++ disc ++ "` is not a field of " ++ @typeName(Record)); } const D = @FieldType(Record, disc); if (@typeInfo(D) != .@"enum") { @compileError(kind ++ " `" ++ name ++ "`: scope_discriminator `" ++ disc ++ "` must be an enum field, got " ++ @typeName(D)); } // Every field classified exactly once. for (fields) |f| { var seen = 0; for (S.field_rules) |r| { if (std.mem.eql(u8, r.name, f.name)) seen += 1; } if (seen == 0) { @compileError(kind ++ " `" ++ name ++ "`: field `" ++ f.name ++ "` is not classified in `field_rules`. Add one of:\n" ++ " .{ .name = \"" ++ f.name ++ "\" }, // read for every " ++ disc ++ "\n" ++ " .{ .name = \"" ++ f.name ++ "\", .only = &.{.some_value} }, // read only for those\n" ++ " .{ .name = \"" ++ f.name ++ "\", .derived = true }, // zfin derives it; never hand-edited"); } if (seen > 1) { @compileError(kind ++ " `" ++ name ++ "`: field `" ++ f.name ++ "` is classified more than once in `field_rules`"); } } // Every rule names a real field, and every `only` value a real tag. for (S.field_rules) |r| { if (!@hasField(Record, r.name)) { @compileError(kind ++ " `" ++ name ++ "`: `field_rules` entry `" ++ r.name ++ "` is not a field of " ++ @typeName(Record) ++ " (renamed or removed?)"); } if (r.derived and r.only != null) { @compileError(kind ++ " `" ++ name ++ "`: `field_rules` entry `" ++ r.name ++ "` sets both `derived` and `only`; a derived field is never hand-edited for any " ++ disc); } if (r.only) |vals| { if (vals.len == 0) { @compileError(kind ++ " `" ++ name ++ "`: `field_rules` entry `" ++ r.name ++ "` has an empty `only` list; omit `only` for \"read everywhere\" or set `derived`"); } for (vals) |v| { if (!@hasField(D, v)) { @compileError(kind ++ " `" ++ name ++ "`: `field_rules` entry `" ++ r.name ++ "` lists `only` value `" ++ v ++ "`, which is not a tag of " ++ @typeName(D)); } } } } } } // ── Name matching ───────────────────────────────────────────── fn indexOfName(names: []const []const u8, key: []const u8) ?usize { for (names, 0..) |n, i| { if (std.mem.eql(u8, n, key)) return i; } return null; } /// A real field name differing from `key` only by case. SRF matches /// with `std.mem.eql`, so these silently do nothing. fn caseMatch(names: []const []const u8, key: []const u8) ?[]const u8 { for (names) |n| { if (std.ascii.eqlIgnoreCase(n, key)) return n; } return null; } /// A real field name in a prefix relationship with `key`, in either /// direction - the shape of a dropped or doubled trailing character /// (`price_dat`, `price_datee`) or a truncation. /// /// Deliberately exact rather than an edit-distance score: no threshold /// to tune, and it cannot produce a wrong suggestion. It catches /// strictly fewer typos than Levenshtein would, and the renderers make /// up the difference: `audit` prints the file's full valid-name set /// whenever a finding is about a name (`Result.hasNameFindings`), and /// `doctor` points at the reference page, which the doc-sync test keeps /// complete. /// /// Picks the candidate whose LENGTH is closest to `key`'s, because one /// real field can prefix another: `Lot` has both `price` and /// `price_date`, so `price_dat` prefix-matches both and first-hit-wins /// would answer `price`. A regression test walks every registered /// model's fields and asserts each one's truncated and doubled forms /// suggest it back. fn prefixMatch(names: []const []const u8, key: []const u8) ?[]const u8 { if (key.len < min_prefix_len) return null; var best: ?[]const u8 = null; var best_delta: usize = std.math.maxInt(usize); for (names) |n| { if (n.len < min_prefix_len) continue; if (!std.mem.startsWith(u8, n, key) and !std.mem.startsWith(u8, key, n)) continue; const delta = if (n.len > key.len) n.len - key.len else key.len - n.len; if (delta < best_delta) { best_delta = delta; best = n; } } return best; } // ── The walk ────────────────────────────────────────────────── /// One record's raw keys, buffered so the discriminator can be located /// before applicability is judged (it may appear after the field it /// governs). const RecordBuf = struct { // SAFETY: only `keys[0..len]` is ever read, and `push` writes each // slot before incrementing `len`. keys: [max_fields_per_record][]const u8 = undefined, len: usize = 0, overflowed: bool = false, /// Raw string value of the discriminator/tag field, when present. disc_value: ?[]const u8 = null, fn push(self: *RecordBuf, key: []const u8) void { if (self.len == self.keys.len) { self.overflowed = true; return; } self.keys[self.len] = key; self.len += 1; } }; /// Walk `data` as SRF and report every key that `shape` does not /// explain. `data` must outlive the returned `Result`. pub fn check(allocator: std.mem.Allocator, data: []const u8, shape: Shape) !Result { var arena = std.heap.ArenaAllocator.init(allocator); errdefer arena.deinit(); var sink: Sink = .{ .allocator = arena.allocator(), .findings = .empty }; var reader = std.Io.Reader.fixed(data); // A file that isn't SRF at all is `doctor`'s existing parse-check's // problem, not ours - report no findings rather than a confusing // wall of "unknown field". var it = srf.iterator(&reader, arena.allocator(), .{ .parse_allocator = .none }) catch { return .{ .findings = &.{}, .shape = shape, .arena = arena }; }; defer it.deinit(); while (it.next() catch null) |fields| { // Matches `cache/store.zig`'s diagnostics: the record's first // line, captured before the field walk advances it. const line: u32 = @intCast(it.state.line); sink.line = line; var buf: RecordBuf = .{}; const disc_name: ?[]const u8 = switch (shape) { .flat => |f| if (f.scopes) |s| s.discriminator else null, .tagged => |t| t.tag_key, }; while (fields.next() catch null) |f| { buf.push(f.key); if (disc_name) |dn| { if (buf.disc_value == null and std.mem.eql(u8, f.key, dn)) { if (f.value) |v| { if (v == .string) buf.disc_value = v.string; } } } } try checkRecord(&sink, shape, buf); } const findings = try sink.findings.toOwnedSlice(arena.allocator()); return .{ .findings = findings, .shape = shape, .truncated = sink.truncated, .arena = arena, }; } fn checkRecord(sink: *Sink, shape: Shape, buf: RecordBuf) !void { const names: []const []const u8, const scopes: ?Scopes = switch (shape) { .flat => |f| .{ f.names, f.scopes }, .tagged => |t| blk: { const tag_value = buf.disc_value orelse return; // untagged record: `to()` errors on it const variant = t.variantFor(tag_value) orelse return; // unknown tag: ditto break :blk .{ variant.names, null }; }, }; // Which valid names have been consumed, for duplicate detection. var seen = [_]bool{false} ** max_fields_per_record; const disc_value: []const u8 = if (scopes) |s| (buf.disc_value orelse s.default_value) else ""; for (buf.keys[0..buf.len]) |key| { if (indexOfName(names, key)) |idx| { if (idx < seen.len) { if (seen[idx]) { try sink.addStatic(.duplicate, key, "SRF keeps the first occurrence and ignores the rest"); continue; } seen[idx] = true; } // Known field. Is it read for this record? if (scopes) |s| { if (buf.overflowed) continue; const rule = s.ruleFor(key) orelse continue; if (rule.derived) { try sink.addStatic(.derived, key, "derived by zfin; remove it from your file"); } else if (rule.only) |vals| { if (indexOfName(vals, disc_value) == null) { const detail = try describeOnly(sink.allocator, s.discriminator, vals, disc_value); try sink.addOwned(.inapplicable, key, detail); } } } continue; } if (caseMatch(names, key)) |n| { try sink.addSuggestion(.case_mismatch, key, n); } else if (prefixMatch(names, key)) |n| { try sink.addSuggestion(.near_miss, key, n); } else { try sink.addStatic(.unknown, key, ""); } } } /// "ignored for security_type stock; read only for cd". fn describeOnly( allocator: std.mem.Allocator, discriminator: []const u8, only: []const []const u8, actual: []const u8, ) ![]const u8 { var aw: std.Io.Writer.Allocating = .init(allocator); errdefer aw.deinit(); try aw.writer.print("ignored for {s} {s}; read only for ", .{ discriminator, actual }); for (only, 0..) |v, i| { if (i > 0) try aw.writer.writeAll(if (i + 1 == only.len) " and " else ", "); try aw.writer.writeAll(v); } return aw.toOwnedSlice(); } /// Run a schema's model-owned `semanticCheck` and/or `fileCheck` over /// `data`, appending to `result`. /// /// A SECOND typed pass, separate from `check`'s raw one, because SRF's /// iterators are single-pass: `to()` drains the fields that the raw /// walk needs. In the per-record pass, records that fail to coerce are /// skipped silently - the typed parser's own diagnostics (and /// `doctor`'s parse-check) already report those, and duplicating them /// here would double every message. A `fileCheck` owns its own walk, /// so what it reports about an uncoercible record is its decision. pub fn checkSemantic(comptime S: type, result: *Result, data: []const u8, ctx: Context) !void { const has_record = @hasDecl(S, "semanticCheck"); const has_file = @hasDecl(S, "fileCheck"); if (!has_record and !has_file) return; const allocator = result.arena.allocator(); var sink: Sink = .{ .allocator = allocator, .findings = .empty }; try sink.findings.appendSlice(allocator, result.findings); sink.truncated = result.truncated; if (has_record) try recordPass(S, &sink, data, ctx); if (has_file) try S.fileCheck(data, ctx, &sink); result.findings = try sink.findings.toOwnedSlice(allocator); result.truncated = sink.truncated; } fn recordPass(comptime S: type, sink: *Sink, data: []const u8, ctx: Context) !void { var reader = std.Io.Reader.fixed(data); var it = srf.iterator(&reader, sink.allocator, .{ .parse_allocator = .none }) catch return; defer it.deinit(); while (it.next() catch null) |fields| { sink.line = @intCast(it.state.line); const rec = fields.to(S.Record, @import("srf_opts.zig").user_edited) catch continue; try S.semanticCheck(rec, ctx, sink); } } /// Write the valid field names for `shape` to `w`, wrapped to `width` /// columns and indented by `indent` spaces. /// /// Renderers call this whenever `Result.hasNameFindings`. It is /// what makes the deliberately-conservative matchers sufficient: even /// when no suggestion can be offered, the user gets the authoritative /// list - derived from the struct, so unlike the reference docs it /// cannot be out of date. pub fn writeValidNames(w: *std.Io.Writer, shape: Shape, indent: usize, width: usize) !void { switch (shape) { .flat => |f| try writeNameList(w, "", f.names, indent, width), .tagged => |t| { for (t.variants) |v| { var label_buf: [64]u8 = undefined; const label = std.fmt.bufPrint(&label_buf, "{s}::{s} ", .{ t.tag_key, v.tag_value }) catch ""; try writeNameList(w, label, v.names, indent, width); } }, } } fn writeNameList( w: *std.Io.Writer, label: []const u8, names: []const []const u8, indent: usize, width: usize, ) !void { try w.splatByteAll(' ', indent); try w.writeAll(label); var col = indent + label.len; for (names) |n| { // +1 for the separating space. Wrap before overflowing so a // narrow terminal does not ragged-wrap mid-name. if (col > indent and col + n.len + 1 > width) { try w.writeAll("\n"); try w.splatByteAll(' ', indent + 2); col = indent + 2; } try w.writeAll(n); try w.writeAll(" "); col += n.len + 1; } try w.writeAll("\n"); } // ── Registry ────────────────────────────────────────────────── /// Every user-authored SRF file zfin reads, paired with the model that /// owns its schema. Nine one-liners, mirroring `tui.zig`'s /// `tab_modules`. /// /// **Adding a tenth user-authored file means adding it here.** That is /// the one drift this design does not close at comptime - there is no /// way to ask Zig "who references `srf_opts.user_edited`" - but it is /// the cheap kind: a new file goes unchecked, nothing becomes wrong. /// `grep -rn srf_opts.user_edited src/` is the authoritative index, and /// that constant's doc comment carries the same reminder. /// /// `history/imported_values.srf` is deliberately ABSENT. It is /// generated by `tools/import_values.zig` from a spreadsheet export and /// hand-editing it is explicitly disallowed (see the module doc on /// `data/imported_values.zig`), so there are no hand-typed field names /// to get wrong. It is also the only `user_edited` parse site with no /// `docs/reference/config/*-srf.md` page, which independently confirms /// the classification. pub const schemas = .{ @import("models/portfolio.zig").srf_schema, @import("analytics/analysis.zig").srf_schema, @import("models/classification.zig").srf_schema, @import("models/transaction_log.zig").srf_schema, @import("analytics/projections.zig").srf_schema, @import("data/Journal.zig").srf_schema, @import("tui/keybinds.zig").srf_schema, @import("tui/theme.zig").srf_schema, @import("commands/common.zig").srf_schema, }; /// Validate every registered schema at build time. Mirrors /// `tui.zig`'s comptime sweep over `tab_modules`. pub const validated_schemas = blk: { for (schemas) |S| _ = shapeOfSchema(S); break :blk true; }; /// Number of registered schemas. pub const schema_count = schemas.len; // ── Tests ───────────────────────────────────────────────────── const testing = std.testing; // Every field of every registered model must appear in that model's // reference page, or be listed in the schema's `undocumented`. // // This is the mechanism that keeps the docs from drifting the way // `portfolio-srf.md` already had: it documented 14 of `Lot`'s 22 // fields in its table, and adding a field had no consequence. Now it // does - this test fails until the field is documented or explicitly // exempted, and the exemption is a visible decision in the schema. // // A field counts as documented if it appears anywhere outside a fenced // code block, in either spelling the docs use: bare (`` `symbol` ``, // reference tables) or on-wire (`` `symbol::` ``, prose). Code fences // are excluded on purpose - an example that happens to mention a field // is not a description of it. test "doc sync: every model field appears in its reference page" { // Referencing this forces the comptime sweep over `schemas`, so a // schema with a contract violation or a non-exhaustive // `field_rules` fails the build rather than going unvalidated. try testing.expect(validated_schemas); var aw: std.Io.Writer.Allocating = .init(testing.allocator); defer aw.deinit(); var missing: usize = 0; inline for (schemas) |S| { const doc_file = comptime std.fs.path.basename(S.doc_path); const doc = config_docs.find(doc_file) orelse { std.debug.print("srf_schema '{s}': no reference page named '{s}'\n", .{ S.file_label, doc_file }); return error.MissingReferencePage; }; switch (comptime shapeOfSchema(S)) { .flat => |f| missing += try reportUndocumented(&aw.writer, S, doc, f.names, ""), .tagged => |t| { for (t.variants) |v| { missing += try reportUndocumented(&aw.writer, S, doc, v.names, t.tag_key); } }, } } if (missing > 0) { std.debug.print( \\ \\{d} model field(s) are not described in their reference page: \\{s} \\Document each one, or add it to that schema's `undocumented` \\list with a comment saying why it is not user-facing. \\ , .{ missing, aw.written() }); } try testing.expectEqual(@as(usize, 0), missing); } /// Write a line to `w` for each field of `S` absent from its reference /// page, and return how many there were. /// /// Reports through a writer rather than printing: zlint's `no-print` /// rule exempts `test` blocks but not the helpers they call, and /// funnelling the text back to the one caller is both cleaner and /// keeps the diagnostic in a single flush. fn reportUndocumented( w: *std.Io.Writer, comptime S: type, doc: config_docs.Doc, names: []const []const u8, tag_key: []const u8, ) !usize { const exempt: []const []const u8 = if (@hasDecl(S, "undocumented")) &S.undocumented else &.{}; var n: usize = 0; for (names) |name| { // The union tag key is SRF machinery, not a model field. if (tag_key.len > 0 and std.mem.eql(u8, name, tag_key)) continue; if (indexOfName(doc.names, name) != null) continue; if (indexOfName(exempt, name) != null) continue; try w.print(" {s}: field '{s}' is undocumented in {s}\n", .{ S.file_label, name, doc.file }); n += 1; } return n; } test "registry: every schema has a distinct file label and doc page" { try testing.expect(schema_count == 9); inline for (schemas, 0..) |A, i| { inline for (schemas, 0..) |B, j| { if (comptime i >= j) continue; try testing.expect(!std.mem.eql(u8, A.file_label, B.file_label)); try testing.expect(!std.mem.eql(u8, A.doc_path, B.doc_path)); } } } test "registry: every schema builds a usable shape and lints a clean empty file" { inline for (schemas) |S| { var r = try check(testing.allocator, "#!srfv1\n", comptime shapeOfSchema(S)); defer r.deinit(); try testing.expect(r.isClean()); } } test "registry: a one-character typo of any real field suggests that field back" { // The property that matters to a user, checked against the REAL // models rather than a fixture: drop the last character of a field // name, or double it, and the lint must point at the field you // meant. `Lot` alone has `price`, `price_date` and `price_ratio`, // so this is where a naive first-hit-wins matcher goes wrong. var aw: std.Io.Writer.Allocating = .init(testing.allocator); defer aw.deinit(); var bad: usize = 0; inline for (schemas) |S| { switch (comptime shapeOfSchema(S)) { .flat => |f| bad += try reportBadSuggestions(&aw.writer, S.file_label, f.names), .tagged => |t| { for (t.variants) |v| bad += try reportBadSuggestions(&aw.writer, S.file_label, v.names); }, } } if (bad > 0) std.debug.print("\n{s}", .{aw.written()}); try testing.expectEqual(@as(usize, 0), bad); } /// Write a line to `w` for each field whose one-character typo forms /// resolve to the wrong suggestion, and return how many there were. fn reportBadSuggestions(w: *std.Io.Writer, label: []const u8, names: []const []const u8) !usize { var buf: [128]u8 = undefined; var bad: usize = 0; for (names) |name| { if (name.len < min_prefix_len + 1) continue; // Dropped trailing character. If the truncation IS another real // field, an exact match wins and no suggestion is wanted. const truncated = name[0 .. name.len - 1]; if (indexOfName(names, truncated) == null) { const got = prefixMatch(names, truncated); if (got == null or !std.mem.eql(u8, got.?, name)) { try w.print(" {s}: '{s}' (from '{s}') suggested {?s}, want '{s}'\n", .{ label, truncated, name, got, name }); bad += 1; } } // Doubled trailing character. @memcpy(buf[0..name.len], name); buf[name.len] = name[name.len - 1]; const doubled = buf[0 .. name.len + 1]; if (indexOfName(names, doubled) == null) { const got = prefixMatch(names, doubled); if (got == null or !std.mem.eql(u8, got.?, name)) { try w.print(" {s}: '{s}' (from '{s}') suggested {?s}, want '{s}'\n", .{ label, doubled, name, got, name }); bad += 1; } } } return bad; } // ── Fixtures ────────────────────────────────────────────────── const TestLotType = enum { stock, option, cd, cash }; const TestLot = struct { symbol: []const u8 = "", shares: f64, account: ?[]const u8 = null, security_type: TestLotType = .stock, rate: ?f64 = null, strike: ?f64 = null, split_factor: f64 = 1.0, }; const test_lot_schema = struct { pub const Record = TestLot; pub const file_label: []const u8 = "test_lot.srf"; pub const doc_path: []const u8 = "reference/config/test-lot-srf.md"; pub const scope_discriminator: []const u8 = "security_type"; pub const field_rules = [_]Rule{ .{ .name = "symbol" }, .{ .name = "shares" }, .{ .name = "account" }, .{ .name = "security_type" }, .{ .name = "rate", .only = &.{"cd"} }, .{ .name = "strike", .only = &.{"option"} }, .{ .name = "split_factor", .derived = true }, }; }; const TestUnion = union(enum) { config: struct { type: []const u8 = "", horizon: u16 = 0 }, birthdate: struct { date: []const u8 = "", person: u8 = 1 }, }; const test_union_schema = struct { pub const Record = TestUnion; pub const file_label: []const u8 = "test_union.srf"; pub const doc_path: []const u8 = "reference/config/test-union-srf.md"; }; fn lintLot(data: []const u8) !Result { return check(testing.allocator, data, shapeOfSchema(test_lot_schema)); } fn findingFor(r: Result, key: []const u8) ?Finding { for (r.findings) |f| { if (std.mem.eql(u8, f.key, key)) return f; } return null; } // ── check: the raw walk ─────────────────────────────────────── test "check: clean file produces no findings" { const data = \\#!srfv1 \\symbol::VTI,shares:num:100,account::Sample Brokerage \\symbol::SPY,shares:num:50,account::Sample IRA \\ ; var r = try lintLot(data); defer r.deinit(); try testing.expect(r.isClean()); } test "check: unknown key with no relative" { const data = \\#!srfv1 \\symbol::VTI,shares:num:100,cost_basis:num:1000 \\ ; var r = try lintLot(data); defer r.deinit(); try testing.expectEqual(@as(usize, 1), r.findings.len); try testing.expectEqual(Kind.unknown, r.findings[0].kind); try testing.expectEqualStrings("cost_basis", r.findings[0].key); try testing.expectEqual(@as(?[]const u8, null), r.findings[0].suggestion); } test "check: case-only mismatch names the real field" { const data = \\#!srfv1 \\Symbol::VTI,shares:num:100 \\ ; var r = try lintLot(data); defer r.deinit(); const f = findingFor(r, "Symbol") orelse return error.MissingFinding; try testing.expectEqual(Kind.case_mismatch, f.kind); try testing.expectEqualStrings("symbol", f.suggestion.?); } test "check: near miss on a dropped trailing character" { const data = \\#!srfv1 \\symbol::VTI,shares:num:100,accoun::Sample IRA \\ ; var r = try lintLot(data); defer r.deinit(); const f = findingFor(r, "accoun") orelse return error.MissingFinding; try testing.expectEqual(Kind.near_miss, f.kind); try testing.expectEqualStrings("account", f.suggestion.?); } test "check: near miss on a doubled trailing character" { const data = \\#!srfv1 \\symbol::VTI,shares:num:100,accountt::Sample IRA \\ ; var r = try lintLot(data); defer r.deinit(); const f = findingFor(r, "accountt") orelse return error.MissingFinding; try testing.expectEqual(Kind.near_miss, f.kind); try testing.expectEqualStrings("account", f.suggestion.?); } test "check: prefix matching ignores keys below the length floor" { // `sy` is a prefix of `symbol` but too short to suggest against. const data = \\#!srfv1 \\shares:num:100,sy::VTI \\ ; var r = try lintLot(data); defer r.deinit(); const f = findingFor(r, "sy") orelse return error.MissingFinding; try testing.expectEqual(Kind.unknown, f.kind); } test "check: duplicate key in one record" { const data = \\#!srfv1 \\symbol::VTI,shares:num:100,shares:num:200 \\ ; var r = try lintLot(data); defer r.deinit(); const f = findingFor(r, "shares") orelse return error.MissingFinding; try testing.expectEqual(Kind.duplicate, f.kind); } test "check: inapplicable field for the record's discriminator" { const data = \\#!srfv1 \\symbol::VTI,shares:num:100,rate:num:5.25 \\ ; var r = try lintLot(data); defer r.deinit(); const f = findingFor(r, "rate") orelse return error.MissingFinding; try testing.expectEqual(Kind.inapplicable, f.kind); // Default discriminator value is derived from the struct, so an // absent `security_type` still reports as `stock`. try testing.expect(std.mem.indexOf(u8, f.detail, "security_type stock") != null); try testing.expect(std.mem.indexOf(u8, f.detail, "cd") != null); } test "check: applicable field for the right discriminator is clean" { const data = \\#!srfv1 \\symbol::CD1,shares:num:1000,security_type::cd,rate:num:5.25 \\ ; var r = try lintLot(data); defer r.deinit(); try testing.expect(r.isClean()); } test "check: discriminator is honored even when it appears after the field" { // `strike` precedes `security_type`, so the record must be // buffered before applicability is judged. const data = \\#!srfv1 \\symbol::AMZN,shares:num:1,strike:num:200,security_type::option \\ ; var r = try lintLot(data); defer r.deinit(); try testing.expect(r.isClean()); } test "check: derived field in a hand-edited file" { const data = \\#!srfv1 \\symbol::VTI,shares:num:100,split_factor:num:4 \\ ; var r = try lintLot(data); defer r.deinit(); const f = findingFor(r, "split_factor") orelse return error.MissingFinding; try testing.expectEqual(Kind.derived, f.kind); } test "check: repeated typo rolls up instead of repeating" { const data = \\#!srfv1 \\symbol::A,shares:num:1,accoun::X \\symbol::B,shares:num:2,accoun::X \\symbol::C,shares:num:3,accoun::X \\symbol::D,shares:num:4,accoun::X \\ ; var r = try lintLot(data); defer r.deinit(); try testing.expectEqual(@as(usize, 1), r.findings.len); const f = r.findings[0]; try testing.expectEqual(@as(u32, 4), f.count); try testing.expectEqual(@as(u32, 2), f.first_line); // Two more lines retained for the "lines 2, 3, 4, ..." tail. try testing.expectEqual(@as(u8, 2), f.extra_len); try testing.expectEqual(@as(u32, 3), f.extra_lines[0]); try testing.expectEqual(@as(u32, 4), f.extra_lines[1]); } test "check: line numbers point at the offending record" { const data = \\#!srfv1 \\symbol::A,shares:num:1 \\symbol::B,shares:num:2 \\symbol::C,shares:num:3,bogus::x \\ ; var r = try lintLot(data); defer r.deinit(); const f = findingFor(r, "bogus") orelse return error.MissingFinding; try testing.expectEqual(@as(u32, 4), f.first_line); } test "check: non-SRF input reports nothing rather than a wall of unknowns" { var r = try lintLot("this is not an srf file at all\n"); defer r.deinit(); try testing.expect(r.isClean()); } test "check: empty file is clean" { var r = try lintLot("#!srfv1\n"); defer r.deinit(); try testing.expect(r.isClean()); } // ── Tagged unions ───────────────────────────────────────────── test "shapeOfSchema: tagged union dispatches on the tag value" { const shape = shapeOfSchema(test_union_schema); try testing.expectEqualStrings("type", shape.tagged.tag_key); try testing.expectEqual(@as(usize, 2), shape.tagged.variants.len); const data = \\#!srfv1 \\type::config,horizon:num:30 \\type::birthdate,date::1980-01-01,person:num:1 \\ ; var r = try check(testing.allocator, data, shape); defer r.deinit(); try testing.expect(r.isClean()); } test "shapeOfSchema: tag key is valid whether or not the variant redeclares it" { // `config` redeclares `type`; `birthdate` does not. Both must // accept `type::` without reporting it as unknown. const shape = shapeOfSchema(test_union_schema); for (shape.tagged.variants) |v| { try testing.expect(indexOfName(v.names, "type") != null); } // And it appears exactly once, not twice, for the redeclaring one. const cfg = shape.tagged.variantFor("config").?; var type_count: usize = 0; for (cfg.names) |n| { if (std.mem.eql(u8, n, "type")) type_count += 1; } try testing.expectEqual(@as(usize, 1), type_count); } test "check: typo inside a union variant is caught against that variant" { const data = \\#!srfv1 \\type::config,horizonn:num:30 \\ ; var r = try check(testing.allocator, data, shapeOfSchema(test_union_schema)); defer r.deinit(); const f = findingFor(r, "horizonn") orelse return error.MissingFinding; try testing.expectEqual(Kind.near_miss, f.kind); try testing.expectEqualStrings("horizon", f.suggestion.?); } test "check: a field valid on another variant is not valid on this one" { const data = \\#!srfv1 \\type::config,person:num:2 \\ ; var r = try check(testing.allocator, data, shapeOfSchema(test_union_schema)); defer r.deinit(); const f = findingFor(r, "person") orelse return error.MissingFinding; try testing.expectEqual(Kind.unknown, f.kind); } test "check: unknown tag value is left to the typed parser" { const data = \\#!srfv1 \\type::nonsense,whatever::x \\ ; var r = try check(testing.allocator, data, shapeOfSchema(test_union_schema)); defer r.deinit(); try testing.expect(r.isClean()); } // ── Derivation ──────────────────────────────────────────────── test "namesOf: derives the full field set" { const names = namesOf(TestLot); try testing.expectEqual(@as(usize, 7), names.len); try testing.expect(indexOfName(names, "symbol") != null); try testing.expect(indexOfName(names, "split_factor") != null); try testing.expect(indexOfName(names, "nope") == null); } test "scopesOfSchema: default_value is derived from the struct default" { const shape = shapeOfSchema(test_lot_schema); try testing.expectEqualStrings("stock", shape.flat.scopes.?.default_value); try testing.expectEqualStrings("security_type", shape.flat.scopes.?.discriminator); } test "prefixMatch: picks the closest-length candidate, not the first" { // `Lot`'s real shape: a short field that prefixes two longer ones. const names: []const []const u8 = &.{ "price", "price_date", "price_ratio" }; try testing.expectEqualStrings("price_date", prefixMatch(names, "price_dat").?); try testing.expectEqualStrings("price_date", prefixMatch(names, "price_datee").?); try testing.expectEqualStrings("price_ratio", prefixMatch(names, "price_rati").?); try testing.expectEqualStrings("price", prefixMatch(names, "pricee").?); // Below the floor, no guess at all. try testing.expectEqual(@as(?[]const u8, null), prefixMatch(names, "pr")); } test "check: findings are capped so a garbage file cannot flood the report" { var aw: std.Io.Writer.Allocating = .init(testing.allocator); defer aw.deinit(); try aw.writer.writeAll("#!srfv1\n"); // Each record carries a DISTINCT bogus key, so rollup cannot // collapse them and the cap is what bounds the output. for (0..max_findings + 50) |i| { try aw.writer.print("shares:num:1,zz{d}::x\n", .{i}); } var r = try lintLot(aw.writer.buffered()); defer r.deinit(); try testing.expectEqual(@as(usize, max_findings), r.findings.len); try testing.expect(r.truncated); } test "Sink.addOwned detail is freed with the result" { // Exercises the arena-ownership contract: `describeOnly` // allocates, and `std.testing.allocator` fails the test if // `Result.deinit` does not release it. const data = \\#!srfv1 \\symbol::VTI,shares:num:100,rate:num:1,strike:num:2 \\ ; var r = try lintLot(data); defer r.deinit(); try testing.expectEqual(@as(usize, 2), r.findings.len); } // ── describe / writeValidNames ──────────────────────────────── fn describeOne(data: []const u8, buf: []u8) ![]const u8 { var r = try lintLot(data); defer r.deinit(); if (r.findings.len == 0) return error.NoFinding; // `describe` writes into `buf`, which outlives `r`, but `key` // borrows from `data` - so this only holds while `data` is alive. return r.findings[0].describe(buf); } test "describe: unknown field" { var buf: [Finding.describe_max]u8 = undefined; const msg = try describeOne("#!srfv1\nshares:num:1,cost_basis:num:5\n", &buf); try testing.expectEqualStrings("unrecognized field 'cost_basis'", msg); } test "describe: near miss names the field it meant" { var buf: [Finding.describe_max]u8 = undefined; const msg = try describeOne("#!srfv1\nshares:num:1,accoun::X\n", &buf); try testing.expectEqualStrings("unrecognized field 'accoun' - did you mean 'account'?", msg); } test "describe: case mismatch explains why it silently did nothing" { var buf: [Finding.describe_max]u8 = undefined; const msg = try describeOne("#!srfv1\nshares:num:1,Account::X\n", &buf); try testing.expect(std.mem.indexOf(u8, msg, "differs only by case from 'account'") != null); try testing.expect(std.mem.indexOf(u8, msg, "case-sensitive") != null); } test "describe: duplicate explains that the first wins" { var buf: [Finding.describe_max]u8 = undefined; const msg = try describeOne("#!srfv1\nshares:num:1,shares:num:2\n", &buf); try testing.expect(std.mem.indexOf(u8, msg, "appears twice") != null); try testing.expect(std.mem.indexOf(u8, msg, "keeps the first") != null); } test "describe: inapplicable names the discriminator and the types that read it" { var buf: [Finding.describe_max]u8 = undefined; const msg = try describeOne("#!srfv1\nshares:num:1,rate:num:5\n", &buf); try testing.expectEqualStrings( "field 'rate' ignored for security_type stock; read only for cd", msg, ); } test "describe: derived field says to remove it" { var buf: [Finding.describe_max]u8 = undefined; const msg = try describeOne("#!srfv1\nshares:num:1,split_factor:num:4\n", &buf); try testing.expectEqualStrings("field 'split_factor' is derived by zfin; remove it from your file", msg); } test "describe: rolled-up repeat reports the count and the first lines" { const data = \\#!srfv1 \\shares:num:1,accoun::X \\shares:num:2,accoun::X \\shares:num:3,accoun::X \\shares:num:4,accoun::X \\ ; var buf: [Finding.describe_max]u8 = undefined; const msg = try describeOne(data, &buf); try testing.expect(std.mem.indexOf(u8, msg, "(4 records: lines 2, 3, 4, ...)") != null); } test "describe: exactly three occurrences omits the ellipsis" { const data = \\#!srfv1 \\shares:num:1,accoun::X \\shares:num:2,accoun::X \\shares:num:3,accoun::X \\ ; var buf: [Finding.describe_max]u8 = undefined; const msg = try describeOne(data, &buf); try testing.expect(std.mem.indexOf(u8, msg, "(3 records: lines 2, 3, 4)") != null); try testing.expect(std.mem.indexOf(u8, msg, "...") == null); } test "describe: a short buffer truncates rather than failing" { var r = try lintLot("#!srfv1\nshares:num:1,cost_basis:num:5\n"); defer r.deinit(); var tiny: [8]u8 = undefined; const msg = r.findings[0].describe(&tiny); try testing.expect(msg.len <= tiny.len); try testing.expectEqualStrings("unrecogn", msg); } test "writeValidNames: flat shape lists every field, wrapped" { var aw: std.Io.Writer.Allocating = .init(testing.allocator); defer aw.deinit(); try writeValidNames(&aw.writer, shapeOfSchema(test_lot_schema), 4, 40); const out = aw.written(); // Every field present, including the derived one - the list is // "what SRF will match", not "what you should write". for (namesOf(TestLot)) |n| { try testing.expect(std.mem.indexOf(u8, out, n) != null); } // Wrapped: more than one line, none wildly over the limit. var it = std.mem.splitScalar(u8, std.mem.trimEnd(u8, out, "\n"), '\n'); var lines: usize = 0; while (it.next()) |line| { lines += 1; try testing.expect(line.len <= 44); } try testing.expect(lines > 1); } test "writeValidNames: tagged shape lists each variant separately" { var aw: std.Io.Writer.Allocating = .init(testing.allocator); defer aw.deinit(); try writeValidNames(&aw.writer, shapeOfSchema(test_union_schema), 2, 100); const out = aw.written(); try testing.expect(std.mem.indexOf(u8, out, "type::config") != null); try testing.expect(std.mem.indexOf(u8, out, "type::birthdate") != null); try testing.expect(std.mem.indexOf(u8, out, "horizon") != null); try testing.expect(std.mem.indexOf(u8, out, "person") != null); } // ── checkSemantic / Result.sort ──────────────────────────────── /// A schema whose `semanticCheck` reports a model-specific rule, used to /// exercise the typed second pass without depending on any real model's /// current rule set. const SemRecord = struct { name: []const u8 = "", pct: f64 = 0, }; /// Fixed `today` for tests, so a date rule's verdict cannot change as /// the calendar moves. const test_ctx: Context = .{ .today = Date.fromYmd(2026, 1, 1) }; const sem_schema = struct { pub const Record = SemRecord; pub const file_label: []const u8 = "sem.srf"; pub const doc_path: []const u8 = "reference/config/sem-srf.md"; pub fn semanticCheck(rec: Record, ctx: Context, sink: *Sink) !void { _ = ctx; if (rec.pct > 100) { try sink.addOwned(.semantic, "pct", try std.fmt.allocPrint( sink.allocator, "'{s}': pct must be <= 100 (got {d})", .{ rec.name, rec.pct }, )); } } }; test "checkSemantic: model-owned rules reach the findings list" { const data = \\#!srfv1 \\name::ok,pct:num:50 \\name::bad,pct:num:150 \\ ; var r = try check(testing.allocator, data, shapeOfSchema(sem_schema)); defer r.deinit(); // Field names are all valid, so the raw pass finds nothing. try testing.expect(r.isClean()); try checkSemantic(sem_schema, &r, data, test_ctx); try testing.expectEqual(@as(usize, 1), r.findings.len); try testing.expectEqual(Kind.semantic, r.findings[0].kind); try testing.expectEqual(@as(u32, 3), r.findings[0].first_line); var buf: [Finding.describe_max]u8 = undefined; try testing.expectEqualStrings("'bad': pct must be <= 100 (got 150)", r.findings[0].describe(&buf)); } test "checkSemantic: a schema without the hook is a no-op" { const data = "#!srfv1\nsymbol::VTI,shares:num:1,account::Sample Brokerage\n"; var r = try check(testing.allocator, data, shapeOfSchema(test_lot_schema)); defer r.deinit(); try checkSemantic(test_lot_schema, &r, data, test_ctx); try testing.expect(r.isClean()); } test "checkSemantic: records that fail to coerce are skipped, not reported twice" { // `pct` is a string where a number is declared; under // `user_edited` coercion that is tolerated, so the record still // reaches `semanticCheck`. A record missing a required field would // be skipped - the typed parser's own diagnostics already cover it. const data = \\#!srfv1 \\name::bad,pct::150 \\ ; var r = try check(testing.allocator, data, shapeOfSchema(sem_schema)); defer r.deinit(); try checkSemantic(sem_schema, &r, data, test_ctx); try testing.expectEqual(@as(usize, 1), r.findings.len); } test "Result.sort: orders by line so the two passes interleave correctly" { // Raw-pass finding on line 4, semantic-pass findings on 2 and 3 - // the order they are produced in is not the order to read them in. const data = \\#!srfv1 \\name::a,pct:num:150 \\name::b,pct:num:200 \\name::c,pct:num:1,bogus::x \\ ; var r = try check(testing.allocator, data, shapeOfSchema(sem_schema)); defer r.deinit(); try checkSemantic(sem_schema, &r, data, test_ctx); try testing.expectEqual(@as(usize, 3), r.findings.len); // Unsorted, the raw finding comes first. try testing.expectEqual(@as(u32, 4), r.findings[0].first_line); r.sort(); try testing.expectEqual(@as(u32, 2), r.findings[0].first_line); try testing.expectEqual(@as(u32, 3), r.findings[1].first_line); try testing.expectEqual(@as(u32, 4), r.findings[2].first_line); } test "rollup: semantic findings on different records stay separate" { // Regression guard. Identity used to be `(kind, key)` alone, which // merged these two into one line naming only 'a' - silently hiding // that 'b' was broken too. Both records trip the same field with a // different detail, so both must survive. const data = \\#!srfv1 \\name::a,pct:num:150 \\name::b,pct:num:200 \\ ; var r = try check(testing.allocator, data, shapeOfSchema(sem_schema)); defer r.deinit(); try checkSemantic(sem_schema, &r, data, test_ctx); try testing.expectEqual(@as(usize, 2), r.findings.len); var buf: [Finding.describe_max]u8 = undefined; r.sort(); try testing.expect(std.mem.indexOf(u8, r.findings[0].describe(&buf), "'a'") != null); try testing.expect(std.mem.indexOf(u8, r.findings[1].describe(&buf), "'b'") != null); } test "rollup: identical generic findings still collapse to one line" { // The other half of the same property: a static per-kind detail // means the same typo across many records is ONE finding, so a // copy-pasted mistake does not produce a wall of output. const data = \\#!srfv1 \\shares:num:1,accoun::X \\shares:num:2,accoun::X \\shares:num:3,accoun::X \\ ; var r = try lintLot(data); defer r.deinit(); try testing.expectEqual(@as(usize, 1), r.findings.len); try testing.expectEqual(@as(u32, 3), r.findings[0].count); } test "check: a record wider than the buffer is still name-checked" { // The `max_fields_per_record` guard. Applicability needs the whole // record buffered (the discriminator may come last), so an // over-wide record stops contributing `inapplicable` findings - but // it must still get its names checked, and must not crash. var aw: std.Io.Writer.Allocating = .init(testing.allocator); defer aw.deinit(); try aw.writer.writeAll("#!srfv1\nshares:num:1"); for (0..max_fields_per_record + 10) |i| { try aw.writer.print(",zz{d}::x", .{i}); } try aw.writer.writeAll("\n"); var r = try lintLot(aw.writer.buffered()); defer r.deinit(); // Bounded by the buffer, so not every bogus key is reported - but // the ones that fit are, and nothing panicked. try testing.expect(r.findings.len > 0); try testing.expect(r.findings.len <= max_fields_per_record); } test "Result.sort: two findings on one line are ordered by key" { // The tiebreaker. Without it the order of same-line findings // depends on field declaration order, which makes report diffs // noisy for no reason. const data = \\#!srfv1 \\shares:num:1,zebra::x,alpha::y \\ ; var r = try lintLot(data); defer r.deinit(); try testing.expectEqual(@as(usize, 2), r.findings.len); r.sort(); try testing.expectEqualStrings("alpha", r.findings[0].key); try testing.expectEqualStrings("zebra", r.findings[1].key); } // ── Context ─────────────────────────────────────────────────── const DatedRecord = struct { name: []const u8 = "", on: ?Date = null, }; /// Flags any `on` date after `ctx.today`. Exercises the one thing the /// real date rules depend on: that `checkSemantic` delivers the caller's /// `today`, not some other clock. const dated_schema = struct { pub const Record = DatedRecord; pub const file_label: []const u8 = "dated.srf"; pub const doc_path: []const u8 = "reference/config/dated-srf.md"; pub fn semanticCheck(rec: Record, ctx: Context, sink: *Sink) !void { const on = rec.on orelse return; if (ctx.today.lessThan(on)) try sink.addStatic(.semantic, "on", "in the future"); } }; test "checkSemantic: ctx.today reaches the hook" { const data = \\#!srfv1 \\name::a,on::2026-06-01 \\ ; // Same record, two different `today`s, opposite verdicts - so the // hook must be reading the context rather than a clock of its own. { var r = try check(testing.allocator, data, shapeOfSchema(dated_schema)); defer r.deinit(); try checkSemantic(dated_schema, &r, data, .{ .today = Date.fromYmd(2026, 1, 1) }); try testing.expectEqual(@as(usize, 1), r.findings.len); } { var r = try check(testing.allocator, data, shapeOfSchema(dated_schema)); defer r.deinit(); try checkSemantic(dated_schema, &r, data, .{ .today = Date.fromYmd(2027, 1, 1) }); try testing.expect(r.isClean()); } } test "Result.hasNameFindings: only name problems warrant the valid-field list" { { var r = try lintLot("#!srfv1\nshares:num:1,accoun::X\n"); defer r.deinit(); try testing.expect(r.hasNameFindings()); } { var r = try lintLot("#!srfv1\nshares:num:1,Account::X\n"); defer r.deinit(); try testing.expect(r.hasNameFindings()); } { // Real fields, wrong use - the list would tell the user nothing. var r = try lintLot("#!srfv1\nshares:num:1,shares:num:2,rate:num:1,split_factor:num:2\n"); defer r.deinit(); try testing.expect(!r.isClean()); try testing.expect(!r.hasNameFindings()); } } // ── fileCheck ───────────────────────────────────────────────── const LimitedRecord = struct { name: []const u8 = "" }; /// A cross-record rule no per-record hook can express: at most two /// records. Walks its own records and stamps `sink.line`, per the /// `fileCheck` contract. const limited_schema = struct { pub const Record = LimitedRecord; pub const file_label: []const u8 = "limited.srf"; pub const doc_path: []const u8 = "reference/config/limited-srf.md"; pub fn fileCheck(data: []const u8, ctx: Context, sink: *Sink) !void { _ = ctx; var reader = std.Io.Reader.fixed(data); var it = srf.iterator(&reader, sink.allocator, .{ .parse_allocator = .none }) catch return; defer it.deinit(); var n: usize = 0; while (it.next() catch null) |fields| { sink.line = @intCast(it.state.line); _ = fields.to(Record, .{}) catch continue; n += 1; if (n > 2) try sink.addStatic(.semantic, "name", "limit of 2 records reached; ignoring record"); } } }; test "checkSemantic: fileCheck sees every record and reports cross-record rules" { const data = \\#!srfv1 \\name::a \\name::b \\name::c \\name::d \\ ; var r = try check(testing.allocator, data, shapeOfSchema(limited_schema)); defer r.deinit(); try checkSemantic(limited_schema, &r, data, test_ctx); // Records 3 and 4 trip the same rule with the same detail, so they // roll up into one finding carrying both line numbers. try testing.expectEqual(@as(usize, 1), r.findings.len); try testing.expectEqual(@as(u32, 2), r.findings[0].count); try testing.expectEqual(@as(u32, 4), r.findings[0].first_line); try testing.expectEqual(@as(u32, 5), r.findings[0].extra_lines[0]); } test "checkSemantic: fileCheck findings merge with raw-pass findings" { const data = \\#!srfv1 \\name::a \\name::b \\name::c,nmae::x \\ ; var r = try check(testing.allocator, data, shapeOfSchema(limited_schema)); defer r.deinit(); try checkSemantic(limited_schema, &r, data, test_ctx); r.sort(); try testing.expectEqual(@as(usize, 2), r.findings.len); try testing.expectEqual(Kind.semantic, r.findings[0].kind); // key "name" < "nmae" try testing.expectEqual(Kind.unknown, r.findings[1].kind); }