zfin/build/gen_config_docs.zig
Emil Lerch 40869cb3a4
lint srf files in audit / doctor commands
This also provides ensurance through zig build test that all
fields are documented in markdown so they do not get out of sync
2026-09-24 06:39:23 -07:00

139 lines
5 KiB
Zig

/// Build-time generator: extracts the set of backticked identifiers
/// from each `docs/reference/config/*.md` reference page and emits them
/// as a Zig source file.
///
/// Exists because `@embedFile` cannot escape the package path, so
/// `src/` code has no way to read `docs/`. The alternative - reading
/// the markdown at test time via cwd - would make the doc-sync check
/// dependent on where the test binary was launched from, and a check
/// that can silently skip is a check that rots.
///
/// Only the NAME SET is extracted, never the prose. That keeps the
/// output free of string-escaping concerns and small (a few hundred
/// identifiers), and it is exactly what the consumer needs:
/// `srf_lint`'s doc-sync test asserts every field of each model appears
/// in that model's reference page.
///
/// An "identifier" here is a backtick-delimited token matching
/// `[a-z][a-z0-9_]*` - the shape of every SRF field name. Prose words
/// in backticks (`true`, `null`, `stock`) come along harmlessly; the
/// test only ever asks whether a known field name is PRESENT, never
/// whether an extracted name is a real field.
const std = @import("std");
pub fn main(init: std.process.Init) !void {
const allocator = init.arena.allocator();
const io = init.io;
const args = try init.minimal.args.toSlice(allocator);
if (args.len < 3) {
var stderr_buf: [160]u8 = undefined;
var stderr = std.Io.File.stderr().writer(io, &stderr_buf);
try stderr.interface.writeAll("Usage: gen_config_docs <output.zig> <doc.md>...\n");
try stderr.interface.flush();
std.process.exit(1);
}
const out_file = try std.Io.Dir.cwd().createFile(io, args[1], .{});
defer out_file.close(io);
var out_buf: [4096]u8 = undefined;
var file_writer = out_file.writer(io, &out_buf);
const w = &file_writer.interface;
try w.writeAll(
\\// Auto-generated from docs/reference/config/*.md - do not edit.
\\// Regenerate: zig build (runs build/gen_config_docs.zig)
\\
\\/// One reference page's backticked identifiers.
\\pub const Doc = struct {
\\ /// Basename, e.g. "portfolio-srf.md".
\\ file: []const u8,
\\ names: []const []const u8,
\\};
\\
\\pub const docs = [_]Doc{
\\
);
for (args[2..]) |path| {
const body = try std.Io.Dir.cwd().readFileAlloc(io, path, allocator, .limited(4 * 1024 * 1024));
const base = std.fs.path.basename(path);
var names: std.ArrayList([]const u8) = .empty;
try collectIdents(allocator, body, &names);
try w.print(" .{{ .file = \"{s}\", .names = &.{{", .{base});
for (names.items, 0..) |n, i| {
if (i % 6 == 0) try w.writeAll("\n ");
try w.print("\"{s}\", ", .{n});
}
try w.writeAll("\n } },\n");
}
try w.writeAll(
\\};
\\
\\/// The `Doc` for `file`, or null when that page was not built in.
\\pub fn find(file: []const u8) ?Doc {
\\ for (docs) |d| {
\\ if (std.mem.eql(u8, d.file, file)) return d;
\\ }
\\ return null;
\\}
\\
\\const std = @import("std");
\\
);
try w.flush();
}
/// Append every deduplicated backticked `[a-z][a-z0-9_]*` token in
/// `body` to `out`. Slices borrow from `body`.
fn collectIdents(
allocator: std.mem.Allocator,
body: []const u8,
out: *std.ArrayList([]const u8),
) !void {
var i: usize = 0;
while (i < body.len) {
if (body[i] != '`') {
i += 1;
continue;
}
// Skip fenced code blocks wholesale - a ```zig sample can
// mention a field in a context the reference table does not,
// and counting it would let a field be "documented" by an
// example alone.
if (std.mem.startsWith(u8, body[i..], "```")) {
const rest = body[i + 3 ..];
const close = std.mem.indexOf(u8, rest, "```") orelse break;
i += 3 + close + 3;
continue;
}
const rest = body[i + 1 ..];
const close = std.mem.indexOfScalar(u8, rest, '`') orelse break;
const raw = rest[0..close];
i += 1 + close + 1;
// The docs name a field both bare (`symbol`, in reference
// tables) and on-wire (`symbol::`, `shares:num:100`, in prose
// and examples). Cut at the first separator so both spellings
// count as documenting the field.
const tok = raw[0 .. std.mem.indexOfScalar(u8, raw, ':') orelse raw.len];
if (!isIdent(tok)) continue;
for (out.items) |seen| {
if (std.mem.eql(u8, seen, tok)) break;
} else {
try out.append(allocator, tok);
}
}
}
fn isIdent(tok: []const u8) bool {
if (tok.len == 0) return false;
if (tok[0] < 'a' or tok[0] > 'z') return false;
for (tok) |c| {
const ok = (c >= 'a' and c <= 'z') or (c >= '0' and c <= '9') or c == '_';
if (!ok) return false;
}
return true;
}