zfin/src/service.zig
Emil Lerch 1a9a27156b
All checks were successful
Generic zig build / build (push) Successful in 5m21s
Generic zig build / publish-macos (push) Successful in 12s
Generic zig build / deploy (push) Successful in 20s
fix flaky test
2026-09-30 10:30:30 -07:00

7395 lines
348 KiB
Zig

//! DataService -- unified data access layer for zfin.
//!
//! Encapsulates the "check cache -> fresh? return -> else fetch from provider -> cache -> return"
//! pattern that was previously duplicated between CLI and TUI. Both frontends should use this
//! as their sole data source.
//!
//! Provider selection is internal: each data type routes to the appropriate provider
//! based on available API keys. Callers never need to know which provider was used.
const std = @import("std");
const builtin = @import("builtin");
const log = std.log.scoped(.service);
const Date = @import("Date.zig");
const Candle = @import("models/candle.zig").Candle;
const Dividend = @import("models/dividend.zig").Dividend;
const Split = @import("models/split.zig").Split;
const OptionsChain = @import("models/option.zig").OptionsChain;
const EarningsEvent = @import("models/earnings.zig").EarningsEvent;
const Quote = @import("models/quote.zig").Quote;
const EtfProfile = @import("models/etf_profile.zig").EtfProfile;
const Holding = @import("models/etf_profile.zig").Holding;
const SectorWeight = @import("models/etf_profile.zig").SectorWeight;
const Config = @import("Config.zig");
const cache = @import("cache/store.zig");
const freshness = @import("cache/freshness.zig");
const srf = @import("srf");
const srf_opts = @import("srf_opts.zig");
const analysis = @import("analytics/analysis.zig");
const transaction_log = @import("models/transaction_log.zig");
const TwelveData = @import("providers/twelvedata.zig").TwelveData;
const Polygon = @import("providers/polygon.zig").Polygon;
const Fmp = @import("providers/fmp.zig").Fmp;
const Cboe = @import("providers/cboe.zig").Cboe;
const OpenFigi = @import("providers/openfigi.zig");
const Yahoo = @import("providers/yahoo.zig").Yahoo;
const Tiingo = @import("providers/tiingo.zig").Tiingo;
const Wikidata = @import("providers/Wikidata.zig");
const Edgar = @import("providers/Edgar.zig");
const classification = @import("models/classification.zig");
const fmt = @import("format.zig");
const performance = @import("analytics/performance.zig");
const http = @import("net/http.zig");
const atomic = @import("atomic.zig");
const market = @import("market.zig");
// File-as-struct: the import binds the whole file struct, so its inline
// tests are discovered by the test runner (see AGENTS.md "Test
// discovery").
const LiveStream = @import("net/LiveStream.zig");
// ── Wall-clock policy ────────────────────────────────────────
//
// `FetchResult.timestamp` records when a given fetch or cached-read
// completed. Each `std.Io.Timestamp.now(self.io, .real)` call in
// this file stamps one specific fetch - a single command invocation
// produces many fetches, each with its own real-time stamp. Threading
// `now_s` in from the caller would collapse all per-fetch timestamps to
// the command-entry time, which is not what callers want when they
// display "fetched 3s ago" for some symbols and "cached 2d ago" for
// others in the same command.
pub const DataError = error{
NoApiKey,
FetchFailed,
CacheError,
ParseError,
OutOfMemory,
/// Transient provider failure (server error, connection issue).
/// Caller should stop and retry later.
TransientError,
/// Provider auth failure (bad API key). Entire refresh should stop.
AuthError,
/// Provider returned a rate-limit response (e.g. SEC EDGAR's
/// 10-req/sec ceiling, or a free-tier candle API's per-minute
/// cap). Caller should stop the current batch and surface a
/// "try again later" message;
/// retrying immediately will just hit the same limit.
RateLimited,
/// Provider responded but doesn't have data for the requested
/// symbol (404, "Error Message" body, or equivalent). Distinct
/// from `FetchFailed` so callers (e.g. `enrich`) can tell the
/// user "this symbol isn't in the provider's catalog; mark it
/// manually" instead of an opaque "fetch failed."
NotFound,
};
/// Per-call options controlling cache vs network behavior. Drives
/// the `--refresh-data` global flag's three modes:
///
/// - `--refresh-data=auto` -> `.{}` (default; respect TTL, fetch on stale/miss).
/// - `--refresh-data=never` -> `.{ .skip_network = true }` (offline mode;
/// return cached data even if stale, treat cache miss as unavailable).
/// - `--refresh-data=force` -> `.{ .force_refresh = true }` (ignore cache TTL,
/// fetch fresh from provider).
///
/// `skip_network` and `force_refresh` represent contradictory intents.
/// The CLI flag cannot produce the combination - `RefreshPolicy` is a
/// 3-variant enum, so the user can never set both. But because the
/// underlying shape is two independent booleans, an internal caller
/// constructing `FetchOptions` directly *could* produce the
/// combination. When both are true, **`skip_network` wins**:
///
/// - The call returns cached data (fresh or stale, whatever's there).
/// - `force_refresh` has no effect - no network is touched.
///
/// This is the safe default: when in doubt, don't reach the network.
/// Internal callers that genuinely want fresh data should set
/// `force_refresh = true, skip_network = false`.
pub const FetchOptions = struct {
/// Skip provider fetches and server sync. Returns cached data
/// (even if stale) or null/empty on cache miss. Wins over
/// `force_refresh` when both are set.
skip_network: bool = false,
/// Force a fresh fetch ignoring cache TTL. No-op when
/// `skip_network` is also set.
force_refresh: bool = false,
};
/// Decide whether a provider failure is permanent enough to merit a
/// negative-cache entry. Negative entries suppress retries until the
/// next manual `--refresh-data=force` / `cache clear`, so writing one is only
/// safe when we're confident more attempts won't succeed.
///
/// Today the only certain-permanent failure is `NotFound`: the symbol
/// just doesn't have data of this type at this provider. Everything
/// else (rate limit, network blip, server 5xx, auth, parse error) is
/// either transient or fixable; recording a negative entry would
/// silently suppress retries for hours/days.
///
/// Rate-limit (`error.RateLimited`) is excluded here because callers
/// handle it specially (single retry after backoff). Anything that
/// reaches this classifier and isn't `NotFound` returns false ->
/// caller returns `FetchFailed` without poisoning the cache.
pub fn isPermanentProviderFailure(err: anyerror) bool {
return err == error.NotFound;
}
/// Spread applied to `Ttl.tiingo_backoff` when a symbol is demoted off
/// Tiingo by a 404.
///
/// Policy lives here rather than in `TtlSpec` (see its doc comment).
/// 7% of 30 days is roughly +/-2 days, so a batch of symbols demoted in
/// the same window re-probes across five distinct days instead of all
/// on one. Sized against the cron cadence: with the refresh running
/// twice a day, five days of spread keeps the re-probe burst to a
/// handful of extra 404s per run.
const tiingo_backoff_jitter_pct: u8 = 7;
/// Result of a CUSIP-to-ticker lookup (provider-agnostic).
pub const CusipResult = OpenFigi.FigiResult;
/// Result of an EDGAR ticker-map fallback lookup. Returned by
/// `DataService.lookupEdgarFallback` so commands consume a
/// digested shape instead of pulling in `TickerMap` /
/// `MutualFundTickerEntry` / `CompanyTickerEntry` (those are
/// provider-internal).
///
/// `enrich` uses this to decide what metadata.srf line to emit
/// when Wikidata had no match for a symbol.
pub const EdgarLookup = union(enum) {
/// Symbol matched the EDGAR mutual-fund / managed-fund map.
/// Generic "Fund" label (the `tickers_funds.srf` file mixes
/// mutual funds and series-of-trust ETFs; we can't tell
/// which without digging into submissions metadata).
managed_fund,
/// Symbol matched the EDGAR company / UIT map. `title` is
/// the entry's `title` (e.g. "SPDR S&P 500 ETF TRUST"),
/// allocated by the service's allocator - caller frees with
/// `freeEdgarLookup` when done. The `is_etf` flag is set
/// when the title contains "ETF" or "TRUST" - operating
/// companies usually have Wikidata coverage and wouldn't
/// reach this fallback, so a UIT-style hit is almost
/// certainly an ETF.
company_or_uit: struct { title: ?[]const u8, is_etf: bool },
/// Symbol not in either EDGAR map.
none,
};
/// Free any owned strings inside an `EdgarLookup`. Currently
/// only `.company_or_uit.title` is owned; `.managed_fund` and
/// `.none` are no-ops.
pub fn freeEdgarLookup(allocator: std.mem.Allocator, lookup: EdgarLookup) void {
switch (lookup) {
.company_or_uit => |c| if (c.title) |t| allocator.free(t),
.managed_fund, .none => {},
}
}
/// Look up `sym` in the supplied EDGAR ticker maps. Pure data
/// transform; no I/O. Returns the borrowing-shape result.
///
/// Both maps may be null (caller failed to load one or both).
/// A null map produces a `none` result for that pass.
///
/// On `.company_or_uit`, the returned `title` is duped from the
/// underlying entry using `allocator` so the caller can use it
/// after the maps are freed. Free with `freeEdgarLookup`.
fn lookupInTickerMaps(
allocator: std.mem.Allocator,
sym: []const u8,
mf_map: ?*const Edgar.TickerMap(Edgar.MutualFundTickerEntry),
co_map: ?*const Edgar.TickerMap(Edgar.CompanyTickerEntry),
) EdgarLookup {
if (mf_map) |m| {
if (m.get(sym)) |_| return .managed_fund;
}
if (co_map) |m| {
if (m.get(sym)) |entry| {
const title_owned: ?[]const u8 = if (entry.title) |t|
allocator.dupe(u8, t) catch null
else
null;
const title_for_check = title_owned orelse "";
const is_etf =
std.ascii.indexOfIgnoreCase(title_for_check, "ETF") != null or
std.ascii.indexOfIgnoreCase(title_for_check, "TRUST") != null;
return .{ .company_or_uit = .{ .title = title_owned, .is_etf = is_etf } };
}
}
return .none;
}
/// Indicates whether the returned data came from cache or was freshly fetched.
pub const Source = enum {
cached,
fetched,
};
/// In-memory payload shape for a fetched type `T`.
///
/// Almost everything is a slice of records (`[]Candle`, `[]Dividend`,
/// ...) - the same shape the cache stores. `EtfProfile` is the lone
/// exception: `getEtfProfile` assembles a single struct from the
/// `etf_metrics` cache rather than returning a slice, so its payload
/// is the struct itself. The cache layer never stores `EtfProfile`
/// directly, which is why this single-struct knowledge lives here in
/// the fetch layer rather than in `Store.DataFor`.
fn PayloadFor(comptime T: type) type {
return if (T == EtfProfile) EtfProfile else []T;
}
/// Generic result type for all fetch operations: data payload + provenance metadata.
///
/// `data` is owned by `allocator` - call `result.deinit()` to release
/// it (both the outer slice/struct and any nested owned fields). This
/// replaces the earlier "caller frees with whatever allocator they
/// happen to have" pattern, which was error-prone when the caller's
/// allocator (e.g. an arena) differed from the service's allocator.
pub fn FetchResult(comptime T: type) type {
return struct {
data: PayloadFor(T),
source: Source,
timestamp: i64,
/// Allocator that owns `data`. Populated by the service on
/// every return path; callers use it via `deinit` rather than
/// touching it directly.
allocator: std.mem.Allocator,
/// Free `data` and any nested owned fields.
///
/// Dispatches at comptime:
/// - If `T` has a `freeSlice` helper (Dividend, OptionsChain),
/// call it - handles element deinit plus the outer slice.
/// - Else if `data` is a slice (Candle, Split, EarningsEvent),
/// do a simple slice free.
/// - Else if `T` has a `deinit` method (EtfProfile), call it
/// on the struct itself.
pub fn deinit(self: @This()) void {
const DT = @TypeOf(self.data);
if (@hasDecl(T, "freeSlice")) {
T.freeSlice(self.allocator, self.data);
} else if (@typeInfo(DT) == .pointer) {
self.allocator.free(self.data);
} else if (@hasDecl(T, "deinit")) {
self.data.deinit(self.allocator);
}
}
};
}
// ── PostProcess callbacks ────────────────────────────────────
// `Store.read` parses with `parse_allocator = .{ .allocator = ... }`,
// so SRF dupes every owned string into the caller's allocator
// automatically. PostProcess callbacks remain only for non-trivial
// post-parse logic (e.g. recomputing derived fields). String duping
// is NOT a valid reason to add a postProcess.
/// Recompute surprise/surprise_percent from actual and estimate fields.
/// SRF only stores actual and estimate; surprise is derived.
fn earningsPostProcess(ev: *EarningsEvent, _: std.mem.Allocator) anyerror!void {
if (ev.actual != null and ev.estimate != null) {
ev.surprise = ev.actual.? - ev.estimate.?;
if (ev.estimate.? != 0) {
ev.surprise_percent = (ev.surprise.? / @abs(ev.estimate.?)) * 100.0;
}
}
}
pub const DataService = struct {
/// Thread-safe wrapper over the caller-provided base allocator.
///
/// Why this exists: `parallelServerSync` spawns worker threads that
/// each allocate through `DataService` - HTTP client init, TLS cert
/// bundle parsing, request/response buffers, and `Store.writeRaw`
/// path joins. The CLI's root allocator is an `ArenaAllocator`
/// (`src/main.zig`), which is NOT thread-safe. Unsynchronized
/// concurrent allocs from workers corrupt the arena's free list.
/// Symptoms seen in the wild:
///
/// thread N panic: reached unreachable code
/// std/mem/Allocator.zig:147 grow
/// std/hash_map.zig:1296 addCertsFromFile
/// std/crypto/Certificate/Bundle.zig:206 request
/// std/http/Client.zig:1789 request
/// src/net/http.zig:43 syncFromServer
///
/// and bare segfaults mid-heap on whatever pointer the arena
/// scrambled that run.
///
/// The wrapper serializes every allocation with a mutex. Cost is
/// one lock acquire/release per alloc - negligible next to the I/O
/// Thread-safe allocator used for all DataService-internal allocations.
///
/// In Zig 0.16, the Juicy-Main-provided `init.gpa` (DebugAllocator)
/// is thread-safe by default when not single-threaded, and
/// `ArenaAllocator` is thread-safe and lock-free. Callers should
/// pass whichever thread-safe allocator is appropriate - we no
/// longer wrap it ourselves.
///
/// DO NOT add an "unwrap" method or pass a non-thread-safe
/// allocator. The point is that internal callers don't need to
/// know whether they're running under threads - the allocator
/// itself guarantees safety.
allocator: std.mem.Allocator,
io: std.Io,
config: Config,
// Lazily initialized providers (null until first use)
td: ?TwelveData = null,
pg: ?Polygon = null,
fmp: ?Fmp = null,
cboe: ?Cboe = null,
yh: ?Yahoo = null,
tg: ?Tiingo = null,
wikidata: ?Wikidata = null,
edgar: ?Edgar = null,
/// Live-price websocket stream (push-based intraday quotes), lazily
/// created by `startLiveStream`. Heap-owned (the stream's background
/// thread captures the pointer, so it must be stable). Torn down by
/// `stopLiveStream` / `deinit`.
live_stream: ?*LiveStream = null,
/// Test-only guard: when true, any code path that would touch
/// the network panics with a clear message. Used by offline-mode
/// tests to verify that `FetchOptions.skip_network = true`
/// genuinely doesn't reach the network. Default false; never
/// set in production.
panic_on_network_attempt: bool = false,
/// Service-wide fetch policy, from the global `--refresh-data` flag.
///
/// A call site passing `.{}` is not opting OUT of the user's instruction -
/// it simply has no opinion, so this applies. Without it, a documented
/// GLOBAL flag silently stopped at any call site that did not thread the
/// option through: `zfin --refresh-data=force projections` did not
/// force-refresh the benchmark symbols, because `views/projections.zig`
/// fetched them with a hardcoded `.{}`.
///
/// Threading the option to those call sites instead would have meant ~7
/// signatures and ~19 edits across projections, compare and the TUI, and
/// would have left the next hardcoded `.{}` with the same bug.
///
/// **Defaults to all-false, which is what keeps tests offline.** Test
/// Configs are keyless literals so a provider fetch already fails with
/// `NoApiKey` before any HTTP, and this field cannot loosen that: an
/// all-false policy ORs to a no-op. A test has to set it deliberately.
default_options: FetchOptions = .{},
pub fn init(io: std.Io, allocator: std.mem.Allocator, config: Config) DataService {
const self = DataService{
.allocator = allocator,
.io = io,
.config = config,
};
// Missing-key warnings are noise under `zig build test` where
// every test that spins up a DataService re-emits the whole
// block. Real users always see them at CLI/TUI startup.
if (!builtin.is_test) self.logMissingKeys();
return self;
}
/// Log warnings for missing API keys so users know which features are unavailable.
fn logMissingKeys(self: DataService) void {
// Primary candle provider
if (self.config.tiingo_key == null) {
log.warn("TIINGO_API_KEY not set - candle data will fall back to TwelveData/Yahoo", .{});
}
// Dividend/split data
if (self.config.polygon_key == null) {
log.warn("POLYGON_API_KEY not set - dividend and split data unavailable", .{});
}
// Earnings data
if (self.config.fmp_key == null) {
log.warn("FMP_API_KEY not set - earnings data unavailable", .{});
}
// ETF profiles + portfolio enrichment now go through public
// SEC EDGAR + Wikidata. Both require a contact email in
// outbound User-Agents (SEC's policy).
if (self.config.user_email == null) {
log.warn("ZFIN_USER_EMAIL not set - ETF profiles + enrichment unavailable", .{});
}
// Candle fallback
if (self.config.twelvedata_key == null and self.config.tiingo_key == null) {
log.warn("TWELVEDATA_API_KEY not set - no candle fallback if Yahoo fails", .{});
}
// CUSIP lookups
if (self.config.openfigi_key == null) {
log.info("OPENFIGI_API_KEY not set - CUSIP lookups will use anonymous rate limits", .{});
}
}
pub fn deinit(self: *DataService) void {
if (self.live_stream) |s| s.destroy();
if (self.td) |*td| td.deinit();
if (self.pg) |*pg| pg.deinit();
if (self.fmp) |*fmp| fmp.deinit();
if (self.cboe) |*c| c.deinit();
if (self.yh) |*yh| yh.deinit();
if (self.tg) |*tg| tg.deinit();
if (self.wikidata) |*w| w.deinit();
if (self.edgar) |*e| e.deinit();
}
// ── Provider accessor ──────────────────────────────────────────
fn getProvider(self: *DataService, comptime T: type) DataError!*T {
const field_name = comptime providerField(T);
if (@field(self, field_name)) |*p| return p;
if (T == Cboe or T == Yahoo) {
// CBOE and Yahoo have no API key
@field(self, field_name) = T.init(self.io, self.allocator);
} else if (T == Wikidata or T == Edgar) {
// Open-data providers identified by contact email rather
// than an API key. The email goes in User-Agent + From
// headers per each provider's politeness contract.
const email = self.config.user_email orelse return DataError.NoApiKey;
@field(self, field_name) = T.init(self.io, self.allocator, email);
} else if (T == Tiingo) {
// Tiingo takes a plan-driven hourly rate limit alongside the
// key. The cap comes from `Config.tiingoHourlyLimit` (free
// 50/hour, Power 10,000/hour) so a paying subscriber isn't
// throttled to free-tier limits.
const key = self.config.tiingo_key orelse return DataError.NoApiKey;
@field(self, field_name) = Tiingo.init(self.io, self.allocator, key, .{
.per_hour = self.config.tiingoHourlyLimit(),
});
} else {
// All we're doing here is lower casing the type name, then
// appending _key to it, so Tiingo -> tiingo_key
const config_key = comptime blk: {
const full = @typeName(T);
var start: usize = 0;
for (full, 0..) |c, i| {
if (c == '.') start = i + 1;
}
const short = full[start..];
var buf: [short.len + 4]u8 = undefined;
_ = std.ascii.lowerString(buf[0..short.len], short);
@memcpy(buf[short.len..][0..4], "_key");
break :blk buf[0 .. short.len + 4];
};
const key = @field(self.config, config_key) orelse return DataError.NoApiKey;
@field(self, field_name) = T.init(self.io, self.allocator, key);
}
return &@field(self, field_name).?;
}
fn providerField(comptime T: type) []const u8 {
inline for (std.meta.fields(DataService)) |f| {
if (f.type == ?T) return f.name;
}
@compileError("unknown provider type");
}
// ── Cache helper ─────────────────────────────────────────────
fn store(self: *DataService) cache.Store {
return cache.Store.init(self.io, self.allocator, self.config.cache_dir);
}
/// LEVEL 3 of the three-level freshness check: the shortest gap
/// between two provider asks for the same symbol, when the ask is
/// driven by the Level 2 completeness check rather than by the Level 1
/// expiry clock. Applies to every type with a `needsRefresh` hook -
/// dividends and earnings today.
///
/// WHY THIS EXISTS. When a hook says something is overdue, we ask the
/// provider. Often the provider does not have it yet - that is the
/// normal case, not a failure, because an ETF publishes its amount
/// around its own ex-date and FMP posts an actual some hours after the
/// report. The answer is "nothing new", and the cache gets rewritten
/// with a fresh clock but the SAME records. So the hook still says
/// "overdue", and the very next command asks again.
///
/// Without a floor that is one provider call per command, not per day.
/// A run touching fifteen symbols in their due windows makes fifteen
/// calls, and the next run ninety seconds later makes fifteen more. At
/// Polygon's four-per-minute limit that is four minutes of waiting,
/// repeated.
///
/// WHY TWELVE HOURS. The floor is also the worst case for how late we
/// notice something, so it has to stay well inside the shortest gap
/// between an event and the money moving. For dividends the tightest
/// real ex-to-pay gap is one day (QTUM), so twelve hours leaves half a
/// day of margin. Longer buys almost nothing: asking once a day is
/// already far more often than the weekly reconcile that uses the
/// answer.
const smart_refresh_recheck_interval: i64 = 12 * std.time.s_per_hour;
/// LEVELS 2 AND 3 of the three-level freshness check. Returns true to
/// serve the cached entry as-is; returns false - having already freed
/// `data` - to discard it and refetch.
///
/// ── The three levels, and why the order reads backwards ──────
///
/// Level 1 expiry clock "Is this entry old?"
/// Level 2 completeness check "Is it missing something it should
/// already have?"
/// Level 3 recheck floor "Did we just ask the provider?"
///
/// LEVEL 2 IS A SECOND OPINION ON A "STILL FRESH" VERDICT, not an
/// extra gate in front of a refresh. That is the thing to get right,
/// because the intuition runs the other way.
///
/// This function is only ever called on a FRESH entry, because
/// `fetchCached` reaches it through `Store.read(.fresh_only)` - which
/// returns null when Level 1 says the entry is stale. So when Level 1
/// says "stale", Levels 2 and 3 never run at all; a refetch is already
/// happening and they have nothing to contribute. Level 2 only earns
/// its keep in the opposite case: Level 1 says "this is fine, serve
/// it", and Level 2 gets to answer "no it is not - a payment has
/// happened that this entry does not contain".
///
/// Level 3 then exists to stop Level 2 being too eager, and can
/// overrule it back to "serve the cache".
///
/// `skip_network` short-circuits between Levels 1 and 2: offline mode
/// never refetches, so discarding a usable entry could only turn a
/// served result into a failure.
///
/// One function rather than two inline blocks because `fetchCached`
/// has two fresh-cache read sites (local, and post-server-sync) and
/// the rule must be identical at both. It owns the free so a caller
/// cannot decide to discard and then leak.
fn serveFreshOrDiscard(
self: *DataService,
comptime T: type,
comptime needsRefresh: ?*const fn ([]const T, Date) bool,
comptime ttlFor: ?*const fn ([]const T) cache.TtlSpec,
symbol: []const u8,
cached: cache.Store.CacheResult(T),
skip_network: bool,
) bool {
if (needsRefresh) |hook| {
if (skip_network) return true;
// wall-clock required: the hook asks whether a scheduled
// event has come due, which is a question about the actual
// current day. Threading `today` down would put a date
// parameter on four public getters for one type's benefit.
if (!hook(cached.data, fmt.todayDate(self.io))) return true;
if (self.askedRecently(T, ttlFor, symbol, cached)) {
log.debug("{s}: {s} is overdue but we asked within the last {d}h; serving cache", .{ symbol, @tagName(comptime cache.Store.dataTypeFor(T)), @divFloor(smart_refresh_recheck_interval, std.time.s_per_hour) });
return true;
}
T.freeSlice(self.allocator, cached.data);
return false;
}
return true;
}
/// LEVEL 3 of the three-level freshness check: did we already ask the
/// provider about this symbol inside `smart_refresh_recheck_interval`?
///
/// Only reached when Level 2 has already said the entry is incomplete,
/// so a true answer here means "incomplete, but asking again this soon
/// would be wasted" - see `serveFreshOrDiscard` for the full ordering.
///
/// WHY `#!expires=` AND NOT `#!created=`. Both directives sit in the
/// same file and `created` looks like the obvious choice, but it is
/// the wrong one. `serializeWithMeta` re-stamps `created` on EVERY
/// write, and dividend rows also arrive as a side effect of a Tiingo
/// candle fetch (`writeSupplement`). So `created` answers "when was
/// this file last touched", which a candle refresh changes without
/// anyone asking Polygon anything.
///
/// `expires` answers the question we actually want. Only a fetch from
/// the primary provider moves it (`ExpiryPolicy.bump`); a supplement
/// deliberately puts the old value back (`.preserve`). So it marks the
/// last time we really asked.
///
/// Recovering the write time from it is exact rather than approximate,
/// because `computeExpires` derives its jitter from a hash of the
/// symbol - the same symbol always gets the same offset. Passing a
/// zero "now" therefore returns precisely the TTL that was applied,
/// and subtracting it from `expires` gives the write time to the
/// second.
///
/// Two cases fall back to "no, ask again", which is the safe
/// direction:
/// - No `#!expires=` in the file at all.
/// - The symbol crossed from unknown-schedule to known-schedule
/// since it was written, so it was stored under the short TTL and
/// we recompute with the long one. That makes the write look
/// older than it was, so we ask sooner than strictly needed.
fn askedRecently(
self: *DataService,
comptime T: type,
comptime ttlFor: ?*const fn ([]const T) cache.TtlSpec,
symbol: []const u8,
cached: cache.Store.CacheResult(T),
) bool {
const expires = cached.expires orelse return false;
const spec = if (ttlFor) |f| f(cached.data) else comptime cache.Store.dataTypeFor(T).ttl();
// `computeExpires` from a zero epoch returns the jittered TTL
// itself, which is what makes this exact - see the doc above.
const applied_ttl = cache.computeExpires(0, spec, symbol);
const written_at = expires - applied_ttl;
// wall-clock required: "recently" is relative to now.
const now_s = std.Io.Timestamp.now(self.io, .real).toSeconds();
return now_s - written_at < smart_refresh_recheck_interval;
}
/// The TTL to stamp on a write: from the records when the type
/// supplies a `ttlFor` hook, otherwise from the type alone.
///
/// Separate from the call sites because `fetchCached` writes in two
/// places (the normal path and the rate-limit retry) and they must
/// agree - a retry that stamped a different clock than the first
/// attempt would be a silently different freshness policy for
/// nobody's benefit.
fn ttlSpecFor(
comptime T: type,
comptime ttlFor: ?*const fn ([]const T) cache.TtlSpec,
items: cache.Store.DataFor(T),
) cache.TtlSpec {
if (ttlFor) |f| return f(items);
return comptime cache.Store.dataTypeFor(T).ttl();
}
/// Generic fetch-or-cache for simple data types (dividends, splits, options).
/// Checks cache first; on miss, fetches from the appropriate provider,
/// writes to cache, and returns. On permanent fetch failure, writes a negative
/// cache entry. Rate limit failures are retried once.
///
/// `opts.skip_network = true` -> returns cached data even if stale,
/// returns FetchFailed on cache miss without touching the network.
/// `opts.force_refresh = true` -> treats cache as stale and fetches.
///
/// `needsRefresh` is LEVEL 2 of the three-level freshness check: given
/// a FRESH cache entry it answers "is this nonetheless incomplete?".
/// It exists because freshness and completeness are different
/// questions - `dividendsNeedRefresh` documents the case that forced
/// it. Null means the expiry clock is the only gate, which is the
/// behaviour every type had before the hook existed.
///
/// NOTE WHERE IT SITS. Levels 2 and 3 live behind the
/// `Store.read(.fresh_only)` calls below, which return null when Level
/// 1 (the expiry clock) says the entry is stale. So a stale entry
/// bypasses them entirely - they can only ever overrule a "still
/// fresh" verdict, never reinforce a stale one. `serveFreshOrDiscard`
/// has the full ordering.
///
/// The hook's branch is comptime-elided for a null argument, so types
/// with no `freeSlice` (Split) stay compilable.
///
/// `ttlFor` sets LEVEL 1's length from the records being written
/// rather than from the type alone. Only dividends need it, because
/// the right clock depends on whether the records establish a payment
/// schedule - see `dividendTtl`. Null means "use `DataType.ttl()`",
/// which is what every other type does.
fn fetchCached(
self: *DataService,
comptime T: type,
symbol: []const u8,
comptime postProcess: ?*const fn (*T, std.mem.Allocator) anyerror!void,
comptime needsRefresh: ?*const fn ([]const T, Date) bool,
comptime ttlFor: ?*const fn ([]const T) cache.TtlSpec,
opts_in: FetchOptions,
) DataError!FetchResult(T) {
// See `getCandles` - one fold, covering every type routed through here.
const opts = self.effectiveOptions(opts_in);
var s = self.store();
const data_type = comptime cache.Store.dataTypeFor(T);
// Force-refresh skips the fresh-cache early return; falls
// through to provider fetch. Skip-network does the opposite:
// returns cached even if stale, never touches the network.
if (!opts.force_refresh) {
if (s.read(self.allocator, T, symbol, postProcess, .fresh_only)) |cached| {
if (self.serveFreshOrDiscard(T, needsRefresh, ttlFor, symbol, cached, opts.skip_network)) {
log.debug("{s}: {s} fresh in local cache", .{ symbol, @tagName(data_type) });
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
log.debug("{s}: {s} fresh in local cache but a scheduled event is overdue; refetching", .{ symbol, @tagName(data_type) });
}
}
if (opts.skip_network) {
// Offline mode: return whatever's cached, even if stale.
// Cache miss is FetchFailed (not a network error).
if (s.read(self.allocator, T, symbol, postProcess, .any)) |cached| {
log.info("{s}: {s} stale-cached returned (skip_network)", .{ symbol, @tagName(data_type) });
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
return DataError.FetchFailed;
}
// Try server sync before hitting providers (skipped on force_refresh).
if (!opts.force_refresh and self.syncFromServer(symbol, data_type)) {
if (s.read(self.allocator, T, symbol, postProcess, .fresh_only)) |cached| {
// The hook applies here too. Without it a configured
// ZFIN_SERVER defeats the whole mechanism: the sync
// leaves a still-incomplete entry on disk, this read
// finds it fresh, and the overdue distribution is served
// anyway. Re-asking is also the cheap outcome when the
// server DID have the newer record - that is the tier's
// reason for existing, and we return without spending a
// provider request.
if (self.serveFreshOrDiscard(T, needsRefresh, ttlFor, symbol, cached, opts.skip_network)) {
log.debug("{s}: {s} synced from server and fresh", .{ symbol, @tagName(data_type) });
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
}
log.debug("{s}: {s} synced from server but stale, falling through to provider", .{ symbol, @tagName(data_type) });
}
log.debug("{s}: fetching {s} from provider", .{ symbol, @tagName(data_type) });
self.assertNetworkAllowed("fetchCached fetchFromProvider");
const fetched = self.fetchFromProvider(T, symbol) catch |err| {
if (err == error.RateLimited) {
// Wait and retry once
self.rateLimitBackoff();
const retried = self.fetchFromProvider(T, symbol) catch |retry_err| {
log.warn("{s}: {s} fetch failed after rate-limit retry: {t}", .{ symbol, @tagName(data_type), retry_err });
return DataError.FetchFailed;
};
s.writeWithSource(T, symbol, retried, ttlSpecFor(T, ttlFor, retried), sourceHintFor(T));
return .{ .data = retried, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
// A MISSING KEY IS A CONFIGURATION FAULT, not a data fault,
// and it is the one distinction callers act on differently:
// `commands/earnings.zig` and `tui/earnings_tab.zig` both
// name the environment variable to set, which they cannot do
// from a generic FetchFailed. Propagate it rather than
// collapsing it, and do not negative-cache - nothing is
// wrong with the symbol.
if (err == DataError.NoApiKey) {
log.warn("{s}: {s} unavailable: no API key configured", .{ symbol, @tagName(data_type) });
return DataError.NoApiKey;
}
// Only NotFound (provider says "this symbol genuinely has
// no data of this type") gets a negative-cache entry.
// Transient failures (network, 5xx, auth misconfig, parse
// error) propagate as FetchFailed without poisoning the
// cache, so the next call retries naturally.
//
// Log the provider's own error either way: the typed return
// collapses to FetchFailed, so this line is the only place the
// distinction between RateLimited, Unauthorized and NotFound
// survives. Callers that swallow failures per symbol (see
// `loadAllDividends`) depend on it.
if (isPermanentProviderFailure(err)) {
// The normal "this symbol has no data of this type" outcome.
log.info("{s}: {s} unavailable: {t}", .{ symbol, @tagName(data_type), err });
s.writeNegative(symbol, data_type);
} else {
log.warn("{s}: {s} fetch failed: {t}", .{ symbol, @tagName(data_type), err });
}
return DataError.FetchFailed;
};
s.writeWithSource(T, symbol, fetched, ttlSpecFor(T, ttlFor, fetched), sourceHintFor(T));
return .{ .data = fetched, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
/// Map the model type fetched via `fetchCached` back to the
/// provider it came from, so the merge primitive's `info(cache)`
/// log lines can attribute new entries / field upgrades to a
/// named source. Returns null for types where the source name
/// isn't useful (the merge primitive only consults this for
/// Dividend and Split).
fn sourceHintFor(comptime T: type) ?[]const u8 {
return switch (T) {
Dividend, Split => "polygon",
else => null,
};
}
/// Dispatch a fetch to the correct provider based on model type.
fn fetchFromProvider(self: *DataService, comptime T: type, symbol: []const u8) !cache.Store.DataFor(T) {
return switch (T) {
Dividend => {
// Polygon is the primary source: it carries
// forward-looking declared dividends (e.g. ARCC's
// 2026-06-15 ex_date), which Tiingo's price-series
// response does not. Tiingo opportunistically
// supplements the cache via `populateAllFromTiingo`
// when candle fetches happen - that path uses
// `cache.Store.writeSupplement`, whose sorted-union
// merge lets Polygon's and Tiingo's entries coexist in
// `dividends.srf` without overwriting each other, and
// which preserves the `#!expires=` clock so a Tiingo
// candle fetch never masks a due Polygon refresh.
var pg = try self.getProvider(Polygon);
return pg.fetchDividends(self.allocator, symbol, null, null);
},
Split => {
// Same rationale as Dividend above. Polygon also
// carries forward-looking split announcements that
// Tiingo's price-series doesn't surface.
var pg = try self.getProvider(Polygon);
return pg.fetchSplits(self.allocator, symbol);
},
OptionsChain => {
var cboe = try self.getProvider(Cboe);
return cboe.fetchOptionsChain(self.allocator, symbol);
},
EarningsEvent => {
var fmp = try self.getProvider(Fmp);
return fmp.fetchEarnings(self.allocator, symbol);
},
else => @compileError("unsupported type for fetchFromProvider"),
};
}
/// Fetch candles, dividends, and splits from Tiingo in a single
/// HTTP call and write all three caches. Returns the triple so
/// the caller can use the data without re-reading from disk.
///
/// This is the orchestrated "cold cache" path. `getCandles`
/// (cold-cache full fetch) calls this so a single Tiingo HTTP
/// request populates `candles_daily.srf`, `candles_meta.srf`,
/// `dividends.srf`, and `splits.srf` together. Tiingo's
/// per-row `divCash` and `splitFactor` make this almost free.
///
/// For dividends and splits the writes go through
/// `writeWithSource` with `"tiingo"` as the source hint. The
/// underlying `writeMerged` primitive merges Tiingo's view
/// into whatever's already on disk (typically Polygon-sourced
/// records), preserving forward-looking entries Polygon
/// uniquely carries. New entries trigger an `info(cache)` log
/// line attributing the discovery to Tiingo - useful when
/// Tiingo surfaces a corporate action Polygon missed (the
/// canonical case is SPYM's 2017-10-16 4:1 split).
///
/// `from` is fixed at 2000-01-01 to cover any 10Y trailing-return
/// window even when `--as-of` back-dates the reference to the
/// earliest imported portfolio data (currently 2014). The extra
/// few years of pre-2004 candles cost ~150 KB per symbol on disk
/// and a one-time bandwidth bump on cold-cache fetch, both
/// trivial. Also gives a comfortable buffer for older corporate
/// actions (e.g. SPYM's 2017-10-16 split, deep-history reverse
/// splits on legacy tickers).
fn populateAllFromTiingo(self: *DataService, symbol: []const u8) !@import("providers/tiingo.zig").CandleAndCorporateActions {
var tg = try self.getProvider(Tiingo);
const today = fmt.todayDate(self.io);
const from = Date.fromYmd(2000, 1, 1);
const triple = try tg.fetchCandlesAndCorporateActions(self.allocator, symbol, from, today);
var s = self.store();
// Candles + meta - `cacheCandles` writes both candles_daily.srf
// and candles_meta.srf in one shot (last_close, last_date,
// provider, fail_count=0).
if (triple.candles.len > 0) {
// wall-clock required: stamp the candle-meta freshness boundary
// (market-aware next post-close / NAV-availability time).
const now_s = std.Io.Timestamp.now(self.io, .real).toSeconds();
const kind = market.classify(symbol);
s.cacheCandles(symbol, triple.candles, .{ .provider = .tiingo }, expiryAfterFetch(now_s, kind, triple.candles));
}
// Dividends and splits use the supplement write path: Tiingo's
// view merges into existing (typically Polygon-sourced) records
// without resetting the `#!expires=` freshness clock, since
// Polygon - not this opportunistic candle-driven write - owns
// when div/split data is next due for a primary refresh. New
// entries are logged with "tiingo" attribution.
s.writeSupplement(Dividend, symbol, triple.dividends, "tiingo");
s.writeSupplement(Split, symbol, triple.splits, "tiingo");
return triple;
}
/// Fixed start date for any full-history candle fetch. See
/// `populateAllFromTiingo`'s doc comment for the rationale.
const full_history_start: Date = Date.fromYmd(2000, 1, 1);
/// Replace a symbol's entire candle series, restating `adj_close`
/// from whichever provider will serve it.
///
/// This is the only way a cached series' adjustment basis ever
/// advances - see `CandleMeta.adj_basis`. Incremental appends
/// cannot restate the rows already on disk, so once a distribution
/// goes ex behind them the whole series reads low until something
/// rewrites it. That something is this function.
///
/// Tiingo first (it returns dividends and splits in the same
/// response, and its `adj_close` is the one the analytics layer is
/// written against), falling back to a full-range Yahoo fetch when
/// Tiingo does not carry the symbol.
///
/// **Never writes a negative-cache entry.** Callers reach this
/// holding a working series; `writeNegative` would overwrite
/// `candles_daily.srf` with a marker and destroy that history. A
/// restatement that cannot be completed is a "try again later", not
/// a verdict about the symbol. Callers should treat any error as
/// "keep the existing series and carry on".
///
/// Returns `error.NotFound` only when *every* provider says it does
/// not carry the symbol - that is the one outcome a cold-start
/// caller may legitimately turn into a negative-cache entry. Any
/// other failure surfaces as `FetchFailed` / `TransientError` /
/// `AuthError` so a network blip can never be mistaken for
/// "this symbol does not exist".
fn refetchFullHistory(
self: *DataService,
symbol: []const u8,
today: Date,
now_s: i64,
prefer_yahoo: bool,
) (DataError || error{NotFound})![]Candle {
self.assertNetworkAllowed("getCandles refetchFullHistory");
const kind = market.classify(symbol);
var s = self.store();
// Tracks whether each provider affirmatively said "no such
// symbol", as opposed to failing for some other reason. Only
// unanimous 404s justify the caller writing a negative entry.
var tiingo_not_found = false;
if (!prefer_yahoo) {
if (self.populateAllFromTiingo(symbol)) |triple| {
defer Dividend.freeSlice(self.allocator, triple.dividends);
defer self.allocator.free(triple.splits);
if (triple.candles.len > 0) return triple.candles;
// An empty full-history response is not a restatement;
// fall through rather than caching emptiness.
self.allocator.free(triple.candles);
log.warn("{s}: Tiingo full history returned no bars, trying Yahoo", .{symbol});
} else |err| {
// Transient failures must not silently degrade to a
// second provider - the caller needs to know to retry.
if (err == error.RateLimited or isTransientError(err)) return DataError.TransientError;
if (err == error.Unauthorized) {
log.err("{s}: Tiingo auth failed during restatement - check TIINGO_API_KEY", .{symbol});
return DataError.AuthError;
}
tiingo_not_found = isPermanentProviderFailure(err);
log.info("{s}: Tiingo cannot serve full history ({s}), trying Yahoo", .{ symbol, @errorName(err) });
}
}
// Yahoo fallback. `fetchCandles` takes an arbitrary range, so a
// full-history request is just a wide one. Yahoo carries no
// dividend/split endpoint here; those caches are Polygon-primary
// and merged separately, so leaving them untouched is correct.
if (self.getProvider(Yahoo)) |yh| {
if (yh.fetchCandles(self.allocator, symbol, full_history_start, today)) |candles| {
if (candles.len > 0) {
s.cacheCandles(symbol, candles, .{
.provider = .yahoo,
// Preserve the Tiingo verdict this call just
// learned, so the next pass doesn't re-probe.
.tiingo_retry_after_s = if (tiingo_not_found)
cache.computeExpires(
now_s,
.{ .seconds = cache.Ttl.tiingo_backoff, .jitter_pct = tiingo_backoff_jitter_pct },
symbol,
)
else
0,
}, expiryAfterFetch(now_s, kind, candles));
log.info("{s}: full history restated from Yahoo ({d} bars)", .{ symbol, candles.len });
return candles;
}
self.allocator.free(candles);
log.warn("{s}: Yahoo full history returned no bars", .{symbol});
} else |err| {
log.warn("{s}: Yahoo full history failed: {s}", .{ symbol, @errorName(err) });
// Both providers affirmatively disclaim the symbol.
if (tiingo_not_found and isPermanentProviderFailure(err)) return error.NotFound;
}
} else |_| {
log.warn("{s}: Yahoo provider not available for full history", .{symbol});
}
return DataError.FetchFailed;
}
/// Invalidate cached data for a symbol so the next get* call forces a fresh fetch.
pub fn invalidate(self: *DataService, symbol: []const u8, data_type: cache.DataType) void {
var s = self.store();
s.clearData(symbol, data_type);
// Also clear candle metadata when invalidating candle data
if (data_type == .candles_daily) {
s.clearData(symbol, .candles_meta);
}
}
// ── Public data methods ──────────────────────────────────────
/// What a candle fetch learned about Tiingo's coverage of a symbol.
///
/// This is deliberately three-valued. The old code collapsed
/// "Tiingo does not carry this symbol" and "Tiingo failed this
/// request" into a single permanent demotion to Yahoo, which is
/// how 22 of 32 cached symbols drifted off Tiingo. Only
/// `.not_found` is a statement about the symbol; everything else
/// is a statement about one HTTP call and must not be remembered.
pub const TiingoCoverage = enum {
/// Tiingo served the request - it definitely covers this symbol.
covered,
/// Tiingo returned a genuine 404 - it does not carry this symbol.
not_found,
/// Tiingo was not consulted, or failed for a reason that says
/// nothing about coverage (active backoff, no API key, 400,
/// 402, malformed body). Nothing new learned; leave any
/// existing backoff exactly as it was.
unknown,
};
/// Fold a fetch's outcome into a symbol's candle metadata.
///
/// Called on any *successful* candle fetch, so `fail_count` resets
/// to 0. The interesting part is `tiingo_retry_after_s`:
///
/// - `.covered` -> clear the backoff. Tiingo just served this
/// symbol, so any prior 404 is stale news. This
/// is what walks a previously-demoted symbol
/// back onto Tiingo.
/// - `.not_found` -> arm the backoff at `Ttl.tiingo_backoff` with
/// per-symbol jitter, so a batch demoted in the
/// same window does not all re-probe on one day.
/// - `.unknown` -> leave it untouched. Either Tiingo was never
/// asked, or it failed for a reason that says
/// nothing about coverage.
fn applyTiingoCoverage(
meta: cache.Store.CandleMeta,
symbol: []const u8,
now_s: i64,
provider: cache.Store.CandleProvider,
coverage: TiingoCoverage,
) cache.Store.CandleMeta {
var next = meta;
next.provider = provider;
next.fail_count = 0;
// A provider change and a cleared backoff are independent facts,
// and conflating them hid the first one. This used to log only
// when clearing an armed backoff - but the case that mattered
// was a legacy cache carrying `provider = .yahoo` with no
// backoff at all, which is every symbol that had drifted off
// Tiingo before `tiingo_retry_after_s` existed. Twenty-one
// symbols converted back to Tiingo without a single line.
if (meta.provider != provider) {
log.info("{s}: candle provider {t} -> {t}", .{ symbol, meta.provider, provider });
}
switch (coverage) {
.covered => {
if (meta.tiingo_retry_after_s != 0) {
log.info("{s}: Tiingo serving again, clearing backoff", .{symbol});
}
next.tiingo_retry_after_s = 0;
},
.not_found => {
next.tiingo_retry_after_s = cache.computeExpires(
now_s,
.{ .seconds = cache.Ttl.tiingo_backoff, .jitter_pct = tiingo_backoff_jitter_pct },
symbol,
);
log.info("{s}: Tiingo NotFound, backing off until {f}", .{ symbol, Date.fromEpoch(next.tiingo_retry_after_s) });
},
.unknown => {},
}
return next;
}
/// Whether a unanimous `NotFound` from `refetchFullHistory` earns a
/// sticky negative-cache entry, given whatever metadata the symbol
/// carried before the fetch.
///
/// Normally yes: every provider affirmatively disclaimed the
/// symbol, so stop asking. The exception is `.external`, where the
/// bars are managed outside zfin and no provider was ever going to
/// carry them. A negative entry there is not "stop asking a
/// provider that has nothing" - it is unrecoverable data loss.
/// `writeNegative` overwrites `candles_daily.srf` with a marker,
/// negative entries never expire, and `getCandles` short-circuits
/// on `isNegative` *before* it reaches `syncCandlesFromServer`, so
/// the marker makes the only surviving copy of the series
/// permanently unreachable. On a server, where the
/// externally-populated cache IS the master copy, nothing survives
/// at all.
///
/// A `null` prior meta (a true cold start, both cache files absent)
/// still writes the entry: there is no label to consult and nothing
/// local to destroy. That leaves a real hole - a cold start while
/// `ZFIN_SERVER` is unreachable poisons an external symbol until
/// `--refresh-data=force` or `cache clear`. Closing it needs the
/// `isNegative` short-circuit moved after the server tier, which
/// would re-hit the network for every legitimately candle-less
/// symbol (crypto, delisted tickers) on every run. That trade
/// belongs with the deferred external-symbol routing work, not
/// here.
///
/// Deliberately pure so it can be tested directly: reaching the
/// `true` branch through `getCandles` requires two live provider
/// 404s, so an inline conditional at the call site would be
/// permanently uncovered.
fn shouldNegativeCache(prior: ?cache.Store.CandleMeta) bool {
const meta = prior orelse return true;
return meta.provider != .external;
}
/// Fetch candles from providers with error classification.
///
/// Error handling:
/// - ServerError/RateLimited/RequestFailed from Tiingo -> TransientError (stop refresh, retry later)
/// - NotFound/ParseError/InvalidResponse from Tiingo -> try Yahoo (symbol-level issue)
/// - Unauthorized -> TransientError (config problem, stop refresh)
///
/// `prefer_yahoo` skips the Tiingo attempt entirely. Callers derive
/// it from `CandleMeta.tiingo_retry_after_s` - i.e. "Tiingo told us
/// 404 recently, don't waste the call". It is NOT derived from
/// which provider sourced the cache; see the `provider` field's
/// doc comment for why that distinction matters.
fn fetchCandlesFromProviders(
self: *DataService,
symbol: []const u8,
from: Date,
to: Date,
prefer_yahoo: bool,
) (DataError || error{NotFound})!struct {
candles: []Candle,
provider: cache.Store.CandleProvider,
tiingo_coverage: TiingoCoverage,
} {
// Under an active Tiingo backoff, go straight to Yahoo.
if (prefer_yahoo) {
if (self.getProvider(Yahoo)) |yh| {
if (yh.fetchCandles(self.allocator, symbol, from, to)) |candles| {
log.debug("{s}: candles from Yahoo (Tiingo backoff active)", .{symbol});
return .{ .candles = candles, .provider = .yahoo, .tiingo_coverage = .unknown };
} else |err| {
log.warn("{s}: Yahoo (Tiingo backoff active) failed: {s}", .{ symbol, @errorName(err) });
}
} else |_| {}
}
// Primary: Tiingo. `coverage` accumulates what this attempt
// taught us, so the Yahoo fallback below can report it back
// to the caller without re-deriving it from the error.
var coverage: TiingoCoverage = .unknown;
if (self.getProvider(Tiingo)) |tg| {
if (tg.fetchCandles(self.allocator, symbol, from, to)) |candles| {
log.debug("{s}: candles from Tiingo", .{symbol});
return .{ .candles = candles, .provider = .tiingo, .tiingo_coverage = .covered };
} else |err| {
log.warn("{s}: Tiingo failed: {s}", .{ symbol, @errorName(err) });
if (err == error.Unauthorized) {
log.err("{s}: Tiingo auth failed - check TIINGO_API_KEY", .{symbol});
return DataError.AuthError;
}
if (err == error.RateLimited) {
// Rate limited: back off and retry - this is expected, not a failure
log.info("{s}: Tiingo rate limited, backing off", .{symbol});
self.rateLimitBackoff();
if (tg.fetchCandles(self.allocator, symbol, from, to)) |candles| {
log.debug("{s}: candles from Tiingo (after rate limit backoff)", .{symbol});
return .{ .candles = candles, .provider = .tiingo, .tiingo_coverage = .covered };
} else |retry_err| {
log.warn("{s}: Tiingo retry after backoff failed: {s}", .{ symbol, @errorName(retry_err) });
if (retry_err == error.RateLimited) {
// Still rate limited after backoff - one more try
self.rateLimitBackoff();
if (tg.fetchCandles(self.allocator, symbol, from, to)) |candles| {
log.debug("{s}: candles from Tiingo (after second backoff)", .{symbol});
return .{ .candles = candles, .provider = .tiingo, .tiingo_coverage = .covered };
} else |_| {}
}
// Exhausted rate limit retries - treat as transient
return DataError.TransientError;
}
}
if (isTransientError(err)) {
// Server error or connection failure - stop, don't fall back
return DataError.TransientError;
}
// NotFound, ParseError, InvalidResponse - fall back to
// Yahoo for this call. Only a genuine 404 is a
// statement about Tiingo's coverage; a 400 / 402 /
// malformed body says something about this request and
// must not earn a remembered demotion. Mirrors the rule
// `isPermanentProviderFailure` already applies to the
// negative cache.
if (isPermanentProviderFailure(err)) {
coverage = .not_found;
log.info("{s}: Tiingo does not carry this symbol, trying Yahoo", .{symbol});
} else {
log.info("{s}: Tiingo request failed ({s}) - not a coverage verdict, trying Yahoo for this call only", .{ symbol, @errorName(err) });
}
}
} else |_| {
log.warn("{s}: Tiingo provider not available (no API key?)", .{symbol});
}
// Fallback: Yahoo. Skipped when we already tried Yahoo first
// (active backoff) and it failed - no point asking twice.
if (!prefer_yahoo) {
if (self.getProvider(Yahoo)) |yh| {
if (yh.fetchCandles(self.allocator, symbol, from, to)) |candles| {
log.info("{s}: candles from Yahoo (Tiingo fallback)", .{symbol});
return .{ .candles = candles, .provider = .yahoo, .tiingo_coverage = coverage };
} else |err| {
log.warn("{s}: Yahoo fallback also failed: {s}", .{ symbol, @errorName(err) });
}
} else |_| {
log.warn("{s}: Yahoo provider not available", .{symbol});
}
}
return DataError.FetchFailed;
}
/// Classify whether a provider error is transient (provider is down).
/// ServerError = HTTP 5xx, RequestFailed = connection/network failure.
/// Note: RateLimited and Unauthorized are handled separately.
fn isTransientError(err: anyerror) bool {
return err == error.ServerError or
err == error.RequestFailed;
}
/// Centralized "are we about to touch the network?" gate. Tests
/// set `panic_on_network_attempt` to assert that offline-mode
/// paths never reach this site. Production callers always pass.
/// Inline so the panic body is only generated when the field is
/// actually checked (no overhead on the false branch).
/// `opts` combined with the service-wide policy.
///
/// A union, not an override: a caller asking for `force_refresh` gets it
/// even under `.auto`, and the global flag applies to callers with no
/// opinion. `skip_network` continues to win over `force_refresh` wherever
/// both end up set, which is the existing documented precedence.
fn effectiveOptions(self: *DataService, opts: FetchOptions) FetchOptions {
return .{
.force_refresh = opts.force_refresh or self.default_options.force_refresh,
.skip_network = opts.skip_network or self.default_options.skip_network,
};
}
inline fn assertNetworkAllowed(self: *DataService, context: []const u8) void {
if (self.panic_on_network_attempt) {
std.debug.panic("network attempted in offline-mode test: {s}", .{context});
}
}
// ── Live-price stream (push-based intraday quotes) ──────────
//
// The transport mirrors the REST live-quote provider
// (Config.effectiveLiveQuoteProvider): a paid Tiingo subscriber
// streams via Tiingo's official IEX feed, everyone else via Yahoo's
// free (unofficial) feed.
/// Start (or ensure running) the live-price stream for `symbols`.
/// Idempotent while running: the symbol set is fixed for the
/// stream's lifetime, so to change it call `stopLiveStream` first.
/// `symbols` is copied by the stream, so the caller need not keep it
/// alive.
///
/// This is the push counterpart to a one-shot quote fetch: once
/// running, `liveStreamSnapshot` returns continuously-updated prices
/// with no further requests. Transport follows
/// `effectiveLiveQuoteProvider`: `.tiingo` (official real-time IEX,
/// key-guarded) for a Power subscriber or anyone who set
/// `ZFIN_LIVE_QUOTE_PROVIDER=tiingo` (IEX level-6 works on the free
/// tier too - one request per connect), else `.yahoo` (keyless, all
/// tiers, but ~15-min delayed/unofficial).
pub fn startLiveStream(self: *DataService, symbols: []const []const u8) DataError!void {
self.assertNetworkAllowed("startLiveStream");
if (self.live_stream == null) {
// `effectiveLiveQuoteProvider` already key-guards Tiingo, so
// `.tiingo` here implies `tiingo_key != null`.
const use_tiingo = self.config.effectiveLiveQuoteProvider() == .tiingo;
const transport: LiveStream.Transport = if (use_tiingo) .tiingo else .yahoo;
const key: ?[]const u8 = if (use_tiingo) self.config.tiingo_key else null;
self.live_stream = LiveStream.create(self.allocator, self.io, transport, key) catch return DataError.OutOfMemory;
}
self.live_stream.?.start(symbols) catch |err| {
log.warn("startLiveStream: {t}", .{err});
return DataError.FetchFailed;
};
}
/// Stop and tear down the live stream if running.
pub fn stopLiveStream(self: *DataService) void {
if (self.live_stream) |s| {
s.destroy();
self.live_stream = null;
}
}
/// True if the live stream is currently running.
pub fn liveStreamActive(self: *DataService) bool {
return if (self.live_stream) |s| s.isRunning() else false;
}
/// Copy the latest streamed prices for `symbols` into `out` (keys
/// borrow `symbols`, mirroring a one-shot quote map). No-op when the
/// stream isn't running; symbols not yet seen are left absent so the
/// caller falls back to the last cached close.
pub fn liveStreamSnapshot(self: *DataService, symbols: []const []const u8, out: *std.StringHashMap(f64)) void {
if (self.live_stream) |s| s.snapshotInto(symbols, out);
}
/// Fetch daily candles for a symbol (10+ years for trailing returns).
/// Checks cache first; fetches from Tiingo (primary) or Yahoo (fallback) if stale/missing.
/// Uses incremental updates: when the cache is stale, only fetches
/// candles newer than the last cached date rather than re-fetching
/// the entire history.
///
/// `opts.skip_network = true` -> returns cached data even if stale,
/// returns FetchFailed on cache miss without touching the network.
/// `opts.force_refresh = true` -> treats cache as stale and fetches.
pub fn getCandles(self: *DataService, symbol: []const u8, opts_in: FetchOptions) DataError!FetchResult(Candle) {
// Fold in the service-wide policy once, here, so every decision below
// sees the user's `--refresh-data` instruction even when the caller
// passed `.{}`. See `default_options`.
const opts = self.effectiveOptions(opts_in);
var s = self.store();
// Negative cache: this symbol is known to have no candle data on
// any provider (e.g. crypto like DOGE-USD that Tiingo can't
// serve). The marker lives in candles_daily.srf - written by the
// permanent-NotFound path below, or synced verbatim from a
// ZFIN_SERVER. Bail before any server sync or provider fetch so
// repeated calls don't re-hit the network every invocation.
// --refresh-data=force (force_refresh) bypasses this to retry.
if (!opts.force_refresh and s.isNegative(symbol, .candles_daily))
return DataError.FetchFailed;
const today = fmt.todayDate(self.io);
// wall-clock required: candle-meta freshness uses a market-aware
// boundary (next post-close / NAV-availability time in ET; see
// market.nextCandleExpiry). Captured once here and threaded to
// every candle-meta write below so a single invocation stamps a
// consistent boundary.
const now_s = std.Io.Timestamp.now(self.io, .real).toSeconds();
const kind = market.classify(symbol);
// Check candle metadata for freshness (tiny file, no candle deserialization)
const meta_result = s.readCandleMeta(symbol);
if (meta_result) |mr| {
const m = mr.meta;
// Offline mode: return cached data without touching the
// network. Cache miss / TwelveData-only cache is treated
// as unavailable.
if (opts.skip_network) {
if (m.provider == .twelvedata) {
log.debug("{s}: skip_network and only TwelveData cached - treating as unavailable", .{symbol});
return DataError.FetchFailed;
}
if (s.read(self.allocator, Candle, symbol, null, .any)) |r| {
if (!s.isCandleMetaFresh(symbol)) {
log.info("{s}: candles stale-cached returned (skip_network)", .{symbol});
}
return .{ .data = r.data, .source = .cached, .timestamp = mr.created, .allocator = self.allocator };
}
return DataError.FetchFailed;
}
// If cached data is from TwelveData (deprecated for candles due to
// unreliable adj_close), skip cache and fall through to full re-fetch.
if (m.provider == .twelvedata) {
log.debug("{s}: cached candles from TwelveData - forcing full re-fetch", .{symbol});
} else if (!opts.force_refresh and s.isCandleMetaFresh(symbol)) {
// Fresh - deserialize candles and return
log.debug("{s}: candles fresh in local cache", .{symbol});
if (s.read(self.allocator, Candle, symbol, null, .any)) |r|
return .{ .data = r.data, .source = .cached, .timestamp = mr.created, .allocator = self.allocator };
} else {
// Stale - try server sync before incremental fetch.
// (Force-refresh skips server sync too: the user explicitly
// asked for fresh provider data.)
if (!opts.force_refresh and self.syncCandlesFromServer(symbol)) {
// Re-read meta: the sync wrote the server's bytes
// verbatim, so its view of both freshness AND
// adjustment basis is now ours. `serverBarRegression`
// only guards `last_date` going backwards - it cannot
// see a same-dated file whose historical adj_close is
// stale, so the basis has to be re-checked here.
const synced = if (s.readCandleMeta(symbol)) |sm| sm.meta else m;
const synced_action = freshness.newestCorporateAction(self.allocator, &s, symbol, synced.last_date);
if (s.isCandleMetaFresh(symbol) and
!freshness.adjustmentBasisStale(synced.adj_basis, synced_action))
{
log.debug("{s}: candles synced from server and fresh", .{symbol});
if (s.read(self.allocator, Candle, symbol, null, .any)) |r|
return .{ .data = r.data, .source = .cached, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
log.debug("{s}: candles synced from server but stale, falling through to incremental fetch", .{symbol});
}
// Stale - try incremental update using last_date from meta
const fetch_from = m.last_date.addDays(1);
// Market-aware freshness boundary for any meta write on
// this stale path (next post-close / NAV-availability time).
const expires = market.nextCandleExpiry(now_s, kind);
// A corporate action has gone ex since the basis that
// produced this series, so the cached `adj_close` values
// behind it were never marked down. Appending cannot fix
// that - only replacing the file can. Do this BEFORE the
// incremental fetch so a symbol that needs both a top-up
// and a restatement costs one full fetch, not an append
// followed by a second pass.
//
// Deliberately not checked on the fresh-cache path above:
// the candle TTL lapses at least once per trading day, so
// detection is at most a day behind, and paying a
// dividend/split cache read on every getCandles call would
// tax the hot portfolio-pricing path for nothing.
if (freshness.adjustmentBasisStale(
m.adj_basis,
freshness.newestCorporateAction(self.allocator, &s, symbol, m.last_date),
)) {
log.info("{s}: restating full history (adj_basis {f} predates a corporate action)", .{ symbol, m.adj_basis });
if (self.refetchFullHistory(symbol, today, now_s, now_s < m.tiingo_retry_after_s)) |candles| {
return .{ .data = candles, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
} else |err| {
// Restatement is best-effort. The existing series
// is untouched and still usable, just understated
// by the missed adjustment - fall through to the
// normal top-up and retry on the next stale pass.
log.warn("{s}: full-history restatement failed ({s}); adjustment basis remains stale, falling back to incremental append", .{ symbol, @errorName(err) });
}
}
// Only skip the fetch when we already hold the latest
// *available* bar (weekend/holiday/pre-close gap, or
// caught up): just bump the TTL. Gating on the
// market-aware `shouldRefresh` -- not a naive
// `last_cached + 1 >= today` calendar check -- keeps this
// decision consistent with the `candleFreshness` lag
// report. The old calendar check skipped the fetch when
// `last_date` was merely yesterday, so a just-closed
// session's due bar got cached as "fresh" until the next
// boundary while the lag check reported it lagging (the
// Friday-17:00 deadlock that exited 75 every retry).
if (!market.shouldRefresh(now_s, kind, m.last_date)) {
s.updateCandleMeta(symbol, m, expires);
if (s.read(self.allocator, Candle, symbol, null, .any)) |r|
return .{ .data = r.data, .source = .cached, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
} else {
// Incremental fetch from day after last cached candle
self.assertNetworkAllowed("getCandles incremental fetchCandlesFromProviders");
const result = self.fetchCandlesFromProviders(symbol, fetch_from, today, now_s < m.tiingo_retry_after_s) catch |err| {
if (err == DataError.TransientError) {
// Increment fail_count for this symbol
const new_fail_count = m.fail_count +| 1; // saturating add
log.warn("{s}: transient failure (fail_count now {d})", .{ symbol, new_fail_count });
var degraded = m;
degraded.fail_count = new_fail_count;
s.updateCandleMeta(symbol, degraded, now_s + market.short_retry_s);
// If degraded (fail_count >= 3), return stale data rather than failing
if (new_fail_count >= 3) {
log.warn("{s}: degraded after {d} consecutive failures, returning stale data", .{ symbol, new_fail_count });
if (s.read(self.allocator, Candle, symbol, null, .any)) |r|
return .{ .data = r.data, .source = .cached, .timestamp = mr.created, .allocator = self.allocator };
}
return DataError.TransientError;
}
// Non-transient failure - return stale data if available
if (s.read(self.allocator, Candle, symbol, null, .any)) |r|
return .{ .data = r.data, .source = .cached, .timestamp = mr.created, .allocator = self.allocator };
return DataError.FetchFailed;
};
const new_candles = result.candles;
// Fold what this fetch learned about Tiingo coverage
// into the metadata we're about to write.
const next_meta = applyTiingoCoverage(m, symbol, now_s, result.provider, result.tiingo_coverage);
if (new_candles.len == 0) {
// No new candles. Either a genuine non-trading-day
// gap (weekend/holiday), the provider hasn't posted
// the just-closed bar yet, or the calendar's
// "trading day" was an un-modeled closure (e.g.
// Good Friday). market.staleCandleExpiry picks the
// boundary: a short retry while a due bar is merely
// late, falling back to the next close boundary
// once it's overdue past the lag grace window - so
// an un-modeled closure stops thrashing and waits
// for the next real session.
self.allocator.free(new_candles);
s.updateCandleMeta(symbol, next_meta, market.staleCandleExpiry(now_s, kind, m.last_date));
if (s.read(self.allocator, Candle, symbol, null, .any)) |r|
return .{ .data = r.data, .source = .cached, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
} else {
// Append new candles to existing file + update meta, reset fail_count.
// TTL via `expiryAfterFetch`, NOT the precomputed
// next-boundary `expires`: getting a bar back does not
// mean getting the RIGHT bar back.
s.appendCandles(symbol, new_candles, next_meta, expiryAfterFetch(now_s, kind, new_candles));
if (s.read(self.allocator, Candle, symbol, null, .any)) |r| {
self.allocator.free(new_candles);
return .{ .data = r.data, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
return .{ .data = new_candles, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
}
}
}
// Offline mode + no usable cache - give up.
if (opts.skip_network) {
log.debug("{s}: skip_network and no cached candles - unavailable", .{symbol});
return DataError.FetchFailed;
}
// No usable cache - try server sync first (skipped on force_refresh).
if (!opts.force_refresh and self.syncCandlesFromServer(symbol)) {
if (s.isCandleMetaFresh(symbol)) {
log.debug("{s}: candles synced from server and fresh (no prior cache)", .{symbol});
if (s.read(self.allocator, Candle, symbol, null, .any)) |r|
return .{ .data = r.data, .source = .cached, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
log.debug("{s}: candles synced from server but stale, falling through to full fetch", .{symbol});
}
// No usable cache - full fetch. Tiingo first (it returns
// candles + dividends + splits from one response), falling back
// to a full-range Yahoo fetch when Tiingo does not carry the
// symbol. The fixed start date (see `populateAllFromTiingo`) is
// 2000-01-01, deep enough to cover a 10Y trailing-return window
// even when `--as-of` back-dates the reference into 2014-era
// imported portfolio history, plus a buffer for older corporate
// actions like SPYM's 2017-10-16 split.
//
// The Yahoo fallback matters here beyond convenience: this
// branch used to be Tiingo-only, so a symbol Tiingo does not
// carry could not be cold-started at all - it 404'd, wrote a
// negative-cache marker *over* candles_daily.srf, and stayed
// unavailable until `cache clear`. Any symbol that reached this
// branch with a populated cache would have had its history
// destroyed.
log.debug("{s}: fetching full candle history from provider", .{symbol});
const prior_backoff: i64 = if (meta_result) |mr| mr.meta.tiingo_retry_after_s else 0;
const candles = self.refetchFullHistory(symbol, today, now_s, now_s < prior_backoff) catch |err| {
if (err == DataError.TransientError) {
// Transient: increment fail_count on existing meta so
// we know to back off if this keeps happening.
if (meta_result) |mr| {
var degraded = mr.meta;
degraded.fail_count = mr.meta.fail_count +| 1;
s.updateCandleMeta(symbol, degraded, now_s + market.short_retry_s);
}
return DataError.TransientError;
}
// Only a unanimous NotFound - every provider affirmatively
// disclaims the symbol - earns a sticky negative-cache
// entry. Auth trouble, a malformed response, or a Yahoo
// network blip are not permanent facts about the symbol, so
// they fail this call but stay retryable. This matters
// because the negative marker replaces the candle file: a
// bad negative both suppresses retries until `--refresh`
// and throws away whatever history was cached.
//
// And even a unanimous NotFound is not a verdict when the
// series was never a provider's to serve; see
// `shouldNegativeCache`.
if (err == error.NotFound) {
const prior: ?cache.Store.CandleMeta = if (meta_result) |mr| mr.meta else null;
if (shouldNegativeCache(prior)) {
s.writeNegative(symbol, .candles_daily);
} else {
log.warn("{s}: every provider disclaims it, but its cache is externally managed (provider::external) - refusing the negative entry, which would make the series unreachable via ZFIN_SERVER", .{symbol});
}
}
if (err == DataError.AuthError) return DataError.AuthError;
return DataError.FetchFailed;
};
return .{ .data = candles, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
/// Fetch dividend history for a symbol.
///
/// The only type that uses both extra hooks, and for one reason: a
/// dividend cache can be fresh and still be missing a payment that
/// has already happened. `dividendsNeedRefresh` notices that from the
/// symbol's own payment schedule, and `dividendTtl` sets the clock
/// according to whether that schedule is known.
pub fn getDividends(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!FetchResult(Dividend) {
return self.fetchCached(Dividend, symbol, null, dividendsNeedRefresh, dividendTtl, opts);
}
/// Fetch split history for a symbol.
///
/// No refresh hook: splits are announced weeks to months ahead, so
/// the forward-looking record is cached long before it matters.
pub fn getSplits(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!FetchResult(Split) {
return self.fetchCached(Split, symbol, null, null, null, opts);
}
/// Fetch options chain for a symbol (all expirations, no API key needed).
pub fn getOptions(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!FetchResult(OptionsChain) {
return self.fetchCached(OptionsChain, symbol, null, null, null, opts);
}
/// Days after an expected ex-date during which a still-absent
/// distribution is worth chasing with a re-fetch.
///
/// Bounded for the same reason `earnings_actual_chase_days` is: an
/// unbounded chase refetches forever whenever the cadence estimate
/// is wrong or a sponsor skips a period. Fourteen days covers the
/// prediction error actually observed (ex-dates land 1-3 days off a
/// median-of-gaps forecast) plus the few days a sponsor can take to
/// publish, and then hands back to the TTL.
const dividend_chase_days: i32 = 14;
/// Fewest cached records that can establish a distribution cadence.
///
/// Three records give two gaps, which is the minimum that can
/// disagree - and therefore the minimum where a median means
/// anything. Below it there is no cadence, the predicate declines,
/// and `Ttl.dividends` is the only guard. That is the correct
/// direction: a newly-bought holding has no schedule to be late
/// against, and inventing one from a single gap would fire on
/// noise.
const dividend_cadence_min_records: usize = 3;
/// How many of the newest consecutive gaps the cadence medians over.
const dividend_cadence_gaps: usize = 5;
/// The `out.len` newest ex-dates in `divs`, descending. Returns how
/// many were written.
///
/// Sorts rather than trusting input order. The cache file happens to
/// be written newest-first today, but nothing in the SRF contract
/// promises it, `writeSupplement`'s sorted-union merge means two
/// providers' records interleave, and a cadence silently computed
/// from negative gaps would predict dates in the past forever.
fn newestExDates(divs: []const Dividend, out: []Date) usize {
var n: usize = 0;
for (divs) |d| {
// Insertion position in the descending prefix.
var i: usize = 0;
while (i < n and !out[i].lessThan(d.ex_date)) i += 1;
if (i >= out.len) continue;
var j = @min(n, out.len - 1);
while (j > i) : (j -= 1) out[j] = out[j - 1];
out[i] = d.ex_date;
if (n < out.len) n += 1;
}
return n;
}
/// The symbol's distribution cadence in days, or null when the
/// cache cannot establish one.
///
/// MEDIAN of the newest gaps, not the mean and not the minimum, and
/// the choice matters in both directions. The mean is dragged by a
/// single special distribution landing days after a regular one; the
/// minimum is destroyed by it outright, predicting the next ex-date
/// a few days out and firing for the whole chase window after every
/// payment. A median over up to five gaps absorbs one outlier and
/// still tracks a genuine schedule.
///
/// What it does NOT survive is a cadence CHANGE - a fund moving
/// quarterly to monthly keeps a ~91-day median for four more
/// periods and the forecast runs late. That is a real blind spot,
/// deliberately left to `Ttl.dividends` rather than papered over
/// with a more eager statistic that would cost requests on every
/// symbol to protect against a rare event on one.
fn dividendCadenceDays(divs: []const Dividend) ?i32 {
// SAFETY: only indices below the count returned by
// `newestExDates` are read.
var dates: [dividend_cadence_gaps + 1]Date = undefined;
const n = newestExDates(divs, &dates);
if (n < dividend_cadence_min_records) return null;
// SAFETY: only indices below `g` are read.
var gaps: [dividend_cadence_gaps]i32 = undefined;
var g: usize = 0;
while (g + 1 < n) : (g += 1) gaps[g] = dates[g].days - dates[g + 1].days;
std.mem.sort(i32, gaps[0..g], {}, std.sort.asc(i32));
const mid = g / 2;
const median = if (g % 2 == 1) gaps[mid] else @divFloor(gaps[mid - 1] + gaps[mid], 2);
// Duplicate ex-dates (a regular and a special on the same day,
// repeated) can median to zero. Zero is not a cadence, and it
// would divide by zero below.
return if (median > 0) median else null;
}
/// Whether a fresh-in-cache dividend set warrants a re-fetch: true
/// when at least one full cadence period has elapsed since the
/// newest cached ex-date and the distribution that should have
/// landed in it is still absent.
///
/// This exists because dividend TTL and dividend availability are
/// uncorrelated. An ETF's distribution is not retrievable from the
/// provider until roughly its own ex-date - Fidelity's FDVV
/// publishes a whole-year calendar in February yet had no amount on
/// the day before its September ex-date - and it pays 1-7 days
/// later. So the interval between "the record exists" and "the cash
/// is in the account" is about a week, and any TTL longer than that
/// can be written before the record exists and still be fresh after
/// the payment. Quarter-end clustering makes it systematic: a dozen
/// funds go ex within three days of each other, so one refresh pass
/// puts a dozen entries on the same expiry shelf and every one of
/// them straddles the next quarter.
///
/// The predicate keys off the failure condition itself - "a
/// distribution I should have seen has not arrived" - rather than
/// off elapsed time since an arbitrary fetch, which is why it fires
/// on exactly the overdue symbols and stays silent on the rest.
///
/// ROLLS FORWARD over missed periods. Anchoring the chase window on
/// the first expected date after the newest cached record means a
/// symbol two quarters behind is past its window and can never
/// recover; taking the latest expected date at or before `today`
/// keeps it recoverable. And requiring a FULL period to have elapsed
/// is what stops it firing immediately after a successful fetch,
/// when the record it just stored is days old.
///
/// A newest ex-date in the FUTURE - normal for an issuer like NKE
/// that declares a quarter ahead - yields a negative elapsed and
/// declines, which is correct: nothing is overdue.
fn dividendsNeedRefresh(divs: []const Dividend, today: Date) bool {
const cadence = dividendCadenceDays(divs) orelse return false;
// SAFETY: a non-null cadence guarantees at least one date.
var newest: [1]Date = undefined;
if (newestExDates(divs, &newest) == 0) return false;
const elapsed = today.days - newest[0].days;
if (elapsed < cadence) return false;
const expected = newest[0].addDays(@divFloor(elapsed, cadence) * cadence);
return today.days - expected.days <= dividend_chase_days;
}
/// LEVEL 1's length for dividends, chosen from the records themselves
/// rather than fixed per type.
///
/// Two clocks, because the clock is doing a different job in each
/// case:
///
/// - **Schedule known** (a cadence can be worked out). The
/// prediction is what finds a due payment, so the clock only has
/// to cover what a schedule cannot predict - an off-cycle special
/// distribution. Fourteen days.
///
/// - **Schedule unknown** (fewer than three records, or no usable
/// cadence - typically a newly bought holding). There is nothing
/// to predict from, so the clock is the ONLY thing that will ever
/// make us look again. Six days, which guarantees the entry is
/// stale by the next weekly review.
///
/// Sizing them separately is the whole point. One short clock for
/// everything would re-check long-established payers every few days
/// to guard against a problem they do not have; one long clock would
/// let a new holding's first payment go unnoticed for a fortnight.
fn dividendTtl(divs: []const Dividend) cache.TtlSpec {
const known = dividendCadenceDays(divs) != null;
const base = if (known) cache.Ttl.dividends_scheduled else cache.Ttl.dividends;
// Jitter policy stays with the rest of it in `DataType.ttl()`;
// only the base differs here.
return .{ .seconds = base, .jitter_pct = comptime cache.DataType.dividends.ttl().jitter_pct };
}
/// Days after an earnings report date during which a still-missing
/// `actual` is worth chasing with a re-fetch. Past this window the
/// gap is treated as permanent (FMP won't backfill it; earnings has
/// no secondary source), so the cache is honored until its TTL.
const earnings_actual_chase_days: i32 = 14;
/// Whether a fresh-in-cache earnings set warrants a re-fetch: true
/// when an event whose report date has arrived (date <= today) is
/// still missing its `actual` AND the report is recent enough
/// (within `window_days`) that the actual could still post.
///
/// Earnings has no cross-provider backfill (unlike dividends/splits,
/// which Tiingo supplements via the candle fetch), so a missing
/// actual only arrives through a later FMP fetch. Without the recency
/// bound a permanently-incomplete past row -- e.g. SPY's 2005-2006
/// estimate-only rows FMP never backfills -- forces a re-fetch every
/// run. The 30-day TTL backstops any actual slower than the window.
fn earningsNeedsRefresh(events: []const EarningsEvent, today: Date, window_days: i32) bool {
for (events) |ev| {
if (ev.actual == null and !today.lessThan(ev.date) and today.days - ev.date.days <= window_days) {
return true;
}
}
return false;
}
/// `earningsNeedsRefresh` bound to the production window, in the shape
/// `fetchCached`'s `needsRefresh` hook takes.
///
/// The three-argument form stays separate so tests can sweep the
/// window boundary without reaching for the constant; this adapter is
/// what production passes.
fn earningsNeedsRefreshHook(events: []const EarningsEvent, today: Date) bool {
return earningsNeedsRefresh(events, today, earnings_actual_chase_days);
}
/// Fetch earnings history for a symbol.
///
/// A thin wrapper over `fetchCached`: the cache / offline / server-sync
/// / provider / negative-cache sequence is entirely generic, and the
/// two earnings-specific parts are supplied as hooks -
/// `earningsPostProcess` rebuilds the derived `surprise` field, and
/// `earningsNeedsRefreshHook` is the smart refresh (re-fetch inside the
/// 30-day TTL once a report date has passed but the actual is still
/// missing; see `earningsNeedsRefresh`).
///
/// The only thing that cannot be a hook is the mutual-fund skip, which
/// short-circuits before any cache read: a fund has no quarterly
/// earnings at all, so there is nothing to cache, fetch, or
/// negative-cache.
///
/// `opts.skip_network = true` -> returns cached data even if stale,
/// returns FetchFailed on cache miss without touching the network.
/// `opts.force_refresh = true` -> treats cache as stale and fetches.
pub fn getEarnings(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!FetchResult(EarningsEvent) {
// Mutual funds (5-letter tickers ending in X) don't have quarterly earnings.
if (market.classify(symbol) == .mutual_fund) {
return .{ .data = &.{}, .source = .cached, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
return self.fetchCached(EarningsEvent, symbol, earningsPostProcess, earningsNeedsRefreshHook, null, opts);
}
/// Fetch ETF profile for a symbol. Assembles a unified
/// `EtfProfile` view from the EDGAR `etf_metrics` cache (profile
/// + sectors + holdings) plus the Wikidata `classification`
/// cache (inception_date, fund name fallback). Both underlying
/// caches are managed by `getEtfMetrics` / `getClassification`;
/// this function does not maintain its own cache.
///
/// Several legacy fields that AlphaVantage used to populate
/// (`expense_ratio`, `dividend_yield`, `portfolio_turnover`,
/// `leveraged`) remain on `EtfProfile` but stay null here -
/// EDGAR NPORT-P doesn't carry them. They'll fill in once a
/// prospectus parser lands.
///
/// `opts.skip_network = true` and `opts.force_refresh = true`
/// are forwarded to `getEtfMetrics`.
pub fn getEtfProfile(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!FetchResult(EtfProfile) {
// Primary source: EDGAR ETF metrics. If the symbol isn't a
// fund (or isn't in EDGAR), surface NotFound to the caller -
// matches the old AlphaVantage behavior of returning empty
// profiles for non-ETFs.
const metrics = try self.getEtfMetrics(symbol, opts);
defer metrics.deinit();
// Walk the EtfMetricRecord slice to extract profile + sectors
// + holdings. The slice shape is "one .profile, then N
// .sector, then M .holding" per `appendEtfMetricRecords`.
var name: ?[]const u8 = null;
errdefer if (name) |n| self.allocator.free(n);
var net_assets: ?f64 = null;
var sectors_buf: std.ArrayList(SectorWeight) = .empty;
errdefer {
for (sectors_buf.items) |s| self.allocator.free(s.name);
sectors_buf.deinit(self.allocator);
}
var holdings_buf: std.ArrayList(Holding) = .empty;
errdefer {
for (holdings_buf.items) |h| {
self.allocator.free(h.name);
if (h.symbol) |s| self.allocator.free(s);
if (h.cusip) |c| self.allocator.free(c);
}
holdings_buf.deinit(self.allocator);
}
for (metrics.data) |rec| switch (rec) {
.profile => |p| {
if (p.series_name) |sn| name = try self.allocator.dupe(u8, sn);
net_assets = p.net_assets;
},
.sector => |s| {
try sectors_buf.append(self.allocator, .{
.name = try self.allocator.dupe(u8, s.description),
.weight = s.pct_of_portfolio / 100.0,
});
},
.holding => |h| {
const sym_dup: ?[]const u8 = if (h.ticker) |t|
try self.allocator.dupe(u8, t)
else
null;
errdefer if (sym_dup) |s| self.allocator.free(s);
const cusip_dup: ?[]const u8 = if (h.cusip) |c|
try self.allocator.dupe(u8, c)
else
null;
errdefer if (cusip_dup) |c| self.allocator.free(c);
const name_dup = try self.allocator.dupe(u8, h.name);
errdefer self.allocator.free(name_dup);
try holdings_buf.append(self.allocator, .{
.symbol = sym_dup,
.name = name_dup,
.weight = h.pct_of_portfolio / 100.0,
.cusip = cusip_dup,
});
},
};
// Wikidata classification provides inception_date and a
// higher-quality name. Best-effort: if the fetch fails we
// still return the EDGAR-only profile.
var inception_date: ?Date = null;
if (self.getClassification(symbol, opts)) |class_result| {
defer class_result.deinit();
for (class_result.data) |c| {
if (c.inception_date) |idate_str| {
if (Date.parse(idate_str)) |d| inception_date = d else |_| {}
}
// Prefer Wikidata's name if EDGAR didn't provide one.
if (name == null) {
if (c.name) |n| name = try self.allocator.dupe(u8, n);
}
}
} else |_| {}
const sectors_count = sectors_buf.items.len;
const holdings_count = holdings_buf.items.len;
const profile: EtfProfile = .{
.symbol = try self.allocator.dupe(u8, symbol),
.name = name,
.net_assets = net_assets,
.holdings = if (holdings_count > 0)
try holdings_buf.toOwnedSlice(self.allocator)
else
null,
.total_holdings = if (holdings_count > 0) @intCast(holdings_count) else null,
.sectors = if (sectors_count > 0)
try sectors_buf.toOwnedSlice(self.allocator)
else
null,
.inception_date = inception_date,
};
// Free the empty ArrayLists we didn't consume via toOwnedSlice
// (they own no allocations but the ArrayList struct itself
// needs deinit when not handed off).
if (holdings_count == 0) holdings_buf.deinit(self.allocator);
if (sectors_count == 0) sectors_buf.deinit(self.allocator);
return .{
.data = profile,
.source = metrics.source,
.timestamp = metrics.timestamp,
.allocator = self.allocator,
};
}
// ── Wikidata + EDGAR providers ─────────────────────────────────
/// Fetch the Wikidata classification record for a single symbol
/// (name, sector, industry, country, inception date, CIK,
/// instance-of). Cache-first; on miss, runs a 1-symbol batched
/// SPARQL query.
///
/// `opts.skip_network = true` returns cached data even if stale,
/// `FetchFailed` on cache miss. `opts.force_refresh = true`
/// ignores the cache and re-fetches.
pub fn getClassification(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!FetchResult(Wikidata.ClassificationRecord) {
var s = self.store();
if (!opts.force_refresh) {
if (s.read(self.allocator, Wikidata.ClassificationRecord, symbol, null, .fresh_only)) |cached| {
log.debug("{s}: classification fresh in local cache", .{symbol});
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
}
if (opts.skip_network) {
if (s.read(self.allocator, Wikidata.ClassificationRecord, symbol, null, .any)) |cached| {
log.info("{s}: classification stale-cached returned (skip_network)", .{symbol});
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
return DataError.FetchFailed;
}
// Try server sync before hitting Wikidata.
if (!opts.force_refresh and self.syncFromServer(symbol, .classification)) {
if (s.read(self.allocator, Wikidata.ClassificationRecord, symbol, null, .fresh_only)) |cached| {
log.debug("{s}: classification synced from server", .{symbol});
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
}
log.debug("{s}: fetching classification from Wikidata", .{symbol});
self.assertNetworkAllowed("getClassification wikidata.fetch");
var wd = try self.getProvider(Wikidata);
const symbols = [_][]const u8{symbol};
const fetched = wd.fetch(self.allocator, &symbols) catch |err| {
if (err == error.RateLimited) {
self.rateLimitBackoff();
if (wd.fetch(self.allocator, &symbols)) |retried| {
return self.finalizeClassification(symbol, retried, opts);
} else |_| {}
}
log.warn("{s}: wikidata fetch failed: {s}", .{ symbol, @errorName(err) });
return DataError.FetchFailed;
};
return self.finalizeClassification(symbol, fetched, opts);
}
/// Common post-Wikidata path: decide if the result is useful as
/// returned, otherwise consult EDGAR to fill in the gaps,
/// otherwise negative-cache. Either way the cache gets written
/// and a `FetchResult` is returned (or `DataError.NotFound`).
///
/// Takes ownership of `wikidata_records`. The slice is either
/// returned as the result data, freed and replaced by a
/// synthesized slice, or freed and the symbol negative-cached.
fn finalizeClassification(
self: *DataService,
symbol: []const u8,
wikidata_records: []Wikidata.ClassificationRecord,
opts: FetchOptions,
) DataError!FetchResult(Wikidata.ClassificationRecord) {
var s = self.store();
const ttl = cache.DataType.classification.ttl();
// Wikidata returned a useful row -> populate geo from
// geoFor(country) and cache as-is.
if (wikidata_records.len > 0 and wikidataLooksUseful(wikidata_records[0])) {
try self.populateGeo(&wikidata_records[0]);
s.write(Wikidata.ClassificationRecord, symbol, wikidata_records, ttl);
return .{ .data = wikidata_records, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
// Sparse or empty: try EDGAR fallback. `synthesizeClassification`
// takes ownership of the wikidata slice (frees it, returns a
// new one-element slice with the merged record). Returns
// `error.NotFound` when even EDGAR has nothing.
const merged = self.synthesizeClassification(symbol, wikidata_records, opts) catch |err| {
if (err == error.NotFound) {
s.writeNegative(symbol, .classification);
return DataError.NotFound;
}
return DataError.FetchFailed;
};
s.write(Wikidata.ClassificationRecord, symbol, merged, ttl);
return .{ .data = merged, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
/// Populate `record.geo` from `geoFor(record.country)` when it
/// isn't already set. Best-effort: if duping the geo string
/// fails, leaves the field null and propagates the error so the
/// caller can decide whether to bail.
fn populateGeo(self: *DataService, record: *Wikidata.ClassificationRecord) !void {
if (record.geo != null) return;
const country = record.country orelse return;
const g = classification.geoFor(country);
if (std.mem.eql(u8, g, classification.geo.unknown)) return;
record.geo = try self.allocator.dupe(u8, g);
}
/// Whether a Wikidata classification record carries enough
/// downstream-usable data to skip the EDGAR fallback. A record
/// with at least one of `is_etf`, `sector`, `country`, or
/// `asset_class` set is "useful"; sparse records (e.g. SOXX
/// getting only a `name` from Wikidata) need the EDGAR
/// ticker-map fallback to fill in `is_etf=true,
/// asset_class=ETF, country=US`.
fn wikidataLooksUseful(c: Wikidata.ClassificationRecord) bool {
if (c.is_etf) return true;
if (c.asset_class != null) return true;
if (c.country != null) return true;
if (c.sector != null) return true;
return false;
}
/// Synthesize a `ClassificationRecord` for a symbol that
/// Wikidata couldn't classify usefully. Consults the EDGAR
/// ticker maps; if found, also fetches `getEtfMetrics` to
/// recover the NPORT-P series_name (more authoritative than
/// the company_tickers title). Title-keyword inference fills
/// in `sector` and `geo` when the name carries an unambiguous
/// keyword.
///
/// Takes ownership of `wikidata_records`: frees them at exit.
/// Wikidata's `name`/`industry`/`inception_date`/`cik` fields
/// are preserved into the synthesized record when present.
/// Returns `error.NotFound` when EDGAR has nothing either.
fn synthesizeClassification(
self: *DataService,
symbol: []const u8,
wikidata_records: []Wikidata.ClassificationRecord,
opts: FetchOptions,
) !cache.Store.DataFor(Wikidata.ClassificationRecord) {
defer Wikidata.ClassificationRecord.freeSlice(self.allocator, wikidata_records);
const lookup = self.lookupEdgarFallback(symbol, opts);
defer freeEdgarLookup(self.allocator, lookup);
if (lookup == .none) return error.NotFound;
// For ETF/fund hits, try to get the richer series_name from
// NPORT-P. Cache hit is cheap; cache miss triggers an EDGAR
// fetch but is bounded by EDGAR's rate limiter. If the call
// fails (e.g. money-market funds with no NPORT-P), we fall
// back to the ticker-map title.
var etf_metrics_result: ?FetchResult(Edgar.EtfMetricRecord) = null;
defer if (etf_metrics_result) |*r| r.deinit();
etf_metrics_result = self.getEtfMetrics(symbol, opts) catch null;
// Extract series_name and cik from the etf_metrics profile row.
var series_name: ?[]const u8 = null;
var etf_cik: ?[]const u8 = null;
if (etf_metrics_result) |r| {
for (r.data) |rec| switch (rec) {
.profile => |p| {
if (p.series_name) |sn| series_name = sn;
etf_cik = p.cik;
break;
},
else => {},
};
}
// Pull whatever Wikidata's sparse record carried so we
// don't lose data on the merge.
const wd: ?Wikidata.ClassificationRecord = if (wikidata_records.len > 0) wikidata_records[0] else null;
// Pick the best name source: NPORT-P series_name >
// EDGAR ticker-map title > Wikidata name > nothing.
//
// We're on the EDGAR-fallback path because Wikidata's
// record was sparse. For funds, Wikidata's `name` (when
// present) is frequently the underlying INDEX rather than
// the FUND itself -- e.g. SOXX's Wikidata `name` is "PHLX
// Semiconductor Sector" but the fund is "iShares
// Semiconductor ETF" per NPORT-P seriesName. Prefer the
// fund-authoritative source so downstream comments and
// labels show the fund name, not the index name.
const ticker_title: ?[]const u8 = switch (lookup) {
.company_or_uit => |c| c.title,
else => null,
};
const best_name: ?[]const u8 = blk: {
if (series_name) |n| break :blk n;
if (ticker_title) |n| break :blk n;
if (wd) |w| {
if (w.name) |n| break :blk n;
}
break :blk null;
};
// Name source for title-keyword inference: prefer the
// most-authoritative source for fund-style classification
// even when Wikidata supplied a (different) name. Wikidata's
// name for a fund is often less informative than NPORT-P's
// seriesName (e.g. SOXX's Wikidata name is "PHLX
// Semiconductor Sector" which is the index name, not the
// fund name).
const inference_name: ?[]const u8 = series_name orelse ticker_title orelse if (wd) |w| w.name else null;
const inferred_sector = classification.inferSectorFromTitle(inference_name);
const inferred_geo = classification.inferGeoFromTitle(inference_name);
// `is_etf` here means "this is fund-shaped, emit multi-row
// breakdown" -- true for ANY EDGAR-found symbol. The
// `tickers_funds.srf` map mixes mutual funds and
// series-of-trust ETFs alike. The `tickers_companies.srf`
// map carries operating companies, closed-end funds, and
// UITs; operating companies usually have Wikidata coverage
// and wouldn't reach this fallback, so anything that
// dropped here is also fund-shaped (e.g. PIMCO closed-end
// funds whose title says "FUND" but not "ETF" or "TRUST").
//
// The ETF/TRUST keyword in the title still drives the
// asset_class label below ("ETF" vs "Fund"), but the
// fund-shaped routing decision applies regardless.
const is_etf = true;
const asset_class: []const u8 = switch (lookup) {
.managed_fund => "Fund",
.company_or_uit => |c| if (c.is_etf) "ETF" else "Fund",
.none => unreachable,
};
// Country: prefer Wikidata's. Default to "US" for
// EDGAR-found symbols (they're SEC filers).
const country_str: []const u8 = if (wd) |w| (w.country orelse "US") else "US";
// Sector: prefer Wikidata's existing sector (rare in this
// sparse-fallback path), else fall back to inferred.
const sector_str: ?[]const u8 = blk: {
if (wd) |w| {
if (w.sector) |sec| break :blk sec;
}
break :blk inferred_sector;
};
// CIK: prefer Wikidata's, fall back to NPORT-P's.
const cik_str: ?[]const u8 = blk: {
if (wd) |w| {
if (w.cik) |c| break :blk c;
}
if (etf_cik) |c| break :blk c;
break :blk null;
};
// Geo: prefer the Wikidata-derived geo (computed from
// `geoFor(country)` against the country code), else use
// title-keyword inference. Default to "US" when neither
// is available -- EDGAR-found symbols are SEC filers.
const geo_str: []const u8 = blk: {
if (wd) |w| {
if (w.country) |c| {
const g = classification.geoFor(c);
if (!std.mem.eql(u8, g, classification.geo.unknown)) break :blk g;
}
}
if (inferred_geo) |g| break :blk g;
break :blk classification.geo.us;
};
const today = fmt.todayDate(self.io);
var as_of_buf: [10]u8 = undefined;
const as_of_str = try std.fmt.bufPrint(&as_of_buf, "{f}", .{today});
// Allocate each owned field up front with its own errdefer
// so a partial-build on OOM doesn't leak the earlier
// successful dupes. Once all dupes succeed we assemble the
// record (no fallible ops below this point).
const symbol_owned = try self.allocator.dupe(u8, symbol);
errdefer self.allocator.free(symbol_owned);
const name_owned: ?[]const u8 = if (best_name) |n| try self.allocator.dupe(u8, n) else null;
errdefer if (name_owned) |s| self.allocator.free(s);
const sector_owned: ?[]const u8 = if (sector_str) |s| try self.allocator.dupe(u8, s) else null;
errdefer if (sector_owned) |s| self.allocator.free(s);
const industry_owned: ?[]const u8 = if (wd) |w|
(if (w.industry) |i| try self.allocator.dupe(u8, i) else null)
else
null;
errdefer if (industry_owned) |s| self.allocator.free(s);
const country_owned = try self.allocator.dupe(u8, country_str);
errdefer self.allocator.free(country_owned);
const geo_owned = try self.allocator.dupe(u8, geo_str);
errdefer self.allocator.free(geo_owned);
const asset_class_owned = try self.allocator.dupe(u8, asset_class);
errdefer self.allocator.free(asset_class_owned);
const inception_owned: ?[]const u8 = if (wd) |w|
(if (w.inception_date) |i| try self.allocator.dupe(u8, i) else null)
else
null;
errdefer if (inception_owned) |s| self.allocator.free(s);
const cik_owned: ?[]const u8 = if (cik_str) |c| try self.allocator.dupe(u8, c) else null;
errdefer if (cik_owned) |s| self.allocator.free(s);
const as_of_owned = try self.allocator.dupe(u8, as_of_str);
errdefer self.allocator.free(as_of_owned);
const source_owned = try self.allocator.dupe(u8, "edgar_fallback");
errdefer self.allocator.free(source_owned);
const result = try self.allocator.alloc(Wikidata.ClassificationRecord, 1);
result[0] = .{
.symbol = symbol_owned,
.name = name_owned,
.sector = sector_owned,
.industry = industry_owned,
.country = country_owned,
.geo = geo_owned,
.asset_class = asset_class_owned,
.is_etf = is_etf,
.inception_date = inception_owned,
.cik = cik_owned,
.as_of = as_of_owned,
.source = source_owned,
};
return result;
}
/// Fetch XBRL-derived entity facts for a CIK (currently
/// shares-outstanding; extensible to revenue / net income / EPS
/// as new variants are added to `Edgar.EntityFactRecord`).
///
/// CIK is the cache key - the file lives at
/// `<cache_dir>/<cik>/entity_facts.srf`. A single dual-class
/// issuer (BRK.A / BRK.B) shares one entity_facts file because
/// both class symbols resolve to the same CIK.
pub fn getEntityFacts(self: *DataService, cik: []const u8, opts: FetchOptions) DataError!FetchResult(Edgar.EntityFactRecord) {
var s = self.store();
if (!opts.force_refresh) {
if (s.read(self.allocator, Edgar.EntityFactRecord, cik, null, .fresh_only)) |cached| {
log.debug("CIK {s}: entity_facts fresh in local cache", .{cik});
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
}
if (opts.skip_network) {
if (s.read(self.allocator, Edgar.EntityFactRecord, cik, null, .any)) |cached| {
log.info("CIK {s}: entity_facts stale-cached returned (skip_network)", .{cik});
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
return DataError.FetchFailed;
}
if (!opts.force_refresh and self.syncFromServer(cik, .entity_facts)) {
if (s.read(self.allocator, Edgar.EntityFactRecord, cik, null, .fresh_only)) |cached| {
log.debug("CIK {s}: entity_facts synced from server", .{cik});
return .{ .data = cached.data, .source = .cached, .timestamp = cached.timestamp, .allocator = self.allocator };
}
}
log.debug("CIK {s}: fetching entity facts from EDGAR", .{cik});
self.assertNetworkAllowed("getEntityFacts edgar.fetchSharesOutstanding");
var edgar = try self.getProvider(Edgar);
const so_opt = edgar.fetchSharesOutstanding(self.allocator, cik) catch |err| {
log.warn("CIK {s}: shares fetch failed: {s}", .{ cik, @errorName(err) });
return DataError.FetchFailed;
};
if (so_opt) |so_in| {
var so = so_in;
defer so.deinit(self.allocator);
const today = fmt.todayDate(self.io);
var as_of_buf: [10]u8 = undefined;
// [10]u8 always fits "YYYY-MM-DD" (10 chars exactly).
const as_of = std.fmt.bufPrint(&as_of_buf, "{f}", .{today}) catch
@panic("getEntityFacts: 10-byte buffer cannot hold YYYY-MM-DD - unreachable");
const form_dup: ?[]u8 = if (so.form.len > 0) try self.allocator.dupe(u8, so.form) else null;
const shares_record = Edgar.SharesRecord{
.symbol = try self.allocator.dupe(u8, ""),
.shares_outstanding = so.value,
.period_end = try self.allocator.dupe(u8, so.period_end),
.form = form_dup,
.cik = try self.allocator.dupe(u8, cik),
.as_of = try self.allocator.dupe(u8, as_of),
.source = "edgar_xbrl",
};
const records = try self.allocator.alloc(Edgar.EntityFactRecord, 1);
records[0] = .{ .shares_outstanding = shares_record };
s.write(Edgar.EntityFactRecord, cik, records, cache.DataType.entity_facts.ttl());
return .{ .data = records, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
}
// No shares-outstanding data for this CIK (e.g. 20-F-only
// filers like BP, XBRL-light filers like META). Negative-
// cache so we don't keep retrying.
s.writeNegative(cik, .entity_facts);
return DataError.NotFound;
}
/// Fetch ETF metrics (NPORT-P profile + sectors + holdings) for
/// a fund symbol. Cache-first via `<symbol>/etf_metrics.srf`.
///
/// On cache miss, looks up the symbol in the EDGAR ticker maps
/// (fetched on demand via `getTickerMap*`), then runs the full
/// `Edgar.fetchEtfMetrics` cascade.
pub fn getEtfMetrics(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!FetchResult(Edgar.EtfMetricRecord) {
var s = self.store();
if (!opts.force_refresh) {
if (s.read(self.allocator, Edgar.EtfMetricRecord, symbol, null, .fresh_only)) |cached| {
log.debug("{s}: etf_metrics fresh in local cache", .{symbol});
return .{
.data = cached.data,
.source = .cached,
.timestamp = cached.timestamp,
.allocator = self.allocator,
};
}
}
if (opts.skip_network) {
if (s.read(self.allocator, Edgar.EtfMetricRecord, symbol, null, .any)) |cached| {
log.info("{s}: etf_metrics stale-cached returned (skip_network)", .{symbol});
return .{
.data = cached.data,
.source = .cached,
.timestamp = cached.timestamp,
.allocator = self.allocator,
};
}
return DataError.FetchFailed;
}
if (!opts.force_refresh and self.syncFromServer(symbol, .etf_metrics)) {
if (s.read(self.allocator, Edgar.EtfMetricRecord, symbol, null, .fresh_only)) |cached| {
log.debug("{s}: etf_metrics synced from server", .{symbol});
return .{
.data = cached.data,
.source = .cached,
.timestamp = cached.timestamp,
.allocator = self.allocator,
};
}
}
log.debug("{s}: fetching ETF metrics from EDGAR", .{symbol});
self.assertNetworkAllowed("getEtfMetrics edgar.fetchEtfMetrics");
// Load the ticker maps. These are big (3-5 MB each) but the
// load happens once per CLI invocation and the parsed
// TickerMap stays alive across all getEtfMetrics calls in
// the same process.
var mf_map = self.loadMutualFundTickerMap(opts) catch |err| {
log.warn("failed to load mutual-fund ticker map: {s}", .{@errorName(err)});
return DataError.FetchFailed;
};
defer mf_map.deinit();
var co_map = self.loadCompanyTickerMap(opts) catch |err| {
log.warn("failed to load company ticker map: {s}", .{@errorName(err)});
return DataError.FetchFailed;
};
defer co_map.deinit();
var edgar = try self.getProvider(Edgar);
const result = edgar.fetchEtfMetrics(
self.io,
self.allocator,
&mf_map,
&co_map,
symbol,
20,
) catch |err| {
log.warn("{s}: etf_metrics fetch failed: {s}", .{ symbol, @errorName(err) });
return DataError.FetchFailed;
};
switch (result) {
.full => |m_in| {
var m = m_in;
defer m.deinit(self.allocator);
var records: std.ArrayList(Edgar.EtfMetricRecord) = .empty;
errdefer {
for (records.items) |*r| r.deinit(self.allocator);
records.deinit(self.allocator);
}
try Edgar.appendEtfMetricRecords(self.allocator, &records, m);
const owned = try records.toOwnedSlice(self.allocator);
s.write(Edgar.EtfMetricRecord, symbol, owned, cache.DataType.etf_metrics.ttl());
return .{ .data = owned, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
},
.profile_only => |m_in| {
var m = m_in;
defer m.deinit(self.allocator);
var records: std.ArrayList(Edgar.EtfMetricRecord) = .empty;
errdefer {
for (records.items) |*r| r.deinit(self.allocator);
records.deinit(self.allocator);
}
try Edgar.appendEtfMetricRecords(self.allocator, &records, m);
const owned = try records.toOwnedSlice(self.allocator);
s.write(Edgar.EtfMetricRecord, symbol, owned, cache.DataType.etf_metrics.ttl());
return .{ .data = owned, .source = .fetched, .timestamp = std.Io.Timestamp.now(self.io, .real).toSeconds(), .allocator = self.allocator };
},
.not_a_fund => {
// Not a fund - write a negative entry to suppress
// retries. The user can ask `getEntityFacts(cik)`
// separately for stock-level facts.
s.writeNegative(symbol, .etf_metrics);
return DataError.NotFound;
},
.not_in_edgar => {
// Symbol isn't in either ticker map. No EDGAR data
// available; negative-cache.
s.writeNegative(symbol, .etf_metrics);
return DataError.NotFound;
},
}
}
/// Load the EDGAR mutual-fund ticker map. Reads `[]MutualFundTickerEntry`
/// from cache when fresh; otherwise fetches via the provider
/// and writes the parsed slice to cache. The returned
/// `TickerMap` takes ownership of the entries; caller frees via
/// a single `mf_map.deinit()`.
///
/// Heavy: ~28k entries. Cheap on cache hit (fast SRF read);
/// expensive on miss (one HTTP round-trip + JSON parse).
/// Exposed publicly so commands like `enrich` can use the
/// ticker map as a fallback classifier when Wikidata returns
/// no rows for a symbol.
pub fn loadMutualFundTickerMap(self: *DataService, opts: FetchOptions) !Edgar.TickerMap(Edgar.MutualFundTickerEntry) {
var s = self.store();
if (!opts.force_refresh) {
if (s.read(self.allocator, Edgar.MutualFundTickerEntry, "_edgar", null, .fresh_only)) |cached| {
if (cached.data.len > 0) {
return Edgar.TickerMap(Edgar.MutualFundTickerEntry).fromEntries(self.allocator, cached.data);
}
Edgar.MutualFundTickerEntry.freeSlice(self.allocator, cached.data);
}
}
log.debug("fetching EDGAR mutual-fund ticker map", .{});
self.assertNetworkAllowed("loadMutualFundTickerMap edgar.fetchMutualFundTickerMap");
var edgar = try self.getProvider(Edgar);
// Fetch + parse via the provider (correct UA + From + Accept
// + rate-limit token), cache the parsed slice, then build
// the lookup map (which takes ownership of the slice).
const entries = try edgar.fetchMutualFundTickerMap(self.allocator);
s.write(Edgar.MutualFundTickerEntry, "_edgar", entries, cache.DataType.tickers_funds.ttl());
return Edgar.TickerMap(Edgar.MutualFundTickerEntry).fromEntries(self.allocator, entries);
}
/// Load the EDGAR company ticker map (stocks + UITs). Same shape
/// as `loadMutualFundTickerMap` for the `CompanyTickerEntry`
/// type. See that function's doc-comment for cost / use-case
/// guidance.
pub fn loadCompanyTickerMap(self: *DataService, opts: FetchOptions) !Edgar.TickerMap(Edgar.CompanyTickerEntry) {
var s = self.store();
if (!opts.force_refresh) {
if (s.read(self.allocator, Edgar.CompanyTickerEntry, "_edgar", null, .fresh_only)) |cached| {
if (cached.data.len > 0) {
return Edgar.TickerMap(Edgar.CompanyTickerEntry).fromEntries(self.allocator, cached.data);
}
Edgar.CompanyTickerEntry.freeSlice(self.allocator, cached.data);
}
}
log.debug("fetching EDGAR company ticker map", .{});
self.assertNetworkAllowed("loadCompanyTickerMap edgar.fetchCompanyTickerMap");
var edgar = try self.getProvider(Edgar);
const entries = try edgar.fetchCompanyTickerMap(self.allocator);
s.write(Edgar.CompanyTickerEntry, "_edgar", entries, cache.DataType.tickers_companies.ttl());
return Edgar.TickerMap(Edgar.CompanyTickerEntry).fromEntries(self.allocator, entries);
}
/// Look up a symbol in the EDGAR ticker maps. Used by the
/// `enrich` command as a fallback classifier when Wikidata
/// returns no rows for the symbol. Loads both maps (cache or
/// network), runs the lookup, frees the maps, returns the
/// digested `EdgarLookup` union.
///
/// Commands consume the union directly - they never see
/// `TickerMap` / `MutualFundTickerEntry` / `CompanyTickerEntry`
/// shapes. Provider details stay inside the service layer.
///
/// Caller owns the `title` string when the result is
/// `.company_or_uit{ .title = non-null }`. Free with the
/// allocator passed to this method (typically the same one
/// the service was initialized with).
pub fn lookupEdgarFallback(
self: *DataService,
sym: []const u8,
opts: FetchOptions,
) EdgarLookup {
var mf_opt: ?Edgar.TickerMap(Edgar.MutualFundTickerEntry) = self.loadMutualFundTickerMap(opts) catch null;
defer if (mf_opt) |*m| m.deinit();
var co_opt: ?Edgar.TickerMap(Edgar.CompanyTickerEntry) = self.loadCompanyTickerMap(opts) catch null;
defer if (co_opt) |*m| m.deinit();
return lookupInTickerMaps(
self.allocator,
sym,
if (mf_opt) |*m| m else null,
if (co_opt) |*m| m else null,
);
}
// ──────────────────────────────────────────────────────────────
/// Fetch a real-time quote for a symbol.
/// Yahoo Finance is primary (free, no API key, no 15-min delay).
/// Falls back to TwelveData if Yahoo fails.
///
/// Quotes are never cached, so `opts.force_refresh` is a no-op
/// (every call goes to the provider). `opts.skip_network = true`
/// returns FetchFailed unconditionally - there's no cached price
/// to fall back to.
pub fn getQuote(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!Quote {
if (opts.skip_network) {
log.debug("{s}: skip_network - quote unavailable (never cached)", .{symbol});
return DataError.FetchFailed;
}
self.assertNetworkAllowed("getQuote");
// Primary: Yahoo Finance (free, real-time)
if (self.getProvider(Yahoo)) |yh| {
if (yh.fetchQuote(self.allocator, symbol)) |quote| {
log.debug("{s}: quote from Yahoo", .{symbol});
return quote;
} else |_| {}
} else |_| {}
// Fallback: TwelveData (requires API key, may be 15-min delayed)
var td = try self.getProvider(TwelveData);
log.debug("{s}: quote fallback to TwelveData", .{symbol});
return td.fetchQuote(self.allocator, symbol) catch
return DataError.FetchFailed;
}
/// Compute trailing returns for a symbol (fetches candles + dividends).
/// Returns both as-of-date and month-end trailing returns.
/// As-of-date: end = latest close. Matches Morningstar "Trailing Returns" page.
/// Month-end: end = last business day of prior month. Matches Morningstar "Performance" page.
/// Compute trailing returns for a symbol (fetches candles + dividends + splits).
/// Returns both as-of-date and month-end trailing returns.
/// As-of-date: end = latest close. Matches Morningstar "Trailing Returns" page.
/// Month-end: end = last business day of prior month. Matches Morningstar "Performance" page.
///
/// `*_price` columns are split-adjusted, NOT dividend-adjusted (matches the
/// "price return" numbers public sources like Yahoo's chart-bar / FMP / Barchart
/// publish). `*_total` columns include dividend reinvestment (matches Morningstar
/// "Trailing Returns" / Yahoo "Performance Overview" / Koyfin "Total Return").
/// See `tmp/multi-ticker-audit.md` for the cross-validation evidence.
pub fn getTrailingReturns(self: *DataService, symbol: []const u8, opts: FetchOptions) DataError!struct {
asof_price: performance.TrailingReturns,
/// Total return as of the newest bar. Non-optional: it is always
/// computable once candles exist, because `performance.totalReturns`
/// degrades to the adj_close series internally when no dividend
/// record is available. It was `?TrailingReturns`, and the
/// optionality invited consumers to write
/// `asof_total orelse asof_price` - silently publishing a
/// price-only return labelled as total return, understated by the
/// full dividend yield.
asof_total: performance.TrailingReturns,
me_price: performance.TrailingReturns,
/// Month-end total return. Non-optional for the same reason.
me_total: performance.TrailingReturns,
candles: []Candle,
dividends: ?[]Dividend,
source: Source,
timestamp: i64,
} {
const candle_result = try self.getCandles(symbol, opts);
const c = candle_result.data;
if (c.len == 0) return DataError.FetchFailed;
const today = fmt.todayDate(self.io);
// Splits: needed to make raw `close` ratios meaningful across
// split boundaries (e.g. NVDA 10:1 on 2024-06-10). If the
// splits fetch fails, fall back to a no-splits empty slice -
// the price-return calculation will still be correct for
// tickers with no splits in the window (i.e. most of them).
var splits_buf: ?FetchResult(Split) = null;
defer if (splits_buf) |sb| sb.deinit();
const splits: []const Split = if (self.getSplits(symbol, opts)) |sr| blk: {
splits_buf = sr;
break :blk sr.data;
} else |_| &.{};
// As-of-date (end = last candle)
const asof_price = performance.trailingReturnsPriceOnly(c, splits);
// Month-end (end = last business day of prior month)
const me_price = performance.trailingReturnsPriceOnlyMonthEnd(c, splits, today);
// Total return. `performance.totalReturns` owns the
// dividend-reinvestment vs adj_close merge, including the
// no-dividend-record fallback, so there is one code path here
// rather than a branch per availability case.
var divs: ?[]Dividend = null;
if (self.getDividends(symbol, opts)) |div_result| {
divs = div_result.data;
} else |_| {}
const asof_total = performance.totalReturns(c, divs, today);
const me_total = performance.totalReturnsMonthEnd(c, divs, today);
return .{
.asof_price = asof_price,
.asof_total = asof_total,
.me_price = me_price,
.me_total = me_total,
.candles = c,
.dividends = divs,
.source = candle_result.source,
.timestamp = candle_result.timestamp,
};
}
/// Check if candle data is fresh in cache without full deserialization.
pub fn isCandleCacheFresh(self: *DataService, symbol: []const u8) bool {
var s = self.store();
return s.isCandleMetaFresh(symbol);
}
/// Read only the latest close price from cached candles (no full deserialization).
/// Returns null if no cached data exists.
pub fn getCachedLastClose(self: *DataService, symbol: []const u8) ?f64 {
var s = self.store();
return s.readLastClose(symbol);
}
/// Read the latest cached candle date for `symbol` without deserializing
/// the full candle history. Returns null if no cached metadata exists.
///
/// Callers should pair this with `isCandleCacheFresh` before trusting
/// the date: a stale cache entry can return a date from days or weeks
/// ago, which is fine for diagnostics but wrong for anything that
/// needs "the current market date".
pub fn getCachedLastDate(self: *DataService, symbol: []const u8) ?Date {
var s = self.store();
const mr = s.readCandleMeta(symbol) orelse return null;
return mr.meta.last_date;
}
/// Estimate wait time (in seconds) before a fetch for `data_type`
/// can proceed without blocking on its provider's rate limiter.
/// Returns 0 if a request can be made immediately, or if the
/// provider for this data type has no rate limiter. Returns null
/// if the relevant provider isn't instantiated yet (e.g., no API
/// key, or first call hasn't happened to lazy-init it).
///
/// The caller asks "how long until getX can proceed?" -- the
/// service maps data type to provider internally so the caller
/// doesn't have to know which provider serves which data.
pub fn estimateWaitSeconds(self: *DataService, data_type: cache.DataType) ?u64 {
const ns: u64 = switch (data_type) {
// Polygon-served: dividends and splits.
.dividends, .splits => if (self.pg) |*pg| pg.rate_limiter.estimateWaitNs() else return null,
// FMP-served: earnings.
.earnings => if (self.fmp) |*fmp| fmp.rate_limiter.estimateWaitNs() else return null,
// Cboe-served: options chains.
.options => if (self.cboe) |*cboe| cboe.rate_limiter.estimateWaitNs() else return null,
// EDGAR-served: ETF metrics, entity facts, ticker maps.
.etf_metrics, .entity_facts, .tickers_funds, .tickers_companies => if (self.edgar) |*e| e.rate_limiter.estimateWaitNs() else return null,
// Tiingo-served candles: 50/hour token bucket. When Tiingo
// isn't instantiated (no key), candles fall back to keyless
// Yahoo with no proactive limiter, so report 0 rather than
// null. `candles_meta` shares Tiingo's budget; `meta` isn't
// fetched; Wikidata (classification) has no published quota.
.candles_daily, .candles_meta => if (self.tg) |*tg| tg.rate_limiter.estimateWaitNs() else 0,
.classification, .meta => 0,
};
return if (ns == 0) 0 else @max(1, ns / std.time.ns_per_s);
}
/// Read candles from cache only (no network fetch). Used by TUI for display.
/// Returns null if no cached data exists or if the entry is a negative cache (fetch_failed).
///
/// `allocator` owns the returned `FetchResult.data`. Pass an
/// arena for "lives until reload" use cases (TUI per-portfolio
/// data); pass a per-call arena for CLI batch commands.
pub fn getCachedCandles(self: *DataService, allocator: std.mem.Allocator, symbol: []const u8) ?FetchResult(Candle) {
var s = self.store();
if (s.isNegative(symbol, .candles_daily)) return null;
const result = s.read(allocator, Candle, symbol, null, .any) orelse return null;
return .{ .data = result.data, .source = .cached, .timestamp = result.timestamp, .allocator = allocator };
}
/// Read dividends from cache only (no network fetch). See
/// `getCachedCandles` for the allocator contract.
pub fn getCachedDividends(self: *DataService, allocator: std.mem.Allocator, symbol: []const u8) ?FetchResult(Dividend) {
var s = self.store();
const result = s.read(allocator, Dividend, symbol, null, .any) orelse return null;
return .{ .data = result.data, .source = .cached, .timestamp = result.timestamp, .allocator = allocator };
}
/// Cache-only split read, mirroring `getCachedDividends`.
///
/// Splits travel with dividends wherever raw `close` is being
/// adjusted: `close` is not split-adjusted, so a total-return series
/// built from dividends alone reads a 2:1 split as a -50% month.
/// Callers that fetch one should fetch both.
pub fn getCachedSplits(self: *DataService, allocator: std.mem.Allocator, symbol: []const u8) ?FetchResult(Split) {
var s = self.store();
const result = s.read(allocator, Split, symbol, null, .any) orelse return null;
return .{ .data = result.data, .source = .cached, .timestamp = result.timestamp, .allocator = allocator };
}
// ── Portfolio price loading ──────────────────────────────────
/// Status emitted for each symbol during price loading.
pub const SymbolStatus = enum {
/// Price resolved from fresh cache.
cached,
/// About to attempt an API fetch (emitted before the network call).
fetching,
/// Price fetched successfully from API.
fetched,
/// API fetch failed but stale cached price was used.
failed_used_stale,
/// API fetch failed and no cached price exists.
failed,
};
/// Callback for progress reporting during price loading.
/// `context` is an opaque pointer to caller-owned state.
pub const ProgressCallback = struct {
context: *anyopaque,
on_progress: *const fn (ctx: *anyopaque, index: usize, total: usize, symbol: []const u8, status: SymbolStatus) void,
fn emit(self: ProgressCallback, index: usize, total: usize, symbol: []const u8, status: SymbolStatus) void {
self.on_progress(self.context, index, total, symbol, status);
}
};
// ── Consolidated Price Loading (Parallel Server + Sequential Provider) ──
/// Configuration for loadAllPrices.
/// Result of loadAllPrices operation.
pub const LoadAllResult = struct {
prices: std.StringHashMap(f64),
/// Number of symbols resolved from fresh local cache.
cached_count: usize,
/// Number of symbols synced from server.
server_synced_count: usize,
/// Number of symbols fetched from providers (rate-limited APIs).
provider_fetched_count: usize,
/// Number of symbols that failed all sources but used stale cache.
stale_count: usize,
/// Number of symbols that failed completely (no data).
failed_count: usize,
/// Latest candle date seen.
latest_date: ?Date,
/// Free the prices hashmap. Call this if you don't transfer ownership.
pub fn deinit(self: *LoadAllResult) void {
self.prices.deinit();
}
};
/// Progress callback for aggregate (parallel) progress reporting.
/// Called periodically during parallel operations with current counts.
pub const AggregateProgressCallback = struct {
context: *anyopaque,
on_progress: *const fn (ctx: *anyopaque, completed: usize, total: usize, phase: Phase) void,
pub const Phase = enum {
/// Checking local cache
cache_check,
/// Syncing from ZFIN_SERVER
server_sync,
/// Fetching from rate-limited providers
provider_fetch,
/// Done
complete,
};
fn emit(self: AggregateProgressCallback, completed: usize, total: usize, phase: Phase) void {
self.on_progress(self.context, completed, total, phase);
}
};
/// Thread-safe counter for parallel progress tracking.
const AtomicCounter = struct {
value: std.atomic.Value(usize) = std.atomic.Value(usize).init(0),
fn increment(self: *AtomicCounter) usize {
return self.value.fetchAdd(1, .monotonic);
}
fn load(self: *const AtomicCounter) usize {
return self.value.load(.monotonic);
}
};
/// Per-symbol result from parallel server sync.
const ServerSyncResult = struct {
symbol: []const u8,
success: bool,
};
/// Load prices for portfolio and watchlist symbols with automatic parallelization.
///
/// When ZFIN_SERVER is configured:
/// 1. Check local cache (fast, parallel-safe)
/// 2. Parallel sync from server for cache misses
/// 3. Sequential provider fallback for server failures
///
/// When ZFIN_SERVER is not configured:
/// Falls back to sequential loading with per-symbol progress.
///
/// Progress is reported via `aggregate_progress` for parallel phases
/// and `symbol_progress` for sequential provider fallback.
pub fn loadAllPrices(
self: *DataService,
portfolio_syms: ?[]const []const u8,
watch_syms: []const []const u8,
opts_in: FetchOptions,
aggregate_progress: ?AggregateProgressCallback,
symbol_progress: ?ProgressCallback,
) LoadAllResult {
// See `getCandles`. Folded here as well as there because this reads
// `opts` directly for its cache-scan fast path, not only via getCandles.
const opts = self.effectiveOptions(opts_in);
var result = LoadAllResult{
.prices = std.StringHashMap(f64).init(self.allocator),
.cached_count = 0,
.server_synced_count = 0,
.provider_fetched_count = 0,
.stale_count = 0,
.failed_count = 0,
.latest_date = null,
};
// Combine all symbols
const portfolio_count = if (portfolio_syms) |ps| ps.len else 0;
const watch_count = watch_syms.len;
const total_count = portfolio_count + watch_count;
if (total_count == 0) return result;
// Build combined symbol list
var all_symbols = std.ArrayList([]const u8).initCapacity(self.allocator, total_count) catch return result;
defer all_symbols.deinit(self.allocator);
if (portfolio_syms) |ps| {
for (ps) |sym| all_symbols.append(self.allocator, sym) catch |err| log.warn("loadAllPrices append portfolio sym({s}): {t}", .{ sym, err });
}
for (watch_syms) |sym| all_symbols.append(self.allocator, sym) catch |err| log.warn("loadAllPrices append watch sym({s}): {t}", .{ sym, err });
// force_refresh does NOT wipe the candle cache. It flows
// through to getCandles (the same `opts` we were handed), which
// ignores the TTL and does an incremental top-up - see the
// `--refresh-data=force` contract. The Phase-1 fast path below
// is skipped on force_refresh so every symbol is re-validated
// against the provider. A full wipe + re-download from scratch
// is reserved for `cache clear`.
// Sub-phase timing. `loadAllPrices` is the whole of the pre-paint wait,
// and its three phases fail in completely different ways: the cache
// scan is local I/O over the largest files in the cache, the server
// sync is one parallel round trip, and the provider fallback is
// rate-limited to a handful of requests per minute. A run that feels
// hung needs to say which of those it was sitting in - "prices: 90s"
// on its own does not distinguish a slow disk from a dead server.
const t0 = std.Io.Timestamp.now(self.io, .real);
var mark = t0;
const timing_on = self.config.timing;
const Sub = struct {
fn done(on: bool, io: std.Io, m: *std.Io.Timestamp, name: []const u8, n: usize) void {
const now = std.Io.Timestamp.now(io, .real);
const ms = @divFloor(now.nanoseconds - m.nanoseconds, std.time.ns_per_ms);
m.* = now;
if (on and ms >= 1) log.info("timing: loadAllPrices {s}: {d}ms ({d} symbols)", .{ name, ms, n });
}
};
// Phase 1: Check local cache (fast path)
var needs_fetch: std.ArrayList([]const u8) = .empty;
defer needs_fetch.deinit(self.allocator);
if (aggregate_progress) |p| p.emit(0, total_count, .cache_check);
for (all_symbols.items, 0..) |sym, i| {
if (!opts.force_refresh and self.isCandleCacheFresh(sym)) {
if (self.getCachedLastClose(sym)) |close| {
result.prices.put(sym, close) catch |err| log.warn("loadAllPrices cache-hit put({s}): {t}", .{ sym, err });
self.updateLatestDate(&result, sym);
}
result.cached_count += 1;
} else {
needs_fetch.append(self.allocator, sym) catch |err| log.warn("loadAllPrices needs_fetch append({s}): {t}", .{ sym, err });
}
// Report inside the loop, not just at the ends. This phase reads
// and validates every symbol's candle file - the bulk of the cache
// by size - so on a portfolio of any size it is hundreds of
// milliseconds of the pre-paint wait. Emitting only before and
// after left the display frozen for all of it, which reads as a
// hang rather than as work.
if (aggregate_progress) |p| p.emit(i + 1, total_count, .cache_check);
}
Sub.done(timing_on, self.io, &mark, "cache scan", total_count);
if (needs_fetch.items.len == 0) {
if (aggregate_progress) |p| p.emit(total_count, total_count, .complete);
return result;
}
if (timing_on) log.info("timing: loadAllPrices: {d} of {d} symbols need a fetch", .{ needs_fetch.items.len, total_count });
// Offline mode: skip server sync and provider fetch entirely.
// For symbols without a fresh cache, fall back to stale cache
// before giving up.
if (opts.skip_network) {
for (needs_fetch.items) |sym| {
if (self.getCachedLastClose(sym)) |close| {
result.prices.put(sym, close) catch |err| log.warn("loadAllPrices cache-hit put({s}): {t}", .{ sym, err });
self.updateLatestDate(&result, sym);
result.stale_count += 1;
} else {
result.failed_count += 1;
}
}
if (aggregate_progress) |p| p.emit(total_count, total_count, .complete);
return result;
}
// Phase 2: Server sync (parallel if server configured)
var server_failures: std.ArrayList([]const u8) = .empty;
defer server_failures.deinit(self.allocator);
if (self.config.server_url != null) {
self.parallelServerSync(
needs_fetch.items,
&result,
&server_failures,
aggregate_progress,
total_count,
);
} else {
// No server - all need provider fetch
for (needs_fetch.items) |sym| {
server_failures.append(self.allocator, sym) catch |err| log.warn("loadAllPrices server_failures append({s}): {t}", .{ sym, err });
}
}
Sub.done(timing_on, self.io, &mark, "server sync", needs_fetch.items.len);
// Phase 3: Sequential provider fallback for server failures
if (server_failures.items.len > 0) {
// The expensive one, and the reason this breakdown exists: the
// provider is rate limited, so this scales in minutes where the
// phases above scale in milliseconds.
if (timing_on) log.info("timing: loadAllPrices: {d} symbols fell through to the PROVIDER (rate limited)", .{server_failures.items.len});
}
if (server_failures.items.len > 0) {
if (aggregate_progress) |p| p.emit(
result.cached_count + result.server_synced_count,
total_count,
.provider_fetch,
);
self.sequentialProviderFetch(
server_failures.items,
&result,
symbol_progress,
total_count - server_failures.items.len, // offset for progress display
opts,
);
}
Sub.done(timing_on, self.io, &mark, "provider fallback", server_failures.items.len);
if (aggregate_progress) |p| p.emit(total_count, total_count, .complete);
return result;
}
/// Fetch live intraday quotes for `symbols`, returning a map of
/// symbol -> live last price. Symbols whose quote fetch fails (or
/// that the provider can't price) are simply absent; the caller
/// falls back to the last cached close.
///
/// This is a pure live-price fetch: quotes are never cached, so it
/// neither reads nor writes the candle cache. It exists for the
/// TUI refresh key (`r`), whose job is "give me current prices,"
/// distinct from candle-history maintenance (TTL/startup) and from
/// `--refresh-data=force` (incremental candle top-up).
///
/// Provider is `Config.effectiveLiveQuoteProvider()`:
/// - `.yahoo` (default): keyless, parallel per-symbol fan-out.
/// - `.tiingo`: a single batched `/iex` request returning Tiingo's
/// real-time IEX reference price (`tngoLast`). On ANY failure
/// (no key, network, auth, parse) it degrades to the Yahoo path
/// so a refresh never hard-fails.
///
/// The returned map's keys borrow `symbols`: keep `symbols` alive
/// while using the map, and `deinit()` the map when done.
pub fn loadLiveQuotes(self: *DataService, symbols: []const []const u8) std.StringHashMap(f64) {
var prices = std.StringHashMap(f64).init(self.allocator);
if (symbols.len == 0) return prices;
self.assertNetworkAllowed("loadLiveQuotes");
switch (self.config.effectiveLiveQuoteProvider()) {
.yahoo => self.loadLiveQuotesYahoo(symbols, &prices),
.tiingo => self.loadLiveQuotesTiingo(symbols, &prices) catch |err| {
log.warn("loadLiveQuotes: Tiingo path failed ({t}); falling back to Yahoo", .{err});
// Drop any partial fill so we never mix providers.
prices.clearRetainingCapacity();
self.loadLiveQuotesYahoo(symbols, &prices);
},
}
return prices;
}
/// Yahoo live-quote fan-out: one task per symbol in a single
/// `std.Io.Group`, each with its own `Yahoo` client (a shared
/// `std.http.Client` is not safe across threads - see `tryOneSync`).
/// Yahoo is keyless with no shared rate limiter, so per-worker
/// clients are safe. Relies on a thread-safe `allocator`/`io`, the
/// same assumption the server-sync fan-out makes.
fn loadLiveQuotesYahoo(self: *DataService, symbols: []const []const u8, prices: *std.StringHashMap(f64)) void {
const QuoteSlot = struct { symbol: []const u8, price: ?f64 = null };
const slots = self.allocator.alloc(QuoteSlot, symbols.len) catch return;
defer self.allocator.free(slots);
for (slots, 0..) |*slot, i| slot.* = .{ .symbol = symbols[i] };
const worker = struct {
fn run(io: std.Io, allocator: std.mem.Allocator, slot: *QuoteSlot) std.Io.Cancelable!void {
try io.checkCancel();
var yh = Yahoo.init(io, allocator);
defer yh.deinit();
// Quote borrows `symbol` and carries no owned memory,
// so the f64 close is all we keep - nothing to free.
slot.price = if (yh.fetchQuote(allocator, slot.symbol)) |q| q.close else |_| null;
}
};
var group: std.Io.Group = .init;
for (slots) |*slot| group.async(self.io, worker.run, .{ self.io, self.allocator, slot });
group.await(self.io) catch |err| log.debug("loadLiveQuotes group await: {t}", .{err});
for (slots) |slot| {
if (slot.price) |p| prices.put(slot.symbol, p) catch |err| log.warn("loadLiveQuotes put({s}): {t}", .{ slot.symbol, err });
}
}
/// Tiingo live-quote path: a single batched `/iex` request for all
/// `symbols`, returning the real-time IEX reference price
/// (`tngoLast`). Results are keyed back to the caller's `symbols`
/// slices (case-insensitive, since Tiingo upper-cases tickers and
/// returns them in arbitrary order), so the map's keys still borrow
/// `symbols`. Symbols Tiingo can't price (mutual funds) or with a
/// null reference price are absent -> candle-close fallback. Errors
/// propagate to `loadLiveQuotes`, which degrades to Yahoo.
fn loadLiveQuotesTiingo(self: *DataService, symbols: []const []const u8, prices: *std.StringHashMap(f64)) !void {
const tg = try self.getProvider(Tiingo);
const quotes = try tg.fetchQuotes(self.allocator, symbols);
defer self.allocator.free(quotes);
for (quotes) |q| {
const price = q.tngo_last orelse continue;
// Map Tiingo's echoed-back (often upper-cased, arbitrary
// order) ticker to the caller's own slice so the result
// map's keys borrow `symbols` per the contract.
const key = fmt.findIgnoreCase(symbols, q.ticker()) orelse continue;
prices.put(key, price) catch |err| log.warn("loadLiveQuotes put({s}): {t}", .{ key, err });
}
}
/// Parallel server sync via `std.Io.Group`.
///
/// Concurrency shape: one task per symbol, spawned into a
/// single `Group`. The `std.Io` implementation owns
/// scheduling and concurrency limits (e.g. `Io.Threaded`
/// sizes its pool from CPU count); we don't second-guess it
/// with our own worker cap or work-stealing queue.
///
/// Each task hits `io.checkCancel()` before its sync, so a
/// cancelation request propagating through `Group.await`
/// stops pending work at task granularity.
fn parallelServerSync(
self: *DataService,
symbols: []const []const u8,
result: *LoadAllResult,
failures: *std.ArrayList([]const u8),
aggregate_progress: ?AggregateProgressCallback,
total_count: usize,
) void {
if (aggregate_progress) |p| p.emit(result.cached_count, total_count, .server_sync);
// Shared state for tasks
var completed = AtomicCounter{};
const sync_results = self.allocator.alloc(ServerSyncResult, symbols.len) catch {
// Allocation failed - fall back to marking all as failures
for (symbols) |sym| failures.append(self.allocator, sym) catch |err| log.warn("parallelServerSync slots-alloc-fallback failures append({s}): {t}", .{ sym, err });
return;
};
defer self.allocator.free(sync_results);
// Initialize results
for (sync_results, 0..) |*sr, i| {
sr.* = .{ .symbol = symbols[i], .success = false };
}
const worker = struct {
fn run(io: std.Io, svc: *DataService, slot: *ServerSyncResult, done: *AtomicCounter) std.Io.Cancelable!void {
defer _ = done.increment();
try io.checkCancel();
slot.success = svc.syncCandlesFromServer(slot.symbol);
}
};
// Spawn one task per symbol. Group.async requires an
// eventual Group.await/cancel to release resources; the
// single await below covers all paths.
var group: std.Io.Group = .init;
for (sync_results) |*sr| {
group.async(self.io, worker.run, .{ self.io, self, sr, &completed });
}
// Progress reporting while the group runs
if (aggregate_progress) |p| {
while (completed.load() < symbols.len) {
std.Io.sleep(self.io, std.Io.Duration.fromMilliseconds(50), .awake) catch |err| {
log.debug("parallelServerSync progress-poll sleep interrupted: {t}", .{err});
break;
};
p.emit(result.cached_count + completed.load(), total_count, .server_sync);
}
}
// Wait for all tasks. On cancelation the unstarted tasks
// exit at their checkCancel point; partial results (slots
// that completed) are still processed below - they came
// from successful cache writes.
group.await(self.io) catch |err| {
log.debug("parallelServerSync group await: {t}", .{err});
};
// Process results
for (sync_results) |sr| {
if (sr.success) {
// Server sync succeeded - read from cache
if (self.getCachedLastClose(sr.symbol)) |close| {
result.prices.put(sr.symbol, close) catch |err| log.warn("syncFromServer cache-after-sync put({s}): {t}", .{ sr.symbol, err });
self.updateLatestDate(result, sr.symbol);
result.server_synced_count += 1;
} else {
// Sync said success but can't read cache - treat as failure
failures.append(self.allocator, sr.symbol) catch |err| log.warn("syncFromServer success-but-no-cache failures append({s}): {t}", .{ sr.symbol, err });
}
} else {
failures.append(self.allocator, sr.symbol) catch |err| log.warn("syncFromServer fail-result failures append({s}): {t}", .{ sr.symbol, err });
}
}
}
/// Sequential provider fetch for symbols that failed server sync.
fn sequentialProviderFetch(
self: *DataService,
symbols: []const []const u8,
result: *LoadAllResult,
progress: ?ProgressCallback,
index_offset: usize,
opts: FetchOptions,
) void {
const total = index_offset + symbols.len;
for (symbols, 0..) |sym, i| {
const display_idx = index_offset + i;
// Notify: about to fetch
if (progress) |p| p.emit(display_idx, total, sym, .fetching);
// Try provider fetch
if (self.getCandles(sym, opts)) |candle_result| {
defer self.allocator.free(candle_result.data);
if (candle_result.data.len > 0) {
const last = candle_result.data[candle_result.data.len - 1];
result.prices.put(sym, last.close) catch |err| log.warn("loadAllPrices candle-close put({s}): {t}", .{ sym, err });
if (result.latest_date == null or last.date.days > result.latest_date.?.days) {
result.latest_date = last.date;
}
}
result.provider_fetched_count += 1;
if (progress) |p| p.emit(display_idx, total, sym, .fetched);
continue;
} else |_| {}
// Provider failed - try stale cache
result.failed_count += 1;
if (self.getCachedLastClose(sym)) |close| {
result.prices.put(sym, close) catch |err| log.warn("loadAllPrices stale-fallback put({s}): {t}", .{ sym, err });
result.stale_count += 1;
if (progress) |p| p.emit(display_idx, total, sym, .failed_used_stale);
} else {
if (progress) |p| p.emit(display_idx, total, sym, .failed);
}
}
}
/// Update latest_date in result from cached candle metadata.
fn updateLatestDate(self: *DataService, result: *LoadAllResult, symbol: []const u8) void {
var s = self.store();
if (s.readCandleMeta(symbol)) |cm| {
const d = cm.meta.last_date;
if (result.latest_date == null or d.days > result.latest_date.?.days) {
result.latest_date = d;
}
}
}
// ── CUSIP Resolution ──────────────────────────────────────────
/// Look up multiple CUSIPs in a single batch request via OpenFIGI.
/// Results array is parallel to the input cusips array (same length, same order).
/// Caller owns the returned slice and all strings within each CusipResult.
pub fn lookupCusips(self: *DataService, cusips: []const []const u8) DataError![]CusipResult {
return OpenFigi.lookupCusips(self.io, self.allocator, cusips, self.config.openfigi_key) catch
return DataError.FetchFailed;
}
/// A single CUSIP-to-ticker mapping record in the cache file.
const CusipEntry = struct {
cusip: []const u8 = "",
ticker: []const u8 = "",
};
/// CUSIP->ticker lookup table loaded from `cusip_tickers.srf`.
///
/// Zero-copy: keys and values are slices into `backing` (the raw
/// file bytes parsed with `parse_allocator = .none`). Nothing is
/// duped per entry - the whole-file buffer IS the storage, and it
/// stays alive for the table's lifetime, released together with
/// the map table in `deinit`.
///
/// This is the L1 tier of CUSIP resolution: callers consult it
/// before reaching for the server or OpenFIGI.
pub const CusipTickerMap = struct {
map: std.StringHashMap([]const u8),
/// Raw bytes of `cusip_tickers.srf`; every map key and value
/// points into this buffer. `&.{}` when the file was missing
/// or unreadable (freeing a zero-length slice is a no-op).
backing: []const u8,
pub fn get(self: CusipTickerMap, cusip: []const u8) ?[]const u8 {
return self.map.get(cusip);
}
pub fn contains(self: CusipTickerMap, cusip: []const u8) bool {
return self.map.contains(cusip);
}
pub fn count(self: CusipTickerMap) u32 {
return self.map.count();
}
/// Release the map table and the backing buffer. Both were
/// allocated with the map's allocator at load time, so we
/// reuse it here - the two lifetimes are bound together by
/// construction, which is the whole point of the wrapper.
pub fn deinit(self: *CusipTickerMap) void {
const allocator = self.map.allocator;
self.map.deinit();
allocator.free(self.backing);
}
};
/// Load the CUSIP->ticker cache file into a `CusipTickerMap`. The
/// returned table owns the file bytes; release it with
/// `CusipTickerMap.deinit`.
///
/// Missing file -> empty table (the common first-run case). First
/// occurrence wins on duplicate CUSIPs, which tolerates the
/// historical double-append bug in cache files written before
/// `cacheCusipTicker` learned to dedup.
///
/// The on-disk format is CUSIP-keyed (`cusip::X,ticker::Y`); the
/// returned map is keyed the same way for O(1) forward lookup.
pub fn loadCusipTickerMap(self: *DataService, allocator: std.mem.Allocator) CusipTickerMap {
const map = std.StringHashMap([]const u8).init(allocator);
const path = std.fs.path.join(allocator, &.{ self.config.cache_dir, "cusip_tickers.srf" }) catch
return .{ .map = map, .backing = &.{} };
defer allocator.free(path);
const data = std.Io.Dir.cwd().readFileAlloc(self.io, path, allocator, .limited(4 * 1024 * 1024)) catch
return .{ .map = map, .backing = &.{} };
// From here `data` is the table's backing store: keys and
// values are slices into it (parse_allocator = .none, so the
// parser borrows rather than copies). Freed by
// `CusipTickerMap.deinit`, never here - that's the lifetime
// contract that lets us skip per-entry dupes entirely.
var result: CusipTickerMap = .{ .map = map, .backing = data };
var reader = std.Io.Reader.fixed(data);
var it = srf.iterator(&reader, allocator, .{ .parse_allocator = .none }) catch return result;
defer it.deinit();
while (it.next() catch return result) |fields| {
const entry = fields.to(CusipEntry, srf_opts.machine_written) catch continue;
if (entry.cusip.len == 0 or entry.ticker.len == 0) continue;
// First occurrence wins; getOrPut stores the borrowed
// slices directly - they live in `backing`, no dupe.
const gop = result.map.getOrPut(entry.cusip) catch continue;
if (!gop.found_existing) gop.value_ptr.* = entry.ticker;
}
return result;
}
/// Append CUSIP->ticker mappings to `cusip_tickers.srf`, skipping
/// any whose CUSIP is already on disk and any duplicates within
/// `entries`. One read + one atomic write regardless of batch size.
///
/// Read-append-atomic-write (rather than open-for-append) so a
/// concurrent reader never sees a valid header plus a partial
/// trailing record - see `cache/store.zig appendRaw` for the same
/// pattern and rationale. `#!srfv1` directives are emitted only
/// when the file is being created.
fn appendCusipEntries(self: *DataService, entries: []const CusipEntry) void {
if (entries.len == 0) return;
// One load gives us both the dedup set and the existing bytes
// to concat (`backing`). Missing/empty file -> empty map + empty
// backing -> directives emitted below.
var existing_map = self.loadCusipTickerMap(self.allocator);
defer existing_map.deinit();
const existing = existing_map.backing;
// Keep only entries new to the file and unique within the batch.
var seen = std.StringHashMap(void).init(self.allocator);
defer seen.deinit();
var to_write: std.ArrayList(CusipEntry) = .empty;
defer to_write.deinit(self.allocator);
for (entries) |e| {
if (e.cusip.len == 0 or e.ticker.len == 0) continue;
if (existing_map.contains(e.cusip)) continue;
const gop = seen.getOrPut(e.cusip) catch continue;
if (gop.found_existing) continue;
to_write.append(self.allocator, e) catch continue;
}
if (to_write.items.len == 0) return;
const path = std.fs.path.join(self.allocator, &.{ self.config.cache_dir, "cusip_tickers.srf" }) catch return;
defer self.allocator.free(path);
if (std.fs.path.dirnamePosix(path)) |dir| {
std.Io.Dir.cwd().createDirPath(self.io, dir) catch |err| log.warn("cusip-cache createDirPath({s}): {t}", .{ dir, err });
}
const emit_directives = existing.len == 0;
var aw: std.Io.Writer.Allocating = .init(self.allocator);
defer aw.deinit();
aw.writer.print("{f}", .{srf.fmt(CusipEntry, to_write.items, .{ .emit_directives = emit_directives })}) catch return;
const encoded = aw.writer.buffered();
if (encoded.len == 0) return;
// Concat existing + new, then atomic-write.
const combined = self.allocator.alloc(u8, existing.len + encoded.len) catch return;
defer self.allocator.free(combined);
@memcpy(combined[0..existing.len], existing);
@memcpy(combined[existing.len..], encoded);
atomic.writeFileAtomic(self.io, self.allocator, path, combined) catch |err| log.warn("cusip-cache writeFileAtomic({s}): {t}", .{ path, err });
}
/// Append a single CUSIP->ticker mapping to the cache file
/// (dedup-aware). Thin wrapper over `appendCusipEntries`; the
/// `lookup` command's single-CUSIP path.
pub fn cacheCusipTicker(self: *DataService, cusip: []const u8, ticker: []const u8) void {
self.appendCusipEntries(&.{.{ .cusip = cusip, .ticker = ticker }});
}
/// Resolve a set of CUSIPs to tickers via the three-tier cascade,
/// persisting newly-learned mappings to `cusip_tickers.srf` (union
/// policy: the local file accumulates everything it ever learns and
/// converges toward the shared server set).
///
/// Tiers, cheapest first:
/// L1 local `cusip_tickers.srf` (always; no network)
/// L2 server `GET /cusips` whole-file sync (if ZFIN_SERVER set)
/// L3 OpenFIGI batch lookup (whatever still misses)
///
/// `skip_network = true` restricts resolution to L1 (the local
/// cache) - for offline mode (`--refresh-data=never`). L2/L3 and
/// the persist-back are skipped entirely; cached CUSIPs still
/// resolve, uncached ones stay unresolved.
///
/// Best-effort: network failures degrade to "fewer entries
/// resolved" rather than erroring. The returned `CusipTickerMap` is
/// a zero-copy view over the (possibly just-rewritten) local file
/// and covers every CUSIP any tier could resolve. Callers resolve
/// forward-per-holding: look up each holding's CUSIP against it,
/// which sidesteps the "do I have every CUSIP for this ticker?"
/// completeness problem entirely.
///
/// Empty/duplicate CUSIPs in `cusips` are ignored. The caller owns
/// the returned map (`deinit`); pass a scratch allocator to scope
/// it to a single command invocation.
pub fn resolveCusips(self: *DataService, allocator: std.mem.Allocator, cusips: []const []const u8, skip_network: bool) CusipTickerMap {
var result = self.loadCusipTickerMap(allocator);
// Offline mode serves only L1. Also the warm-cache fast path:
// when nothing is missing there's no scratch, no network, no
// rewrite.
if (skip_network or !anyMissing(result, cusips)) return result;
// Scratch arena for minted entries; decouples their lifetime
// from the server body / OpenFIGI result buffers freed below.
var scratch = std.heap.ArenaAllocator.init(self.allocator);
defer scratch.deinit();
const sa = scratch.allocator();
var minted = std.StringHashMap([]const u8).init(sa); // cusip -> ticker
// L2: server whole-file sync. Degrades to no-op until the
// `GET /cusips` route exists (a 404 surfaces as NotFound from
// client.get); when it lands it's purely additive - no change
// here. The server is expected to serve the file via its
// existing `handleStaticSrfFile` machinery (same shape as
// `/_edgar/tickers_funds`).
if (self.config.server_url) |server_url| {
if (self.fetchServerCusips(server_url)) |body| {
defer self.allocator.free(body);
mergeCusipBody(sa, &minted, result, body);
}
}
// L3: OpenFIGI for whatever still misses.
self.mintMissingViaOpenFigi(sa, &minted, result, cusips);
if (minted.count() == 0) return result; // nothing new learned
// Persist the union, then reload so the returned map is a clean
// single-buffer zero-copy view over the updated file.
var ents: std.ArrayList(CusipEntry) = .empty;
// Reserve up front so the collection loop is infallible. On OOM
// (vanishingly unlikely for a small list), skip persistence and
// return the L1 view - some CUSIPs stay unresolved this run
// rather than erroring.
ents.ensureTotalCapacity(sa, minted.count()) catch return result;
var mit = minted.iterator();
while (mit.next()) |kv| ents.appendAssumeCapacity(.{ .cusip = kv.key_ptr.*, .ticker = kv.value_ptr.* });
self.appendCusipEntries(ents.items);
result.deinit();
return self.loadCusipTickerMap(allocator);
}
/// True if any non-empty CUSIP in `cusips` is absent from `map`.
fn anyMissing(map: CusipTickerMap, cusips: []const []const u8) bool {
for (cusips) |c| {
if (c.len == 0) continue;
if (!map.contains(c)) return true;
}
return false;
}
/// Merge a CUSIP->ticker SRF body (as served by `GET /cusips`) into
/// `out`, skipping any CUSIP already present in `have` or `out`.
/// Strings are duped into `arena`. Pure with respect to I/O, so it's
/// unit-tested directly with fixture bytes (the live L2 path can't
/// be exercised until the server route exists).
fn mergeCusipBody(arena: std.mem.Allocator, out: *std.StringHashMap([]const u8), have: CusipTickerMap, body: []const u8) void {
var reader = std.Io.Reader.fixed(body);
var it = srf.iterator(&reader, arena, .{ .parse_allocator = .none }) catch return;
defer it.deinit();
while (it.next() catch return) |fields| {
const e = fields.to(CusipEntry, srf_opts.machine_written) catch continue;
if (e.cusip.len == 0 or e.ticker.len == 0) continue;
if (have.contains(e.cusip) or out.contains(e.cusip)) continue;
const kc = arena.dupe(u8, e.cusip) catch continue;
const vc = arena.dupe(u8, e.ticker) catch continue;
out.put(kc, vc) catch continue;
}
}
/// Build the auth header slice for a server-sync request. Writes the
/// `X-API-Key` header into `buf` (caller-owned; must outlive the
/// request) and returns a one-element slice of it, or an empty slice
/// when no key is configured. The key points into `Config`
/// (process-lifetime), so the value slice stays valid. Free function
/// (not a method) so it's unit-testable without a `DataService`.
fn serverAuthHeaders(server_api_key: ?[]const u8, buf: *[1]std.http.Header) []const std.http.Header {
const key = server_api_key orelse return &.{};
buf[0] = .{ .name = "X-API-Key", .value = key };
return buf[0..1];
}
/// L2 seam: fetch the whole CUSIP->ticker map from the server via
/// `GET {server}/cusips`. Returns the raw SRF body (caller frees
/// with `self.allocator`) or null on any failure. Best-effort: no
/// retry and no torn-body archival (this is a shared reference
/// file, not per-symbol cache) - a bad/absent response just
/// degrades to the OpenFIGI tier.
fn fetchServerCusips(self: *DataService, server_url: []const u8) ?[]u8 {
const url = std.fmt.allocPrint(self.allocator, "{s}/cusips", .{server_url}) catch return null;
defer self.allocator.free(url);
var client = http.Client.init(self.io, self.allocator);
defer client.deinit();
var hdr_buf: [1]std.http.Header = .{.{ .name = "", .value = "" }};
const extra_headers = serverAuthHeaders(self.config.server_api_key, &hdr_buf);
var response = client.request(.GET, url, null, extra_headers) catch |err| {
log.debug("cusips server sync failed: {s}", .{@errorName(err)});
return null;
};
defer response.deinit();
if (!cache.Store.looksCompleteSrf(response.body)) {
log.debug("cusips server response not complete SRF ({d} bytes) - ignoring", .{response.body.len});
return null;
}
return self.allocator.dupe(u8, response.body) catch null;
}
/// L3: resolve still-missing CUSIPs through OpenFIGI (batched 100
/// per request, the API's job limit), recording hits into `out`
/// (duped into `arena`). De-dups the lookup set against `have`,
/// `out`, and itself. Best-effort: a failed batch logs and is
/// skipped; remaining batches still run.
fn mintMissingViaOpenFigi(self: *DataService, arena: std.mem.Allocator, out: *std.StringHashMap([]const u8), have: CusipTickerMap, cusips: []const []const u8) void {
var seen = std.StringHashMap(void).init(arena);
var to_lookup: std.ArrayList([]const u8) = .empty;
for (cusips) |c| {
if (c.len == 0) continue;
if (have.contains(c) or out.contains(c)) continue;
const gop = seen.getOrPut(c) catch continue;
if (gop.found_existing) continue;
to_lookup.append(arena, c) catch continue;
}
if (to_lookup.items.len == 0) return;
const batch_size = 100; // OpenFIGI accepts up to 100 jobs/request.
var start: usize = 0;
while (start < to_lookup.items.len) : (start += batch_size) {
const end = @min(start + batch_size, to_lookup.items.len);
const batch = to_lookup.items[start..end];
const figi = self.lookupCusips(batch) catch |err| {
log.warn("resolveCusips: OpenFIGI lookup of {d} CUSIP(s) failed: {s}", .{ batch.len, @errorName(err) });
continue;
};
defer {
for (figi) |r| {
if (r.ticker) |t| self.allocator.free(t);
if (r.name) |n| self.allocator.free(n);
if (r.security_type) |s| self.allocator.free(s);
}
self.allocator.free(figi);
}
// Results are parallel to `batch` (same length + order).
for (figi, 0..) |r, i| {
if (!r.found) continue;
const ticker = r.ticker orelse continue;
const kc = arena.dupe(u8, batch[i]) catch continue;
const vc = arena.dupe(u8, ticker) catch continue;
out.put(kc, vc) catch continue;
}
}
}
// ── Utility ──────────────────────────────────────────────────
/// Sleep before retrying after a rate limit error.
/// Uses the provider's rate limiter if available, otherwise a fixed 10s backoff.
fn rateLimitBackoff(self: *DataService) void {
if (self.td) |*td| {
td.rate_limiter.backoff();
} else {
std.Io.sleep(self.io, std.Io.Duration.fromSeconds(10), .awake) catch |err| log.debug("rate-limit backoff sleep interrupted: {t}", .{err});
}
}
// ── Server sync ──────────────────────────────────────────────
/// Try to sync a cache file from the configured zfin-server.
/// Returns true if the file was successfully synced, false on any error.
/// Silently returns false if no server is configured.
///
/// Applies a single retry with a short delay when the first attempt
/// fails at the HTTP layer OR produces a torn body (integrity
/// mismatch / `looksCompleteSrf` rejection). Motivation: refreshes
/// fan out 20+ symbols across 8 parallel threads, and the tear
/// pattern we've observed so far looks transient per-connection.
/// One retry papers over single-packet hiccups without dramatically
/// extending refresh wall time. If the retry also fails the
/// archive grows by one more `.bin`/`.meta` pair - two captures
/// from the same refresh are the most valuable diagnostic signal
/// we can produce (same body shape? same byte offset? same time
/// delta? all answers we can't get from a single failure).
fn syncFromServer(self: *DataService, symbol: []const u8, data_type: cache.DataType) bool {
const server_url = self.config.server_url orelse return false;
const endpoint = switch (data_type) {
.candles_daily => "/candles",
.candles_meta => "/candles_meta",
.dividends => "/dividends",
.earnings => "/earnings",
.options => "/options",
.splits => "/splits",
.meta => return false,
.classification => "/classification",
.etf_metrics => "/etf_metrics",
.entity_facts => "/entity_facts",
// Provider-internal cache files (ticker-map indexes)
// are not served - clients fetch them directly from
// the SEC. The DataService caches the JSON via
// `Store` after fetching; the server has no role.
.tickers_funds, .tickers_companies => return false,
};
const full_url = std.fmt.allocPrint(self.allocator, "{s}/{s}{s}", .{ server_url, symbol, endpoint }) catch return false;
defer self.allocator.free(full_url);
const max_attempts: u8 = 2;
const retry_delay_ms: u64 = 250;
var attempt: u8 = 0;
while (attempt < max_attempts) : (attempt += 1) {
if (attempt > 0) {
log.debug(
"{s}: retrying {s} server sync (attempt {d}/{d}) after {d}ms delay",
.{ symbol, @tagName(data_type), attempt + 1, max_attempts, retry_delay_ms },
);
std.Io.sleep(self.io, std.Io.Duration.fromMilliseconds(retry_delay_ms), .awake) catch |err| log.debug("syncFromServer retry-delay sleep interrupted: {t}", .{err});
}
switch (self.tryOneSync(symbol, data_type, full_url)) {
.ok => return true,
// Torn or network error - retry if attempts remain.
.torn, .net_err => {},
}
}
return false;
}
const SyncAttempt = enum { ok, torn, net_err };
/// One attempt at syncing a file from the server. Archives a torn
/// body when detected but does NOT retry - the caller decides that.
fn tryOneSync(self: *DataService, symbol: []const u8, data_type: cache.DataType, full_url: []const u8) SyncAttempt {
// Per-attempt start/finish trace. The "started" line emits
// before any blocking call; the "finished" line emits on every
// exit path. If a sync wedges in `client.get`, you'll see the
// started line with no matching finished line - the missing
// finished entries identify which symbols are stuck. Pair this
// with the per-stage `http: stage=...` lines from `net/http.zig`
// to pinpoint which transport stage stalled.
//
// wall-clock required: per-attempt elapsed for diagnosing
// partial-success/stall patterns under parallel fan-out.
// `.awake` (monotonic) avoids spurious negatives on clock skew.
const t_start = std.Io.Timestamp.now(self.io, .awake).nanoseconds;
log.debug("{s}: tryOneSync started ({s})", .{ symbol, @tagName(data_type) });
var client = http.Client.init(self.io, self.allocator);
defer client.deinit();
var hdr_buf: [1]std.http.Header = .{.{ .name = "", .value = "" }};
const extra_headers = serverAuthHeaders(self.config.server_api_key, &hdr_buf);
var response = client.request(.GET, full_url, null, extra_headers) catch |err| {
const elapsed_ms = @divTrunc(std.Io.Timestamp.now(self.io, .awake).nanoseconds - t_start, std.time.ns_per_ms);
// Operator-visible: surfaces meaningful failures
// (`NoAddressReturned`, `ConnectionRefused`,
// `TlsInitializationFailed`, etc.) instead of swallowing
// them. Network-shaped errors are exactly what the user
// needs to see when sync stops working - keeping this at
// debug level meant a DNS-truncation bug was visible only
// to anyone running with debug logging on, which cost
// hours of diagnosis time.
log.warn("{s}: server sync failed for {s}: {s} (elapsed_ms={d})", .{ symbol, @tagName(data_type), @errorName(err), elapsed_ms });
log.debug("{s}: tryOneSync finished ({s}) result=net_err elapsed_ms={d}", .{ symbol, @tagName(data_type), elapsed_ms });
return .net_err;
};
defer response.deinit();
// Integrity check: if the server advertised an ETag in
// `"sha256:<hex>"` form, compare the body's actual sha256
// against it. Catches mid-stream truncation that Zig's
// std.http.Client.fetch silently accepts on the Content-Length
// path (EndOfStream from a cut transport is swallowed as a
// normal termination). Archive the mismatching body with the
// advertised etag so post-mortem can see exactly what was
// promised vs what arrived. Deployments with no ETag or a
// non-sha256 etag fall through to `looksCompleteSrf` below
// (backward-compatible with pre-fix servers).
switch (response.verifyIntegrity()) {
.mismatch => |m| {
cache.Store.archiveTornBody(
self.io,
self.allocator,
self.config.cache_dir,
symbol,
data_type,
response.body,
.{
.failure_reason = .etag_mismatch,
.http_status = @intFromEnum(response.status),
.server_url = full_url,
.server_etag = response.etag,
},
) catch |err| {
log.debug(
"{s}: failed to archive etag-mismatch {s} body: {s}",
.{ symbol, @tagName(data_type), @errorName(err) },
);
};
log.debug(
"{s}: {s} server response failed integrity check ({d} bytes, expected sha256={s}, actual={s}) - archived under _torn/, not writing to cache",
.{ symbol, @tagName(data_type), response.body.len, m.expected_hex, m.actual_hex },
);
log.debug("{s}: tryOneSync finished ({s}) result=torn elapsed_ms={d}", .{ symbol, @tagName(data_type), @divTrunc(std.Io.Timestamp.now(self.io, .awake).nanoseconds - t_start, std.time.ns_per_ms) });
return .torn;
},
.ok, .not_applicable => {},
}
// Validate the response body looks like a complete SRF file before
// writing it to cache. This guards against HTTP body truncation
// (TCP reset, Content-Length mismatch, proxy that flushed a
// partial response, etc.) - torn bodies get written atomically
// to the cache otherwise, producing the classic SRF parse error
// on the next read:
// error(srf): custom parse of value YYYY-MM failed : InvalidDateFormat
//
// When the check rejects a body, archive the raw bytes + context
// under `{cache_dir}/_torn/` so the next time this recurs we
// have ammunition for root-cause analysis. The log line is kept
// at debug level on purpose - user explicitly asked that routine
// rejections not be noisy in production runs. The `.meta`
// sidecar on disk is the durable signal.
if (!cache.Store.looksCompleteSrf(response.body)) {
cache.Store.archiveTornBody(
self.io,
self.allocator,
self.config.cache_dir,
symbol,
data_type,
response.body,
.{
.failure_reason = .looks_complete_srf_failed,
.http_status = @intFromEnum(response.status),
.server_url = full_url,
.server_etag = response.etag,
},
) catch |err| {
log.debug(
"{s}: failed to archive torn {s} body: {s}",
.{ symbol, @tagName(data_type), @errorName(err) },
);
};
log.debug(
"{s}: rejecting torn {s} server response ({d} bytes) - archived under _torn/, not writing to cache",
.{ symbol, @tagName(data_type), response.body.len },
);
log.debug("{s}: tryOneSync finished ({s}) result=torn elapsed_ms={d}", .{ symbol, @tagName(data_type), @divTrunc(std.Io.Timestamp.now(self.io, .awake).nanoseconds - t_start, std.time.ns_per_ms) });
return .torn;
}
// Write to local cache
var s = self.store();
// Never let the shared cache move a symbol BACKWARDS. The server's
// bytes are written verbatim, `#!expires=` included, so its view of
// freshness becomes the client's - and if the server's copy carries an
// older bar than the one already here, an unconditional write replaces
// good local data with worse and stamps it authoritative.
//
// Observed: the server's cron fetched at 17:00 ET, some symbols got that
// session's bar and some did not, and every one of them was stamped
// fresh until the next boundary. A client that had already fetched the
// newer bar would have had it overwritten and then believed the older
// one for a full day.
//
// Only candle data can be ordered this way, so only candle data is
// guarded; everything else falls through unchanged.
if (isCandleType(data_type)) {
if (serverBarRegression(&s, symbol, response.body)) |reg| {
// WARN, not debug. A shared cache should never be behind a
// client that draws from it: it is the tier with the refresh
// cron and the provider budget. When it happens, something on
// the server side has stopped keeping up, and every other client
// is being handed the same stale bar - so this is an operator
// event, not a diagnostic detail. Logged at debug, it was
// invisible in exactly the builds people install.
log.warn(
"{s}: shared cache is BEHIND this client for {s} - it offered {f}, local copy has {f}. Refused the sync rather than move the cache backwards; the server needs a refresh (see `zfin cache stale`).",
.{ symbol, @tagName(data_type), reg.incoming, reg.local },
);
log.debug("{s}: tryOneSync finished ({s}) result=ok elapsed_ms={d}", .{ symbol, @tagName(data_type), @divTrunc(std.Io.Timestamp.now(self.io, .awake).nanoseconds - t_start, std.time.ns_per_ms) });
return .ok;
}
}
s.writeRaw(symbol, data_type, response.body) catch |err| {
log.debug("{s}: failed to write synced {s} to cache: {s}", .{ symbol, @tagName(data_type), @errorName(err) });
log.debug("{s}: tryOneSync finished ({s}) result=net_err elapsed_ms={d}", .{ symbol, @tagName(data_type), @divTrunc(std.Io.Timestamp.now(self.io, .awake).nanoseconds - t_start, std.time.ns_per_ms) });
return .net_err;
};
log.debug("{s}: synced {s} from server ({d} bytes)", .{ symbol, @tagName(data_type), response.body.len });
log.debug("{s}: tryOneSync finished ({s}) result=ok elapsed_ms={d}", .{ symbol, @tagName(data_type), @divTrunc(std.Io.Timestamp.now(self.io, .awake).nanoseconds - t_start, std.time.ns_per_ms) });
return .ok;
}
/// Sync candle data (both daily and meta) from the server.
/// Do these bytes describe candle data, i.e. data with a newest-bar date
/// that can be compared for age?
fn isCandleType(data_type: cache.DataType) bool {
return data_type == .candles_daily or data_type == .candles_meta;
}
/// Both dates, when writing `body` would replace the local candle data with
/// an OLDER bar. Null otherwise.
///
/// Returns the pair rather than a bool because the caller has to report
/// them: "the shared cache is behind you" is only actionable with the two
/// dates attached.
///
/// Null whenever the question cannot be answered - no local copy, an
/// unparseable body, no dates on either side - so an unknown never blocks a
/// sync. The guard only fires on a definite regression.
fn serverBarRegression(
s: *cache.Store,
symbol: []const u8,
body: []const u8,
) ?struct { local: Date, incoming: Date } {
const local = s.readCandleMeta(symbol) orelse return null;
const incoming = newestDateIn(body) orelse return null;
if (!incoming.lessThan(local.meta.last_date)) return null;
return .{ .local = local.meta.last_date, .incoming = incoming };
}
/// Newest `last_date::` or `date::` value in an SRF body.
///
/// Deliberately a scan for the maximum rather than a parse: `candles_meta`
/// carries one `last_date`, `candles_daily` carries a `date` per bar, and
/// the ordering of the latter is not something this check should assume.
fn newestDateIn(body: []const u8) ?Date {
var best: ?Date = null;
for ([_][]const u8{ "last_date::", "date::" }) |key| {
var rest = body;
while (std.mem.indexOf(u8, rest, key)) |idx| {
const start = idx + key.len;
rest = rest[start..];
const end = std.mem.indexOfAny(u8, rest, ",\n\r") orelse rest.len;
if (Date.parse(rest[0..end])) |d| {
if (best == null or best.?.lessThan(d)) best = d;
} else |_| {}
}
}
return best;
}
/// TTL to stamp after a fetch that DID return candles.
///
/// A successful fetch is not necessarily a caught-up one. An incremental
/// request asks for everything after the last cached bar, so it can return
/// the bar missed in a PREVIOUS session while the current session's bar is
/// still unposted - a success by every measure the fetch itself can see.
///
/// Stamping the next-boundary TTL there writes the session off after a
/// partial catch-up, and for a symbol whose provider posts later than
/// `provider_lag_grace_s` that never converges: each session it recovers the
/// previous session's bar and is immediately a session behind again.
/// Observed on AMZN and NKE, which sat exactly one session behind for days
/// while the bar had been sitting at the provider the whole time -
///
/// Fri 18:36 zero bars, 99min past grace -> next boundary (Mon 16:55)
/// Mon 17:00 returned FRIDAY's bar -> next boundary (Tue 16:55) <- here
/// Tue 16:55 returns MONDAY's bar -> next boundary (Wed 16:55)
///
/// `staleCandleExpiry` asks the question the zero-bar branch already asks -
/// is the newest bar we now hold the one that should be available? - and
/// returns the next boundary when it is, so a genuinely caught-up fetch
/// behaves exactly as before.
///
/// TWO EARLIER DIAGNOSES OF THIS WERE WRONG, recorded so they are not
/// re-derived at the same cost:
///
/// - It is NOT `market.staleCandleExpiry` routing `.overdue` to the next
/// boundary. That path was never reached on the observed run:
/// `created=Mon 17:00` beside `expires=Tue 16:55` proves it, because five
/// minutes past the target is well inside `provider_lag_grace_s`, so the
/// verdict there was `.lagging`, not `.overdue`.
/// - It is NOT the 90-minute grace window being too short. It behaved
/// exactly as designed; the defect was upstream of it, in treating any
/// non-empty fetch result as proof of catching up.
///
/// Takes the maximum rather than the last element: provider ordering is not
/// something this decision should depend on.
fn expiryAfterFetch(now_s: i64, kind: market.InstrumentKind, candles: []const Candle) i64 {
if (candles.len == 0) return market.nextCandleExpiry(now_s, kind);
var newest = candles[0].date;
for (candles[1..]) |c| {
if (newest.lessThan(c.date)) newest = c.date;
}
return market.staleCandleExpiry(now_s, kind, newest);
}
fn syncCandlesFromServer(self: *DataService, symbol: []const u8) bool {
const daily = self.syncFromServer(symbol, .candles_daily);
const meta = self.syncFromServer(symbol, .candles_meta);
return daily and meta;
}
// ── User config files ─────────────────────────────────────────
/// Load and parse accounts.srf from the same directory as the given portfolio path.
/// Returns null if the file doesn't exist or can't be parsed.
/// Caller owns the returned AccountMap and must call deinit().
pub fn loadAccountMap(self: *DataService, allocator: std.mem.Allocator, portfolio_path: []const u8) ?analysis.AccountMap {
const dir_end = if (std.mem.lastIndexOfScalar(u8, portfolio_path, std.fs.path.sep)) |idx| idx + 1 else 0;
const acct_path = std.fmt.allocPrint(self.allocator, "{s}accounts.srf", .{portfolio_path[0..dir_end]}) catch return null;
defer self.allocator.free(acct_path);
const data = std.Io.Dir.cwd().readFileAlloc(self.io, acct_path, self.allocator, .limited(1024 * 1024)) catch return null;
defer self.allocator.free(data);
return analysis.parseAccountsFile(allocator, data) catch null;
}
/// Load and parse `transaction_log.srf` from the same directory as
/// the given portfolio path. Returns null if the file doesn't
/// exist or can't be parsed - the contributions pipeline falls
/// back to the pre-transaction-log behavior (no transfer netting)
/// when null is returned.
///
/// Caller owns the returned `TransactionLog` and must call
/// `deinit()`.
pub fn loadTransferLog(self: *DataService, portfolio_path: []const u8) ?transaction_log.TransactionLog {
const dir_end = if (std.mem.lastIndexOfScalar(u8, portfolio_path, std.fs.path.sep)) |idx| idx + 1 else 0;
const path = std.fmt.allocPrint(self.allocator, "{s}transaction_log.srf", .{portfolio_path[0..dir_end]}) catch return null;
defer self.allocator.free(path);
const data = std.Io.Dir.cwd().readFileAlloc(self.io, path, self.allocator, .limited(1024 * 1024)) catch return null;
defer self.allocator.free(data);
return transaction_log.parseTransactionLogFile(self.allocator, data) catch null;
}
/// Read the per-symbol `splits_current_through` opt-in from the
/// metadata.srf sibling of `portfolio_path` into a `symbol -> cutover`
/// map. A symbol absent from the map has NOT opted in - its lots keep
/// `split_factor` 1.0 (today's behavior). An empty map (no file, no
/// symbol carries the field) means split adjustment is fully off.
/// Caller owns the map AND each key: free every key, then `deinit`.
/// See the "Share model" block in models/portfolio.zig.
pub fn loadSplitsCutovers(self: *DataService, allocator: std.mem.Allocator, portfolio_path: []const u8) std.StringHashMap(Date) {
var map = std.StringHashMap(Date).init(allocator);
const dir_end = if (std.mem.lastIndexOfScalar(u8, portfolio_path, std.fs.path.sep)) |idx| idx + 1 else 0;
const path = std.fmt.allocPrint(self.allocator, "{s}metadata.srf", .{portfolio_path[0..dir_end]}) catch return map;
defer self.allocator.free(path);
const data = std.Io.Dir.cwd().readFileAlloc(self.io, path, self.allocator, .limited(1024 * 1024)) catch return map;
defer self.allocator.free(data);
var cm = classification.parseClassificationFile(self.allocator, data) catch return map;
defer cm.deinit();
for (cm.entries) |e| {
const cutover = e.splits_current_through orelse continue;
// A symbol can repeat across blended-fund rows; keep the first
// cutover and dupe its key exactly once (no leak on repeats).
const gop = map.getOrPut(e.symbol) catch continue;
if (!gop.found_existing) {
gop.key_ptr.* = allocator.dupe(u8, e.symbol) catch {
_ = map.remove(e.symbol);
continue;
};
gop.value_ptr.* = cutover;
}
}
return map;
}
/// Fetch split history for each symbol into a corpus keyed by
/// symbol, for lot effective-share enrichment (`enrichSplits`).
/// Symbols whose fetch fails or carry no splits are simply absent
/// (their lots keep `split_factor` 1.0 - a safe no-adjustment
/// fallback). Caller owns the returned map AND each value slice:
/// free every value, then `deinit` the map.
pub fn loadAllSplits(
self: *DataService,
allocator: std.mem.Allocator,
syms: []const []const u8,
opts: FetchOptions,
) std.StringHashMap([]const Split) {
var corpus = std.StringHashMap([]const Split).init(allocator);
for (syms) |sym| {
const fr = self.getSplits(sym, opts) catch continue;
defer fr.deinit();
if (fr.data.len == 0) continue;
const owned = allocator.dupe(Split, fr.data) catch continue;
corpus.put(sym, owned) catch {
allocator.free(owned);
continue;
};
}
return corpus;
}
/// Warm the dividend cache for each symbol. Warm-only: nothing is
/// returned, because the sole consumer (`PortfolioData`'s dividends
/// worker) reads the cache immediately afterward. That is the whole
/// difference from `loadAllSplits`, whose corpus feeds `enrichSplits`
/// inline.
///
/// Why this exists: dividends were the only per-symbol data type with
/// a portfolio-wide reader and no portfolio-wide writer. Candles are
/// warmed by `loadAllPrices` and splits by `loadAllSplits`, but nothing
/// warmed dividends, so `getCachedDividends` read a cache that only
/// per-symbol commands (`divs`, `perf`) had ever populated. The visible
/// symptom was `views/review.zig` silently reporting price-only
/// trailing returns for every symbol the user had never inspected
/// individually.
///
/// Sequential on purpose. The expensive phase is the provider, and
/// Polygon serves both dividends and splits from one 4/min bucket, so
/// concurrency cannot speed that up; the server-sync phase is a
/// sub-second serial cost for a normal portfolio. Rate limiting needs
/// no wiring here - it lives in the provider, so every `getDividends`
/// call is already throttled.
///
/// Failures are swallowed per symbol: a missing dividend history
/// degrades a total return to price-only, which is not worth failing
/// a whole portfolio load over. `fetchCached` logs the provider's own
/// error (rate limit vs auth vs no-such-data) before collapsing it to
/// `FetchFailed`, so a swallowed failure is still diagnosable.
pub fn loadAllDividends(
self: *DataService,
syms: []const []const u8,
opts: FetchOptions,
) void {
for (syms) |sym| {
// Per symbol, not once before the loop. Every cache miss here is a
// provider round trip under a 4-request/minute budget, so a batch
// whose TTLs happened to lapse together runs for minutes. The
// dividends worker is cancelled on TUI teardown
// (`PortfolioData.cancelLoad`), and cancelling waits for the
// worker - so without a check inside the loop, quitting blocked
// until the whole batch drained.
self.io.checkCancel() catch return;
const fr = self.getDividends(sym, opts) catch continue;
fr.deinit();
}
}
};
// ── Tests ─────────────────────────────────────────────────────────
test "serverAuthHeaders: key present yields one X-API-Key header" {
var buf: [1]std.http.Header = .{.{ .name = "", .value = "" }};
const h = DataService.serverAuthHeaders("s3cret", &buf);
try std.testing.expectEqual(@as(usize, 1), h.len);
try std.testing.expectEqualStrings("X-API-Key", h[0].name);
try std.testing.expectEqualStrings("s3cret", h[0].value);
}
test "serverAuthHeaders: null key yields no headers" {
var buf: [1]std.http.Header = .{.{ .name = "", .value = "" }};
const h = DataService.serverAuthHeaders(null, &buf);
try std.testing.expectEqual(@as(usize, 0), h.len);
}
test "isPermanentProviderFailure: NotFound is permanent" {
try std.testing.expect(isPermanentProviderFailure(error.NotFound));
}
test "isPermanentProviderFailure: RequestFailed is transient" {
try std.testing.expect(!isPermanentProviderFailure(error.RequestFailed));
}
test "isPermanentProviderFailure: ServerError is transient" {
try std.testing.expect(!isPermanentProviderFailure(error.ServerError));
}
test "isPermanentProviderFailure: Unauthorized is transient" {
// Auth misconfigs are user-fixable (set the API key); not a reason
// to permanently suppress retries.
try std.testing.expect(!isPermanentProviderFailure(error.Unauthorized));
}
test "isPermanentProviderFailure: InvalidResponse is transient" {
// Parse errors are usually a provider format change or one-off
// garbage response - retrying later is fine.
try std.testing.expect(!isPermanentProviderFailure(error.InvalidResponse));
}
test "isPermanentProviderFailure: PaymentRequired is transient" {
// FMP marks plan-locked symbols with HTTP 402; user can upgrade
// their plan or rotate providers, so don't poison the cache.
try std.testing.expect(!isPermanentProviderFailure(error.PaymentRequired));
}
test "isPermanentProviderFailure: RateLimited is transient" {
// Rate-limit is the textbook transient case; the caller already
// handles it specially with backoff + retry.
try std.testing.expect(!isPermanentProviderFailure(error.RateLimited));
}
test "DataService init/deinit lifecycle" {
const allocator = std.testing.allocator;
const config = Config{
.cache_dir = "/tmp/zfin-test-cache",
};
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
// Should be able to access config
try std.testing.expectEqualStrings("/tmp/zfin-test-cache", svc.config.cache_dir);
// Providers should be null (lazy init)
try std.testing.expect(svc.td == null);
try std.testing.expect(svc.pg == null);
try std.testing.expect(svc.fmp == null);
try std.testing.expect(svc.yh == null);
try std.testing.expect(svc.tg == null);
}
test "DataService live-stream accessors no-op when no stream is running" {
const allocator = std.testing.allocator;
const config = Config{
.cache_dir = "/tmp/zfin-test-cache",
};
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
// No stream created yet: active is false, snapshot leaves `out`
// empty, and stop is a safe no-op (must not crash on null).
try std.testing.expect(!svc.liveStreamActive());
var out = std.StringHashMap(f64).init(allocator);
defer out.deinit();
const syms = [_][]const u8{ "SPY", "AAPL" };
svc.liveStreamSnapshot(&syms, &out);
try std.testing.expectEqual(@as(usize, 0), out.count());
svc.stopLiveStream();
try std.testing.expect(!svc.liveStreamActive());
}
test "getProvider(Tiingo): hourly rate limit follows the configured plan" {
const allocator = std.testing.allocator;
// Free plan -> 50/hour token bucket.
{
var svc = DataService.init(std.testing.io, allocator, .{
.cache_dir = "/tmp/zfin-test-cache",
.tiingo_key = "test-key",
.tiingo_plan = .free,
});
defer svc.deinit();
const tg = try svc.getProvider(Tiingo);
try std.testing.expectEqual(@as(usize, 50), tg.rate_limiter.max_tokens);
}
// Power plan -> 10,000/hour token bucket.
{
var svc = DataService.init(std.testing.io, allocator, .{
.cache_dir = "/tmp/zfin-test-cache",
.tiingo_key = "test-key",
.tiingo_plan = .power,
});
defer svc.deinit();
const tg = try svc.getProvider(Tiingo);
try std.testing.expectEqual(@as(usize, 10_000), tg.rate_limiter.max_tokens);
}
}
test "getProvider(Tiingo): missing key -> NoApiKey" {
const allocator = std.testing.allocator;
var svc = DataService.init(std.testing.io, allocator, .{ .cache_dir = "/tmp/zfin-test-cache" });
defer svc.deinit();
try std.testing.expectError(DataError.NoApiKey, svc.getProvider(Tiingo));
}
test "getProvider: keyless/keyed dispatch and lazy caching" {
const allocator = std.testing.allocator;
var svc = DataService.init(std.testing.io, allocator, .{
.cache_dir = "/tmp/zfin-test-cache",
.twelvedata_key = "td-key",
});
defer svc.deinit();
// Keyless provider (Yahoo): no key needed; lazily created, then
// cached - a second call returns the same pointer.
const yh1 = try svc.getProvider(Yahoo);
const yh2 = try svc.getProvider(Yahoo);
try std.testing.expect(yh1 == yh2);
// Keyed provider via the generic branch (TwelveData, key present).
_ = try svc.getProvider(TwelveData);
// Keyed provider whose key is unset -> NoApiKey (Polygon).
try std.testing.expectError(DataError.NoApiKey, svc.getProvider(Polygon));
}
test "getProvider: open-data providers need a contact email" {
const allocator = std.testing.allocator;
// With an email, Wikidata/Edgar init via the email branch.
{
var svc = DataService.init(std.testing.io, allocator, .{
.cache_dir = "/tmp/zfin-test-cache",
.user_email = "test@example.com",
});
defer svc.deinit();
_ = try svc.getProvider(Wikidata);
_ = try svc.getProvider(Edgar);
}
// Without an email -> NoApiKey.
{
var svc = DataService.init(std.testing.io, allocator, .{ .cache_dir = "/tmp/zfin-test-cache" });
defer svc.deinit();
try std.testing.expectError(DataError.NoApiKey, svc.getProvider(Wikidata));
}
}
test "loadLiveQuotes: empty symbol list returns empty without touching the network" {
const allocator = std.testing.allocator;
// panic_on_network_attempt asserts the early return happens BEFORE
// any network gate is reached.
var svc = DataService.init(std.testing.io, allocator, .{ .cache_dir = "/tmp/zfin-test-cache" });
svc.panic_on_network_attempt = true;
defer svc.deinit();
var prices = svc.loadLiveQuotes(&.{});
defer prices.deinit();
try std.testing.expectEqual(@as(usize, 0), prices.count());
}
test "getCandles: negative candle cache short-circuits without touching the network" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, .{ .cache_dir = dir_path });
defer svc.deinit();
// Seed a negative candle entry: this symbol has no candle data on
// any provider (the crypto / defunct-ticker case). The marker lives
// in candles_daily.srf; candles_meta.srf is never created.
var s = svc.store();
try s.ensureSymbolDir("DOGE-USD");
s.writeNegative("DOGE-USD", .candles_daily);
// Any server sync or provider fetch would trip this panic guard.
svc.panic_on_network_attempt = true;
// getCandles must recognize the negative entry and fail fast, with
// no network round-trip (no panic).
try std.testing.expectError(DataError.FetchFailed, svc.getCandles("DOGE-USD", .{}));
}
test "DataService store helper creates valid store" {
const allocator = std.testing.allocator;
const config = Config{
.cache_dir = "/tmp/zfin-test-cache",
};
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
const s = svc.store();
try std.testing.expectEqualStrings("/tmp/zfin-test-cache", s.cache_dir);
}
test "DataService getProvider returns NoApiKey without key" {
const allocator = std.testing.allocator;
const config = Config{
.cache_dir = "/tmp/zfin-test-cache",
// No API keys set
};
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
// TwelveData requires API key
const td_result = svc.getProvider(TwelveData);
try std.testing.expectError(DataError.NoApiKey, td_result);
// Polygon requires API key
const pg_result = svc.getProvider(Polygon);
try std.testing.expectError(DataError.NoApiKey, pg_result);
// Yahoo doesn't require API key
const yh_result = svc.getProvider(Yahoo);
try std.testing.expect(yh_result != error.NoApiKey);
}
test "DataService getProvider initializes provider with key" {
const allocator = std.testing.allocator;
const config = Config{
.cache_dir = "/tmp/zfin-test-cache",
.tiingo_key = "test-tiingo-key",
};
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
// First call initializes
const tg1 = try svc.getProvider(Tiingo);
try std.testing.expect(svc.tg != null);
// Second call returns same instance
const tg2 = try svc.getProvider(Tiingo);
try std.testing.expect(tg1 == tg2);
}
test "DataService LoadAllResult default values" {
const allocator = std.testing.allocator;
var result = DataService.LoadAllResult{
.prices = std.StringHashMap(f64).init(allocator),
.cached_count = 0,
.server_synced_count = 0,
.provider_fetched_count = 0,
.stale_count = 0,
.failed_count = 0,
.latest_date = null,
};
defer result.deinit();
try std.testing.expectEqual(@as(usize, 0), result.prices.count());
}
test "FetchResult type construction" {
// Verify FetchResult works for different types
const candle_result = FetchResult(Candle){
.data = &.{},
.source = .cached,
.timestamp = 0,
.allocator = std.testing.allocator,
};
try std.testing.expect(candle_result.source == .cached);
const div_result = FetchResult(Dividend){
.data = &.{},
.source = .fetched,
.timestamp = 12345,
.allocator = std.testing.allocator,
};
try std.testing.expect(div_result.source == .fetched);
try std.testing.expectEqual(@as(i64, 12345), div_result.timestamp);
}
test "FetchOptions default is fully permissive" {
// Default-init should allow normal fetch behavior.
const opts: FetchOptions = .{};
try std.testing.expect(!opts.skip_network);
try std.testing.expect(!opts.force_refresh);
}
test "getCandles offline mode returns cached data without network" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
// Construct a service with a cache pre-populated with candle data.
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
// Pre-populate cache via the Store API.
var store = svc.store();
var candles = [_]Candle{
.{ .date = Date.fromYmd(2026, 5, 19), .open = 100, .high = 105, .low = 99, .close = 104, .adj_close = 104, .volume = 1000 },
.{ .date = Date.fromYmd(2026, 5, 20), .open = 104, .high = 106, .low = 103, .close = 105, .adj_close = 105, .volume = 1100 },
};
store.cacheCandles("TEST", candles[0..], .{ .provider = .tiingo }, market.nextCandleExpiry(std.Io.Timestamp.now(io, .real).toSeconds(), .equity));
// Set the test guard: any network call would panic. We expect
// the offline-mode path NOT to touch the network.
svc.panic_on_network_attempt = true;
const result = try svc.getCandles("TEST", .{ .skip_network = true });
defer result.deinit();
try std.testing.expectEqual(@as(usize, 2), result.data.len);
try std.testing.expect(result.data[0].date.eql(Date.fromYmd(2026, 5, 19)));
try std.testing.expect(result.data[1].date.eql(Date.fromYmd(2026, 5, 20)));
try std.testing.expectEqual(Source.cached, result.source);
}
test "getCandles offline mode with no cache returns FetchFailed" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
// Network guard is on. With no cache and skip_network=true,
// we must return FetchFailed without panicking.
svc.panic_on_network_attempt = true;
const err = svc.getCandles("NEVERHEARDOFIT", .{ .skip_network = true });
try std.testing.expectError(DataError.FetchFailed, err);
}
test "fetchCached offline mode returns stale-cached data" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
// Pre-populate dividend cache with a TTL in the past (stale).
var store = svc.store();
var divs = [_]Dividend{
.{ .ex_date = Date.fromYmd(2026, 3, 15), .amount = 0.50, .type = .regular },
};
// Manually set TTL to 1 second (long since expired) by writing
// through writeWithSource with a tiny TTL.
store.writeWithSource(Dividend, "TEST", divs[0..], .{ .seconds = -1_000_000 }, "test");
svc.panic_on_network_attempt = true;
// Even though the cache is stale, skip_network must return it
// rather than touching the network.
const result = try svc.getDividends("TEST", .{ .skip_network = true });
defer result.deinit();
try std.testing.expectEqual(@as(usize, 1), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
}
// ── Schedule-aware dividend refresh ──────────────────────────────
//
// `dividendsNeedRefresh` and its two helpers. The suite is deliberately
// heavy on real-corpus regressions: the predicate replaced a TTL that
// looked adequate on a mid-quarter snapshot and was in fact missing
// roughly half the portfolio's payers across two consecutive weekly
// runs, so the fixtures below are the actual cached ex-dates that
// produced that failure. Symbol identities are public tickers and fund
// distribution schedules - no account data is involved.
/// Build a dividend slice from ex-dates given as `.{ y, m, d }`.
fn divsFromYmd(comptime dates: anytype) [dates.len]Dividend {
// SAFETY: every element is assigned in the loop below.
var out: [dates.len]Dividend = undefined;
inline for (dates, 0..) |d, i| {
out[i] = .{ .ex_date = Date.fromYmd(d[0], d[1], d[2]), .amount = 1.0, .type = .regular };
}
return out;
}
/// A quarterly series of `n` ex-dates ending on `newest`, newest-first.
fn quarterlySeries(comptime n: usize, newest: Date, cadence: i32) [n]Dividend {
// SAFETY: every element is assigned in the loop below.
var out: [n]Dividend = undefined;
for (0..n) |i| {
out[i] = .{
.ex_date = newest.addDays(-cadence * @as(i32, @intCast(i))),
.amount = 1.0,
.type = .regular,
};
}
return out;
}
test "newestExDates: empty input writes nothing" {
var out: [4]Date = @splat(Date.epoch);
try std.testing.expectEqual(@as(usize, 0), DataService.newestExDates(&.{}, &out));
}
test "newestExDates: fewer records than the buffer yields them all, descending" {
const divs = divsFromYmd(.{ .{ 2026, 3, 17 }, .{ 2026, 6, 15 }, .{ 2025, 12, 16 } });
var out: [6]Date = @splat(Date.epoch);
try std.testing.expectEqual(@as(usize, 3), DataService.newestExDates(&divs, &out));
try std.testing.expect(out[0].eql(Date.fromYmd(2026, 6, 15)));
try std.testing.expect(out[1].eql(Date.fromYmd(2026, 3, 17)));
try std.testing.expect(out[2].eql(Date.fromYmd(2025, 12, 16)));
}
test "newestExDates: more records than the buffer keeps only the newest" {
// Six years of quarterly history, buffer of three. The oldest
// entries must not displace a newer one, and the newest must not be
// lost to an early-full buffer.
const divs = quarterlySeries(24, Date.fromYmd(2026, 6, 15), 91);
var out: [3]Date = @splat(Date.epoch);
try std.testing.expectEqual(@as(usize, 3), DataService.newestExDates(&divs, &out));
try std.testing.expect(out[0].eql(Date.fromYmd(2026, 6, 15)));
try std.testing.expect(out[1].eql(Date.fromYmd(2026, 6, 15).addDays(-91)));
try std.testing.expect(out[2].eql(Date.fromYmd(2026, 6, 15).addDays(-182)));
}
test "newestExDates: input order does not matter" {
// The cache file happens to be newest-first, but `writeSupplement`
// merges two providers' records and nothing in SRF promises an
// order. Ascending and shuffled inputs must agree with descending.
const descending = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 3, 17 }, .{ 2025, 12, 16 }, .{ 2025, 9, 16 } });
const ascending = divsFromYmd(.{ .{ 2025, 9, 16 }, .{ 2025, 12, 16 }, .{ 2026, 3, 17 }, .{ 2026, 6, 15 } });
const shuffled = divsFromYmd(.{ .{ 2025, 12, 16 }, .{ 2026, 6, 15 }, .{ 2025, 9, 16 }, .{ 2026, 3, 17 } });
var a: [4]Date = @splat(Date.epoch);
var b: [4]Date = @splat(Date.epoch);
var c: [4]Date = @splat(Date.epoch);
try std.testing.expectEqual(@as(usize, 4), DataService.newestExDates(&descending, &a));
try std.testing.expectEqual(@as(usize, 4), DataService.newestExDates(&ascending, &b));
try std.testing.expectEqual(@as(usize, 4), DataService.newestExDates(&shuffled, &c));
for (a, b, c) |x, y, z| {
try std.testing.expect(x.eql(y));
try std.testing.expect(x.eql(z));
}
}
test "newestExDates: duplicate ex-dates are both retained" {
// A regular and a special on the same day. Silently collapsing them
// would understate the record count and could flip the
// minimum-records gate.
const divs = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 6, 15 }, .{ 2026, 3, 17 } });
var out: [6]Date = @splat(Date.epoch);
try std.testing.expectEqual(@as(usize, 3), DataService.newestExDates(&divs, &out));
try std.testing.expect(out[0].eql(Date.fromYmd(2026, 6, 15)));
try std.testing.expect(out[1].eql(Date.fromYmd(2026, 6, 15)));
try std.testing.expect(out[2].eql(Date.fromYmd(2026, 3, 17)));
}
test "newestExDates: a one-slot buffer yields the single newest" {
const divs = divsFromYmd(.{ .{ 2025, 12, 16 }, .{ 2026, 6, 15 }, .{ 2026, 3, 17 } });
var out: [1]Date = @splat(Date.epoch);
try std.testing.expectEqual(@as(usize, 1), DataService.newestExDates(&divs, &out));
try std.testing.expect(out[0].eql(Date.fromYmd(2026, 6, 15)));
}
test "dividendCadenceDays: below the minimum record count there is no cadence" {
// One gap cannot disagree with anything, so a median over it is a
// guess. Declining hands the symbol to the TTL, which is the safe
// direction for a newly-bought holding.
const none: []const Dividend = &.{};
try std.testing.expect(DataService.dividendCadenceDays(none) == null);
const one = divsFromYmd(.{.{ 2026, 6, 15 }});
try std.testing.expect(DataService.dividendCadenceDays(&one) == null);
const two = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 3, 17 } });
try std.testing.expect(DataService.dividendCadenceDays(&two) == null);
// Three is the first count that yields one.
const three = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 3, 17 }, .{ 2025, 12, 16 } });
try std.testing.expect(DataService.dividendCadenceDays(&three) != null);
}
test "dividendCadenceDays: recognises quarterly, monthly, semi-annual and annual" {
// The four cadences present in a real portfolio. An annual payer
// matters as much as a quarterly one: getting its cadence wrong
// would fire the predicate for eleven months of the year.
const q = quarterlySeries(6, Date.fromYmd(2026, 6, 15), 91);
try std.testing.expectEqual(@as(i32, 91), DataService.dividendCadenceDays(&q).?);
const m = quarterlySeries(6, Date.fromYmd(2026, 9, 1), 30);
try std.testing.expectEqual(@as(i32, 30), DataService.dividendCadenceDays(&m).?);
const semi = quarterlySeries(4, Date.fromYmd(2026, 6, 15), 181);
try std.testing.expectEqual(@as(i32, 181), DataService.dividendCadenceDays(&semi).?);
const annual = quarterlySeries(3, Date.fromYmd(2025, 12, 12), 364);
try std.testing.expectEqual(@as(i32, 364), DataService.dividendCadenceDays(&annual).?);
}
test "dividendCadenceDays: median absorbs a single outlier gap" {
// The case that rules out both the mean and the minimum. A special
// distribution three days after a regular one creates one tiny gap.
// The minimum would read the cadence as 3 and fire for the whole
// chase window after every payment; the mean would be dragged low.
// Gaps here are 3, 91, 91, 91, 91 -> median 91.
var divs = [_]Dividend{
.{ .ex_date = Date.fromYmd(2026, 6, 18), .amount = 2.0, .type = .special },
.{ .ex_date = Date.fromYmd(2026, 6, 15), .amount = 1.0, .type = .regular },
.{ .ex_date = Date.fromYmd(2026, 3, 16), .amount = 1.0, .type = .regular },
.{ .ex_date = Date.fromYmd(2025, 12, 15), .amount = 1.0, .type = .regular },
.{ .ex_date = Date.fromYmd(2025, 9, 15), .amount = 1.0, .type = .regular },
.{ .ex_date = Date.fromYmd(2025, 6, 16), .amount = 1.0, .type = .regular },
};
try std.testing.expectEqual(@as(i32, 91), DataService.dividendCadenceDays(divs[0..]).?);
}
test "dividendCadenceDays: an even gap count averages the two middles" {
// Four records -> three gaps (odd, true middle). Five -> four gaps
// (even, floored average). Both branches exercised explicitly
// because an off-by-one in the median index is silent.
const odd_gaps = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 3, 16 }, .{ 2025, 12, 15 }, .{ 2025, 9, 15 } });
// Gaps: 91, 91, 91.
try std.testing.expectEqual(@as(i32, 91), DataService.dividendCadenceDays(&odd_gaps).?);
// Gaps 10, 20, 30, 40 -> sorted the two middles are 20 and 30 ->
// floored average 25.
var uneven = [_]Dividend{
.{ .ex_date = Date.epoch.addDays(100), .amount = 1.0 },
.{ .ex_date = Date.epoch.addDays(90), .amount = 1.0 },
.{ .ex_date = Date.epoch.addDays(70), .amount = 1.0 },
.{ .ex_date = Date.epoch.addDays(40), .amount = 1.0 },
.{ .ex_date = Date.epoch, .amount = 1.0 },
};
try std.testing.expectEqual(@as(i32, 25), DataService.dividendCadenceDays(uneven[0..]).?);
}
test "dividendCadenceDays: all-duplicate ex-dates yield no cadence, not a zero" {
// A zero cadence is not a schedule, and it would divide by zero in
// `dividendsNeedRefresh`. This is the guard for that.
const divs = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 6, 15 }, .{ 2026, 6, 15 }, .{ 2026, 6, 15 } });
try std.testing.expect(DataService.dividendCadenceDays(&divs) == null);
}
test "dividendCadenceDays: only the newest gaps count, so old history cannot drag it" {
// Monthly for the last six records, quarterly before that. The
// window must read 30, not something between.
var divs = [_]Dividend{
.{ .ex_date = Date.fromYmd(2026, 9, 1), .amount = 1.0 },
.{ .ex_date = Date.fromYmd(2026, 8, 2), .amount = 1.0 },
.{ .ex_date = Date.fromYmd(2026, 7, 3), .amount = 1.0 },
.{ .ex_date = Date.fromYmd(2026, 6, 3), .amount = 1.0 },
.{ .ex_date = Date.fromYmd(2026, 5, 4), .amount = 1.0 },
.{ .ex_date = Date.fromYmd(2026, 4, 4), .amount = 1.0 },
// Quarterly history further back - outside the gap window.
.{ .ex_date = Date.fromYmd(2026, 1, 3), .amount = 1.0 },
.{ .ex_date = Date.fromYmd(2025, 10, 4), .amount = 1.0 },
.{ .ex_date = Date.fromYmd(2025, 7, 5), .amount = 1.0 },
};
try std.testing.expectEqual(@as(i32, 30), DataService.dividendCadenceDays(divs[0..]).?);
}
test "dividendsNeedRefresh: declines when there is no cadence to be late against" {
const today = Date.fromYmd(2026, 9, 19);
const none: []const Dividend = &.{};
try std.testing.expect(!DataService.dividendsNeedRefresh(none, today));
// Two records: a gap exists but no median does.
const two = divsFromYmd(.{ .{ 2026, 3, 17 }, .{ 2025, 12, 16 } });
try std.testing.expect(!DataService.dividendsNeedRefresh(&two, today));
// Duplicates: three records, zero cadence.
const dup = divsFromYmd(.{ .{ 2026, 3, 17 }, .{ 2026, 3, 17 }, .{ 2026, 3, 17 } });
try std.testing.expect(!DataService.dividendsNeedRefresh(&dup, today));
}
test "dividendsNeedRefresh: a full cadence period must elapse before it fires" {
// The property that stops it refiring immediately after a
// successful fetch, when the record it just stored is days old.
const newest = Date.fromYmd(2026, 6, 15);
const divs = quarterlySeries(6, newest, 91);
// Day of, mid-period, and the last day before due: all quiet.
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, newest));
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, newest.addDays(45)));
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, newest.addDays(90)));
// Exactly one cadence period: due.
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, newest.addDays(91)));
}
test "dividendsNeedRefresh: the chase window has both bounds" {
const newest = Date.fromYmd(2026, 6, 15);
const divs = quarterlySeries(6, newest, 91);
const chase = DataService.dividend_chase_days;
// Last day inside the window fires; the next day does not. Past it
// the sponsor is off-schedule and the TTL takes over - an unbounded
// chase would refetch forever on a wrong cadence estimate.
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, newest.addDays(91 + chase)));
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, newest.addDays(91 + chase + 1)));
}
test "dividendsNeedRefresh: rolls forward so a long-neglected symbol stays recoverable" {
// Anchoring the chase window on the FIRST expected date after the
// newest cached record means a symbol two periods behind is past its
// window forever and the predicate can never recover it. Taking the
// latest expected date at or before today fixes that.
const newest = Date.fromYmd(2026, 3, 17);
const divs = quarterlySeries(6, newest, 91);
// Two periods on: due again, and within the second window.
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, newest.addDays(182)));
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, newest.addDays(182 + 5)));
// Eight periods on, just after that period's expected date.
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, newest.addDays(91 * 8 + 2)));
}
test "dividendsNeedRefresh: quiet in the dead zone between two expected dates" {
// The other half of rolling forward. Between periods there is
// nothing outstanding, and firing there would spend a request per
// run for no possible gain.
const newest = Date.fromYmd(2026, 3, 17);
const divs = quarterlySeries(6, newest, 91);
const chase = DataService.dividend_chase_days;
// One period + past the window, but before the second period.
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, newest.addDays(91 + chase + 1)));
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, newest.addDays(150)));
// ...and it wakes up again at the second period.
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, newest.addDays(182)));
}
test "dividendsNeedRefresh: a forward-declared ex-date is not overdue" {
// NKE's shape - it declares roughly a quarter ahead, so its newest
// cached ex-date is normally in the future. A negative elapsed must
// read as "nothing outstanding", not wrap into a fire.
const divs = divsFromYmd(.{ .{ 2026, 12, 1 }, .{ 2026, 9, 1 }, .{ 2026, 6, 1 }, .{ 2026, 3, 2 } });
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, Date.fromYmd(2026, 9, 19)));
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, Date.fromYmd(2026, 10, 1)));
}
test "dividendsNeedRefresh: regression - the real corpus that a 14-day TTL missed" {
// Cached ex-dates as they actually stood on 2026-09-19, when four
// symbols had a distribution that had already paid and a fresh cache
// that did not contain it. Public tickers and fund schedules only.
const today = Date.fromYmd(2026, 9, 19);
// Overdue: newest cached ex-date is a full quarter old and the
// September distribution is absent.
const soxx = quarterlySeries(6, Date.fromYmd(2026, 6, 15), 91);
const fdvv = quarterlySeries(6, Date.fromYmd(2026, 6, 18), 91);
const hfxi = quarterlySeries(6, Date.fromYmd(2026, 6, 18), 91);
const spym = quarterlySeries(6, Date.fromYmd(2026, 6, 12), 91);
try std.testing.expect(DataService.dividendsNeedRefresh(&soxx, today));
try std.testing.expect(DataService.dividendsNeedRefresh(&fdvv, today));
try std.testing.expect(DataService.dividendsNeedRefresh(&hfxi, today));
try std.testing.expect(DataService.dividendsNeedRefresh(&spym, today));
// Not yet due on that date - their ex-dates were still days away.
// Firing on these would be the false-positive cost of the mechanism,
// and there is none.
const rsp = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const xmmo = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const xlv = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const idmo = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const qqq = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const qtum = quarterlySeries(6, Date.fromYmd(2026, 6, 24), 91);
const schd = quarterlySeries(6, Date.fromYmd(2026, 6, 24), 91);
const frdm = quarterlySeries(6, Date.fromYmd(2026, 6, 29), 91);
const ivlu = quarterlySeries(4, Date.fromYmd(2026, 6, 15), 181);
const sphy = quarterlySeries(6, Date.fromYmd(2026, 9, 1), 30);
const nvda = quarterlySeries(6, Date.fromYmd(2026, 9, 10), 92);
const fdscx = quarterlySeries(3, Date.fromYmd(2025, 12, 12), 364);
for ([_][]const Dividend{ &rsp, &xmmo, &xlv, &idmo, &qqq, &qtum, &schd, &frdm, &ivlu, &sphy, &nvda, &fdscx }) |corpus| {
try std.testing.expect(!DataService.dividendsNeedRefresh(corpus, today));
}
}
test "dividendsNeedRefresh: regression - the quarter-end cluster one week on" {
// The same corpus at the following weekly run. Six funds went ex
// 2026-09-21..09-23 and paid within days; their caches were all
// written in the same mid-September pass and so all expired well
// after the payments. This is the cluster the TTL cannot see.
const next_week = Date.fromYmd(2026, 9, 26);
const rsp = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const xmmo = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const xlv = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const idmo = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const qqq = quarterlySeries(6, Date.fromYmd(2026, 6, 22), 91);
const qtum = quarterlySeries(6, Date.fromYmd(2026, 6, 24), 91);
const schd = quarterlySeries(6, Date.fromYmd(2026, 6, 24), 91);
for ([_][]const Dividend{ &rsp, &xmmo, &xlv, &idmo, &qqq, &qtum, &schd }) |corpus| {
try std.testing.expect(DataService.dividendsNeedRefresh(corpus, next_week));
}
// A monthly payer that has not reached its next ex-date stays quiet
// even in the same run.
const sphy = quarterlySeries(6, Date.fromYmd(2026, 9, 1), 30);
try std.testing.expect(!DataService.dividendsNeedRefresh(&sphy, next_week));
}
test "dividendsNeedRefresh: once the overdue record lands, it goes quiet for a period" {
// Convergence. The whole point is to stop asking as soon as the
// answer arrives - otherwise the mechanism costs a request per run
// forever.
const today = Date.fromYmd(2026, 9, 19);
const before = quarterlySeries(6, Date.fromYmd(2026, 6, 15), 91);
try std.testing.expect(DataService.dividendsNeedRefresh(&before, today));
// The refetch stores the 2026-09-15 record.
var after = [_]Dividend{
.{ .ex_date = Date.fromYmd(2026, 9, 15), .amount = 0.325046, .type = .regular },
} ++ before;
try std.testing.expect(!DataService.dividendsNeedRefresh(after[0..], today));
// ...and stays quiet until the next period comes due.
try std.testing.expect(!DataService.dividendsNeedRefresh(after[0..], Date.fromYmd(2026, 12, 1)));
try std.testing.expect(DataService.dividendsNeedRefresh(after[0..], Date.fromYmd(2026, 12, 15)));
}
test "dividendsNeedRefresh: a monthly payer is due a month on, not a quarter on" {
// SPHY's shape. A cadence read as quarterly would leave two monthly
// distributions unseen, so this pins the short-cadence arithmetic
// rather than trusting the quarterly cases to cover it.
const newest = Date.fromYmd(2026, 9, 1);
const divs = quarterlySeries(6, newest, 30);
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, newest.addDays(29)));
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, newest.addDays(30)));
// Two months on, rolled forward.
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, newest.addDays(60)));
}
test "dividendsNeedRefresh: KNOWN LIMITATION - a cadence change forecasts late" {
// Documented, not fixed. A fund that moves quarterly to monthly
// keeps a ~91-day median for several periods, so the predicate does
// not fire until a quarter has passed and the intervening monthly
// distributions are invisible to it. `Ttl.dividends` is what covers
// this, and the trade is deliberate: a more eager statistic (the
// minimum gap) would spend requests on every symbol, every period,
// to protect against a rare event on one.
//
// This test exists so that changing the statistic shows up as a
// deliberate diff here rather than as a silent behaviour change.
const divs = quarterlySeries(6, Date.fromYmd(2026, 6, 15), 91);
// A monthly schedule would have gone ex on 07-15 and 08-15. The
// predicate is silent through both.
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, Date.fromYmd(2026, 7, 20)));
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, Date.fromYmd(2026, 8, 20)));
// It only wakes at the quarterly boundary.
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, Date.fromYmd(2026, 9, 14)));
}
// ── The refresh hook inside fetchCached ──────────────────────────
//
// These go through the real cache and the real `fetchCached`, so they
// also pin the free-on-discard: `std.testing.allocator` fails the test
// if the discarded entry leaks, and a double free would trip its
// bookkeeping too.
//
// Fixtures are built relative to the actual current day because the hook
// reads the wall clock (see `serveFreshOrDiscard`). Hardcoded dates would
// make these pass or fail depending on when they run.
/// A quarterly dividend corpus positioned relative to `today`.
/// `age_days` is how old the newest cached ex-date is, so `age_days <
/// 91` is "not due" and `>= 91` is "overdue".
fn corpusAged(today: Date, age_days: i32) [6]Dividend {
return quarterlySeries(6, today.addDays(-age_days), 91);
}
/// Write a cache entry as though the write had happened `age_s` ago.
///
/// Needed because `askedRecently` recovers the write time from
/// `#!expires=` minus the TTL that was applied. A freshly written entry
/// therefore always looks like "we just asked", which suppresses the
/// refresh hook - correct in production, useless in a test that wants to
/// exercise the hook. Back-dating the expires by the TTL is the only way
/// to make a test entry look old without sleeping.
///
/// `spec` must be the TTL the production path would have used for this
/// data, since that is what `askedRecently` will recompute.
fn writeAged(
comptime T: type,
store: *cache.Store,
symbol: []const u8,
items: cache.Store.DataFor(T),
spec: cache.TtlSpec,
age_s: i64,
) void {
// What the production write would have stamped, jitter included.
const applied = cache.computeExpires(0, spec, symbol);
// Exact expires, no jitter of our own, so the recompute lines up.
store.writeWithSource(T, symbol, items, .{ .seconds = applied - age_s }, null);
}
test "fetchCached hook: a null hook leaves the fresh-cache path untouched" {
// The regression guard for every other type. Splits pass no hook and
// have no `freeSlice`; if the hook branch were not comptime-elided
// this would not compile, and if it were consulted anyway this would
// panic on the network assertion.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
var store = svc.store();
var splits = [_]Split{.{ .date = Date.fromYmd(2024, 3, 7), .numerator = 3, .denominator = 1 }};
store.write(Split, "TEST", splits[0..], cache.DataType.splits.ttl());
svc.panic_on_network_attempt = true;
const result = try svc.getSplits("TEST", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 1), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
}
test "fetchCached hook: a fresh, complete dividend cache is still served" {
// Nothing outstanding, so the hook must not discard a usable entry.
// `panic_on_network_attempt` is the assertion: reaching the provider
// at all would be a spurious refetch on every symbol, every run.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var divs = corpusAged(today, 10);
var store = svc.store();
store.write(Dividend, "TEST", divs[0..], cache.DataType.dividends.ttl());
svc.panic_on_network_attempt = true;
const result = try svc.getDividends("TEST", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 6), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
}
test "fetchCached hook: skip_network suppresses it rather than discarding for nothing" {
// Offline mode never refetches, so consulting the hook could only
// turn a served result into a failure. Same rule `getEarnings`
// applies. The corpus here IS overdue - without the suppression this
// would discard the entry and then fail.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var divs = corpusAged(today, 95);
try std.testing.expect(DataService.dividendsNeedRefresh(divs[0..], today));
var store = svc.store();
store.write(Dividend, "TEST", divs[0..], cache.DataType.dividends.ttl());
svc.panic_on_network_attempt = true;
const result = try svc.getDividends("TEST", .{ .skip_network = true });
defer result.deinit();
try std.testing.expectEqual(@as(usize, 6), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
}
test "fetchCached hook: an overdue distribution discards the fresh entry and refetches" {
// The behaviour the whole change exists for. With no Polygon key the
// provider call fails before any network I/O, so FetchFailed is the
// observable proof that the fresh cache was NOT served. If the hook
// did not fire, this would return six cached records instead.
//
// It also pins the free: the discarded slice was allocated by the
// cache read, and `std.testing.allocator` fails the test if it leaks
// or is freed twice.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var divs = corpusAged(today, 95);
var store = svc.store();
// Aged two days: a just-written entry looks like "we already asked"
// to the recheck floor, which would suppress the refresh. The spec
// must match what production would have stamped (`dividendTtl`), since
// that is what the floor recomputes.
writeAged(Dividend, &store, "TEST", divs[0..], DataService.dividendTtl(divs[0..]), 2 * std.time.s_per_day);
// Sanity: the entry really is fresh, so TTL alone would have served
// it. That is the failure this replaces.
{
const fresh = svc.getCachedDividends(allocator, "TEST") orelse return error.TestUnexpectedResult;
defer fresh.deinit();
try std.testing.expectEqual(@as(usize, 6), fresh.data.len);
}
// NoApiKey, not FetchFailed: reaching the provider at all is the
// proof that the fresh cache was discarded, and the specific error
// is what tells us we got as far as key resolution.
try std.testing.expectError(DataError.NoApiKey, svc.getDividends("TEST", .{}));
// A transient failure must not poison the cache - the entry is still
// there for the next run, and no negative entry was written.
const after = svc.getCachedDividends(allocator, "TEST") orelse return error.TestUnexpectedResult;
defer after.deinit();
try std.testing.expectEqual(@as(usize, 6), after.data.len);
}
test "fetchCached hook: force_refresh never reaches it" {
// force_refresh bypasses the fresh-cache read entirely, so the hook
// is not consulted and cannot double-free the entry the caller
// already skipped.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
// Deliberately NOT due, so any refetch can only be force_refresh's.
var divs = corpusAged(today, 10);
var store = svc.store();
store.write(Dividend, "TEST", divs[0..], cache.DataType.dividends.ttl());
try std.testing.expectError(DataError.NoApiKey, svc.getDividends("TEST", .{ .force_refresh = true }));
}
test "fetchCached hook: a symbol with too little history is left to the TTL" {
// A newly-bought holding has no cadence, so the hook declines and the
// fresh entry is served. Reaching the provider here would mean every
// new position refetched on every run.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
// Two records, one gap - and deliberately ancient, so the only
// reason not to refetch is the missing cadence.
var divs = [_]Dividend{
.{ .ex_date = today.addDays(-400), .amount = 1.0, .type = .regular },
.{ .ex_date = today.addDays(-491), .amount = 1.0, .type = .regular },
};
var store = svc.store();
store.write(Dividend, "TEST", divs[0..], cache.DataType.dividends.ttl());
svc.panic_on_network_attempt = true;
const result = try svc.getDividends("TEST", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 2), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
}
test "fetchCached hook: getOptions passes no hook and is unaffected" {
// The third type routed through `fetchCached`, and a comptime
// combination neither of the others covers: OptionsChain HAS a
// `freeSlice` but passes no hook, where Split has a hook-less type
// with NO `freeSlice`. Both must reach the plain fresh-cache return.
//
// A negative cache entry is the cheapest always-fresh options
// fixture - `Store.read` returns an empty slice for one under
// `.fresh_only` - and the assertion is that we get there without
// touching the network.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
var store = svc.store();
store.writeNegative("TEST", .options);
svc.panic_on_network_attempt = true;
const result = try svc.getOptions("TEST", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 0), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
}
test "fetchCached: a missing API key propagates as NoApiKey, not FetchFailed" {
// Errors carry information. `commands/earnings.zig` and
// `tui/earnings_tab.zig` both switch on NoApiKey specifically so they
// can name the environment variable to set - "Error fetching earnings
// data" sends the user to read source code instead. Collapsing it
// into FetchFailed erases the one distinction any caller acts on.
//
// Also: a missing key must NOT negative-cache. Nothing is wrong with
// the symbol, and poisoning it would suppress the fetch after the key
// is finally configured.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
// Cold cache, no keys configured, for each type routed through
// `fetchCached` that needs one.
try std.testing.expectError(DataError.NoApiKey, svc.getDividends("TEST", .{}));
try std.testing.expectError(DataError.NoApiKey, svc.getSplits("TEST", .{}));
try std.testing.expectError(DataError.NoApiKey, svc.getEarnings("TEST", .{}));
// No negative entry was written for any of them.
try std.testing.expect(svc.getCachedDividends(allocator, "TEST") == null);
}
test "getEarnings: the mutual-fund skip short-circuits before any cache or network work" {
// The one earnings-specific branch that could not become a hook. A
// fund has no quarterly earnings at all, so this must return empty
// without reading the cache, syncing, or resolving a provider - and
// in particular without the NoApiKey that a real fetch would raise
// here, since no FMP key is configured.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
svc.panic_on_network_attempt = true;
const result = try svc.getEarnings("VBTLX", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 0), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
// A non-fund ticker does NOT take the skip: it reaches the generic
// path, where a cold cache under skip_network is FetchFailed.
try std.testing.expectError(DataError.FetchFailed, svc.getEarnings("TESTA", .{ .skip_network = true }));
}
test "getEarnings: smart refresh survives the move onto fetchCached" {
// The behaviour the refactor had to preserve. A fresh cache holding a
// past report with no actual must re-fetch; the same cache with the
// actual filled in must be served. Before the refactor this logic sat
// in a hand-rolled copy of `fetchCached`; now it is the generic hook,
// and this is what proves the wiring is real.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var store = svc.store();
// A report three days ago with no actual yet -> inside the chase
// window -> must refetch, which with no FMP key surfaces as NoApiKey.
//
// Aged deliberately: a just-written entry looks like "we already
// asked" to the recheck floor, which would suppress the very refresh
// this test is about.
var pending = [_]EarningsEvent{
.{ .symbol = "TEST", .date = today.addDays(-3), .estimate = 1.5 },
};
writeAged(EarningsEvent, &store, "TEST", pending[0..], cache.DataType.earnings.ttl(), 2 * std.time.s_per_day);
try std.testing.expectError(DataError.NoApiKey, svc.getEarnings("TEST", .{}));
// Same report with the actual posted -> nothing outstanding -> served
// from cache without touching the provider.
var settled = [_]EarningsEvent{
.{ .symbol = "TEST", .date = today.addDays(-3), .estimate = 1.5, .actual = 1.62 },
};
store.write(EarningsEvent, "TEST", settled[0..], cache.DataType.earnings.ttl());
svc.panic_on_network_attempt = true;
const result = try svc.getEarnings("TEST", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 1), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
// `earningsPostProcess` still runs through the generic path: surprise
// is derived, not stored.
try std.testing.expect(result.data[0].surprise != null);
try std.testing.expectApproxEqAbs(@as(f64, 0.12), result.data[0].surprise.?, 1e-9);
}
// ── The recheck floor, and the TTL that varies by schedule ────────
test "recheck floor: a just-asked symbol is not asked again" {
// The bug this fixes. When the schedule says a payment is due but the
// provider has nothing yet, the cache is rewritten with a fresh clock
// and the SAME records - so the schedule still says "due", and without
// a floor the next command asks again. Here the entry is written and
// immediately read back, which is exactly that situation.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var divs = corpusAged(today, 95);
// Confirm the schedule really does say "overdue", so the only thing
// that can stop the refetch is the floor.
try std.testing.expect(DataService.dividendsNeedRefresh(divs[0..], today));
var store = svc.store();
store.write(Dividend, "TEST", divs[0..], DataService.dividendTtl(divs[0..]));
// Reaching the provider would panic. Being served proves the floor held.
svc.panic_on_network_attempt = true;
const result = try svc.getDividends("TEST", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 6), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
}
test "recheck floor: once the interval has passed, the symbol is asked again" {
// The other side of the floor. Same overdue corpus, but the entry was
// written longer ago than the interval, so the refetch proceeds.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var divs = corpusAged(today, 95);
var store = svc.store();
const spec = DataService.dividendTtl(divs[0..]);
// A MINUTE either side, not a second. `writeAged` stamps `expires` off
// the wall clock at write time and `askedRecently` recovers the age off
// the wall clock at read time, so the entry ages by however long the gap
// between the two takes. At one second that gap was the margin: under
// kcov the instrumented binary occasionally crossed a second boundary,
// the "inside" entry read as 12h old, the refetch went ahead, and the
// network assertion panicked the coverage run. A minute is far beyond
// any real gap and still nothing against a 12-hour interval.
const margin: i64 = 60;
// Inside the interval -> still served.
writeAged(Dividend, &store, "TEST", divs[0..], spec, DataService.smart_refresh_recheck_interval - margin);
{
svc.panic_on_network_attempt = true;
const held = try svc.getDividends("TEST", .{});
defer held.deinit();
try std.testing.expectEqual(Source.cached, held.source);
svc.panic_on_network_attempt = false;
}
// Past it -> asked again (NoApiKey proves we got to the provider rather
// than being served the cache).
writeAged(Dividend, &store, "TEST", divs[0..], spec, DataService.smart_refresh_recheck_interval + margin);
try std.testing.expectError(DataError.NoApiKey, svc.getDividends("TEST", .{}));
}
test "recheck floor: a Tiingo supplement does not count as having asked" {
// WHY THE FLOOR READS `#!expires=` AND NOT `#!created=`.
//
// Dividend rows also arrive as a side effect of a Tiingo candle fetch,
// via `writeSupplement`. That rewrites the file - so `#!created=`
// moves - but it deliberately puts the old `#!expires=` back, because
// only the primary provider owns the freshness clock.
//
// If the floor keyed off `created`, any candle refresh would look like
// "we just asked Polygon" and suppress a genuinely due refetch for
// twelve hours. Keying off `expires` is immune.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var divs = corpusAged(today, 95);
var store = svc.store();
// Asked two days ago, so the floor has expired.
writeAged(Dividend, &store, "TEST", divs[0..], DataService.dividendTtl(divs[0..]), 2 * std.time.s_per_day);
// Now a candle fetch supplements the same file, moving `created` to
// now and leaving `expires` alone.
store.writeSupplement(Dividend, "TEST", divs[0..], "tiingo");
// Still asked. If this returns cached data, the floor is reading the
// wrong directive.
try std.testing.expectError(DataError.NoApiKey, svc.getDividends("TEST", .{}));
}
test "recheck floor: an entry with no expires directive is asked again" {
// Fails safe. A file carrying no `#!expires=` gives the floor nothing
// to measure from, and guessing "recently" would strand the symbol.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var divs = corpusAged(today, 95);
try std.testing.expect(svc.askedRecently(Dividend, DataService.dividendTtl, "TEST", .{
.data = divs[0..],
.timestamp = 0,
.expires = null,
}) == false);
}
test "recheck floor: the write time is recovered exactly, jitter and all" {
// The floor subtracts the applied TTL from `#!expires=` to get the
// write time. That is only exact because `computeExpires` derives its
// jitter from a hash of the symbol, so the same symbol always gets the
// same offset and it can be recomputed later.
//
// Checked across several symbols because a jitter bug would show up as
// a per-symbol discrepancy, not a uniform one.
for ([_][]const u8{ "TESTA", "TESTB", "TESTC", "TESTD" }) |sym| {
const spec = cache.DataType.dividends.ttl();
const applied = cache.computeExpires(0, spec, sym);
// Jitter must actually be doing something, or this proves nothing.
try std.testing.expect(applied != cache.Ttl.dividends or spec.jitter_pct == 0);
const now_s: i64 = 1_800_000_000;
const expires = cache.computeExpires(now_s, spec, sym);
try std.testing.expectEqual(now_s, expires - applied);
}
}
test "dividendTtl: six days when the schedule is unknown, fourteen when it is known" {
// The two clocks, and the reason they differ. A symbol with a
// established cadence is found by prediction, so its clock can be
// long; one without is found only by the clock, so it must be short
// enough to expire before the next weekly review.
const today = Date.fromYmd(2026, 9, 19);
// Five years of quarterly history -> schedule known -> long clock.
const established = quarterlySeries(6, today.addDays(-30), 91);
try std.testing.expectEqual(cache.Ttl.dividends_scheduled, DataService.dividendTtl(&established).seconds);
// Two records -> no cadence can be inferred -> short clock.
const sparse = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 3, 17 } });
try std.testing.expectEqual(cache.Ttl.dividends, DataService.dividendTtl(&sparse).seconds);
// Nothing at all -> short clock.
const none: []const Dividend = &.{};
try std.testing.expectEqual(cache.Ttl.dividends, DataService.dividendTtl(none).seconds);
// Duplicate ex-dates give a zero cadence, which is not a schedule.
const dup = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 6, 15 }, .{ 2026, 6, 15 } });
try std.testing.expectEqual(cache.Ttl.dividends, DataService.dividendTtl(&dup).seconds);
// Both keep the shared jitter policy.
const jitter = cache.DataType.dividends.ttl().jitter_pct;
try std.testing.expectEqual(jitter, DataService.dividendTtl(&established).jitter_pct);
try std.testing.expectEqual(jitter, DataService.dividendTtl(&sparse).jitter_pct);
}
test "ttlSpecFor: the hook wins when present, the type default otherwise" {
// The one-line dispatch that decides which clock a write gets. Only
// reachable in production after a successful provider fetch, so it is
// exercised directly here rather than left to a live-network path no
// test can take.
const today = Date.fromYmd(2026, 9, 19);
var established = quarterlySeries(6, today.addDays(-30), 91);
// With the dividend hook: the schedule is known, so the long clock.
try std.testing.expectEqual(
cache.Ttl.dividends_scheduled,
DataService.ttlSpecFor(Dividend, DataService.dividendTtl, established[0..]).seconds,
);
// Without a hook, the same records get the type default - which is the
// cautious short value. This is what every other type sees.
try std.testing.expectEqual(
cache.Ttl.dividends,
DataService.ttlSpecFor(Dividend, null, established[0..]).seconds,
);
// A type that has no hook at all.
var splits = [_]Split{.{ .date = Date.fromYmd(2024, 3, 7), .numerator = 3, .denominator = 1 }};
try std.testing.expectEqual(
cache.Ttl.splits,
DataService.ttlSpecFor(Split, null, splits[0..]).seconds,
);
}
test "dividendTtl: the chosen clock is what actually lands on disk" {
// `dividendTtl` returning the right number is worth nothing if the
// write path ignores it, so this reads `#!expires=` back off the file.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var store = cache.Store.init(io, allocator, dir_path);
const today = fmt.todayDate(io);
// `store.write` reads the clock itself, so a single `now` captured
// here can be a second behind the one it stamps - which flaked this
// test as `expected 516344, found 516345`. Bracket each write
// instead: the stamped expiry must be `offset` past SOME instant in
// [before, after].
const Bracket = struct {
fn expect(expires: i64, offset: i64, before_s: i64, after_s: i64) !void {
try std.testing.expect(expires >= before_s + offset);
try std.testing.expect(expires <= after_s + offset);
}
};
// Established payer -> ~14 days.
var established = quarterlySeries(6, today.addDays(-30), 91);
const long_before = std.Io.Timestamp.now(io, .real).toSeconds();
store.write(Dividend, "TESTA", established[0..], DataService.dividendTtl(&established));
const long_after = std.Io.Timestamp.now(io, .real).toSeconds();
const long_expires = readExpires(io, allocator, dir_path, "TESTA") orelse return error.TestUnexpectedResult;
try Bracket.expect(
long_expires,
cache.computeExpires(0, DataService.dividendTtl(&established), "TESTA"),
long_before,
long_after,
);
// Sparse history -> ~6 days.
var sparse = divsFromYmd(.{ .{ 2026, 6, 15 }, .{ 2026, 3, 17 } });
const short_before = std.Io.Timestamp.now(io, .real).toSeconds();
store.write(Dividend, "TESTB", sparse[0..], DataService.dividendTtl(&sparse));
const short_after = std.Io.Timestamp.now(io, .real).toSeconds();
const short_expires = readExpires(io, allocator, dir_path, "TESTB") orelse return error.TestUnexpectedResult;
try Bracket.expect(
short_expires,
cache.computeExpires(0, DataService.dividendTtl(&sparse), "TESTB"),
short_before,
short_after,
);
// And the long one really is longer, so nobody can "simplify" the two
// constants into one without this failing.
try std.testing.expect(long_expires > short_expires);
}
test "dividendTtl: a supplement to an unseen symbol gets the cautious clock" {
// `writeSupplement` has no records to reason about when the file does
// not exist yet, so it falls back to `DataType.dividends.ttl()`. That
// must be the SHORT value, because a symbol we have never seen before
// is precisely the unknown-schedule case.
try std.testing.expectEqual(cache.Ttl.dividends, cache.DataType.dividends.ttl().seconds);
}
test "full cycle: a quarterly ETF through one whole period" {
// The end-to-end story, on the pattern from the discussion that
// produced this code: ex-dates on 1/1, 4/1, 7/1 and 10/1, five years
// of history, newest cached record 2026-07-01.
//
// Median gap is 91 days, so the forecast for the next ex-date is
// 2026-09-30 - one day early, because Jul->Oct is actually 92. Being
// early is harmless; the window is fourteen days wide.
const newest = Date.fromYmd(2026, 7, 1);
var divs: [19]Dividend = undefined;
{
var i: usize = 0;
var y: i16 = 2026;
while (y >= 2022) : (y -= 1) {
for ([_]u8{ 10, 7, 4, 1 }) |m| {
const d = Date.fromYmd(y, m, 1);
if (!newest.lessThan(d) and i < divs.len) {
divs[i] = .{ .ex_date = d, .amount = 1.0, .type = .regular };
i += 1;
}
}
}
try std.testing.expectEqual(divs.len, i);
}
try std.testing.expectEqual(@as(i32, 91), DataService.dividendCadenceDays(&divs).?);
// Schedule known, so the clock is the long one.
try std.testing.expectEqual(cache.Ttl.dividends_scheduled, DataService.dividendTtl(&divs).seconds);
// NOT EXPECTING ANYTHING: the whole quarter between payments.
for ([_]Date{
Date.fromYmd(2026, 7, 2),
Date.fromYmd(2026, 8, 1),
Date.fromYmd(2026, 9, 1),
Date.fromYmd(2026, 9, 29),
}) |d| {
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, d));
}
// OVERDUE BY PREDICTION: from the forecast date to fourteen days on.
for ([_]Date{
Date.fromYmd(2026, 9, 30),
Date.fromYmd(2026, 10, 1),
Date.fromYmd(2026, 10, 7),
Date.fromYmd(2026, 10, 14),
}) |d| {
try std.testing.expect(DataService.dividendsNeedRefresh(&divs, d));
}
// GAVE UP: past the window the schedule is treated as wrong and the
// clock takes over, so we stop asking every run.
try std.testing.expect(!DataService.dividendsNeedRefresh(&divs, Date.fromYmd(2026, 10, 15)));
// The 10/1 record arrives. Asking stops immediately...
var with_oct: [20]Dividend = undefined;
with_oct[0] = .{ .ex_date = Date.fromYmd(2026, 10, 1), .amount = 1.0, .type = .regular };
@memcpy(with_oct[1..], &divs);
try std.testing.expect(!DataService.dividendsNeedRefresh(&with_oct, Date.fromYmd(2026, 10, 1)));
try std.testing.expect(!DataService.dividendsNeedRefresh(&with_oct, Date.fromYmd(2026, 12, 1)));
// ...and the forecast self-corrects: the newest gaps now median to 92,
// which lands the next expected ex-date exactly on 2027-01-01.
try std.testing.expectEqual(@as(i32, 92), DataService.dividendCadenceDays(&with_oct).?);
try std.testing.expect(DataService.dividendsNeedRefresh(&with_oct, Date.fromYmd(2027, 1, 1)));
}
/// Read the raw `#!expires=` directive off a cached dividend file.
fn readExpires(io: std.Io, allocator: std.mem.Allocator, dir_path: []const u8, symbol: []const u8) ?i64 {
const path = std.fs.path.join(allocator, &.{ dir_path, symbol, "dividends.srf" }) catch return null;
defer allocator.free(path);
const data = std.Io.Dir.cwd().readFileAlloc(io, path, allocator, .limited(1024 * 1024)) catch return null;
defer allocator.free(data);
var reader = std.Io.Reader.fixed(data);
var it = srf.iterator(&reader, allocator, .{ .parse_allocator = .none }) catch return null;
defer it.deinit();
return it.expires;
}
test "getEarnings: skip_network suppresses the smart refresh" {
// Offline mode must serve the incomplete-but-fresh entry rather than
// discard it and fail. Same rule the dividend hook follows, and it now
// comes from one place instead of two.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
const today = fmt.todayDate(io);
var store = svc.store();
var pending = [_]EarningsEvent{
.{ .symbol = "TEST", .date = today.addDays(-3), .estimate = 1.5 },
};
store.write(EarningsEvent, "TEST", pending[0..], cache.DataType.earnings.ttl());
svc.panic_on_network_attempt = true;
const result = try svc.getEarnings("TEST", .{ .skip_network = true });
defer result.deinit();
try std.testing.expectEqual(@as(usize, 1), result.data.len);
try std.testing.expectEqual(Source.cached, result.source);
}
test "loadAllDividends: honors skip_network for every symbol, and one miss does not abort the rest" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
// TSTB is cached; TSTA is absent. TSTA comes first, so if a per-symbol
// failure aborted the loop, TSTB would never be reached.
var divs = [_]Dividend{
.{ .ex_date = Date.fromYmd(2026, 3, 15), .amount = 0.50, .type = .regular },
};
var store = svc.store();
store.write(Dividend, "TSTB", divs[0..], cache.DataType.dividends.ttl());
// The whole loop must stay offline, not just the first symbol.
svc.panic_on_network_attempt = true;
svc.loadAllDividends(&.{ "TSTA", "TSTB" }, .{ .skip_network = true });
// The cached symbol survives the pass.
const b = svc.getCachedDividends(allocator, "TSTB") orelse return error.TestUnexpectedResult;
defer b.deinit();
try std.testing.expectEqual(@as(usize, 1), b.data.len);
// The absent symbol stays absent - a warm must not leave a negative or
// empty entry behind that would mask a later real fetch.
try std.testing.expect(svc.getCachedDividends(allocator, "TSTA") == null);
}
test "getCachedSplits: reads cache only and reports absence" {
// Splits travel with dividends wherever raw `close` is adjusted, so
// this mirrors `getCachedDividends` exactly - including that an
// absent symbol returns null rather than an empty slice, which is a
// different fact (nothing cached vs. cached-and-never-split).
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var splits = [_]Split{
.{ .date = Date.fromYmd(2024, 6, 10), .numerator = 10, .denominator = 1 },
};
var store = svc.store();
store.write(Split, "TSTB", splits[0..], cache.DataType.splits.ttl());
svc.panic_on_network_attempt = true;
const b = svc.getCachedSplits(allocator, "TSTB") orelse return error.TestUnexpectedResult;
defer b.deinit();
try std.testing.expectEqual(@as(usize, 1), b.data.len);
try std.testing.expectApproxEqAbs(@as(f64, 10.0), b.data[0].ratio(), 1e-9);
try std.testing.expect(svc.getCachedSplits(allocator, "TSTA") == null);
}
test "loadAllDividends: empty symbol list is a no-op" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
// No symbols means no fetches, so this must hold even with network
// otherwise allowed.
svc.panic_on_network_attempt = true;
svc.loadAllDividends(&.{}, .{});
}
test "getQuote offline mode returns FetchFailed (quotes never cached)" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
svc.panic_on_network_attempt = true;
// Quotes have no cache to fall back to in offline mode.
const err = svc.getQuote("AAPL", .{ .skip_network = true });
try std.testing.expectError(DataError.FetchFailed, err);
}
test "loadAllPrices offline mode skips network and returns cached" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var store = svc.store();
// Symbol with fresh cache.
var fresh_candles = [_]Candle{
.{ .date = Date.fromYmd(2026, 5, 20), .open = 100, .high = 105, .low = 99, .close = 104, .adj_close = 104, .volume = 1000 },
};
store.cacheCandles("FRESH", fresh_candles[0..], .{ .provider = .tiingo }, market.nextCandleExpiry(std.Io.Timestamp.now(io, .real).toSeconds(), .equity));
// Symbol with no cache at all.
// (no setup needed - just passes a symbol that doesn't exist)
svc.panic_on_network_attempt = true;
const symbols = [_][]const u8{ "FRESH", "MISSING" };
var result = svc.loadAllPrices(
symbols[0..],
&.{},
.{ .skip_network = true },
null,
null,
);
defer result.prices.deinit();
// FRESH should resolve from cache.
try std.testing.expect(result.prices.contains("FRESH"));
try std.testing.expectEqual(@as(f64, 104), result.prices.get("FRESH").?);
// MISSING should not be in the prices map.
try std.testing.expect(!result.prices.contains("MISSING"));
// failed_count should reflect MISSING.
try std.testing.expectEqual(@as(usize, 1), result.failed_count);
}
// ── adjustment-basis restatement ─────────────────────────────
test "getCandles offline never escalates a stale adjustment basis" {
// Restatement does network I/O, so `skip_network` must return the
// cached series untouched rather than escalating. Pinned with
// `panic_on_network_attempt`, which fires inside
// `refetchFullHistory`'s `assertNetworkAllowed` if this regresses.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var store = svc.store();
var candles = [_]Candle{
.{ .date = Date.fromYmd(2026, 8, 12), .open = 10, .high = 10, .low = 10, .close = 10, .adj_close = 10, .volume = 1 },
.{ .date = Date.fromYmd(2026, 8, 13), .open = 11, .high = 11, .low = 11, .close = 11, .adj_close = 11, .volume = 1 },
};
store.cacheCandles("SMPL", candles[0..], .{ .provider = .tiingo }, market.nextCandleExpiry(std.Io.Timestamp.now(io, .real).toSeconds(), .equity));
// Roll the basis back behind a cached ex-date so the series is
// unambiguously due for restatement.
var divs = [_]Dividend{.{ .ex_date = Date.fromYmd(2026, 7, 1), .amount = 1.0 }};
store.write(Dividend, "SMPL", divs[0..], .{ .seconds = cache.Ttl.dividends });
var meta = (store.readCandleMeta("SMPL") orelse return error.NoCache).meta;
meta.adj_basis = Date.fromYmd(2026, 6, 1);
store.updateCandleMeta("SMPL", meta, 1); // expiry in the past => stale
try std.testing.expect(freshness.adjustmentBasisStale(
meta.adj_basis,
freshness.newestCorporateAction(allocator, &store, "SMPL", meta.last_date),
));
svc.panic_on_network_attempt = true;
const result = try svc.getCandles("SMPL", .{ .skip_network = true });
defer result.deinit();
// Served from cache, series intact, basis untouched.
try std.testing.expectEqual(@as(usize, 2), result.data.len);
const after = (store.readCandleMeta("SMPL") orelse return error.NoCache).meta;
try std.testing.expect(after.adj_basis.eql(Date.fromYmd(2026, 6, 1)));
// And critically, no negative-cache marker was written over the
// candle file.
try std.testing.expect(!store.isNegative("SMPL", .candles_daily));
}
// ── externally-managed candle series (provider::external) ────
// The label is pure provenance and must stay behavior-free, with one
// exception: it vetoes the negative-cache write, because for a series
// no provider carries, that marker is unrecoverable rather than merely
// sticky. These tests pin all three halves - the label changes nothing
// about how a cache is served, it does change whether a unanimous 404
// gets remembered, and it is itself overwritten the moment a provider
// starts serving the symbol.
test "shouldNegativeCache refuses an external-provider symbol" {
// Looped over every variant so a fifth one has to come here and
// declare its intent rather than silently inheriting `true`.
for (std.enums.values(cache.Store.CandleProvider)) |provider| {
const meta = cache.Store.CandleMeta{
.last_close = 10.0,
.last_date = Date.fromYmd(2026, 8, 14),
.provider = provider,
};
const expected = provider != .external;
try std.testing.expectEqual(expected, DataService.shouldNegativeCache(meta));
}
// A true cold start has no label to consult and nothing local to
// destroy, so the entry is still written. This is the documented
// residual hole: an external symbol cold-started while ZFIN_SERVER
// is unreachable stays poisoned until `--refresh-data=force`.
try std.testing.expect(DataService.shouldNegativeCache(null));
}
test "a successful provider fetch relabels an external symbol (the label is not sticky)" {
// `external` records where the bars came from NOW, not a permanent
// property of the symbol. A provider serving it means the cached
// bars genuinely came from that provider, so relabelling is the
// honest outcome - and it is the good outcome, because the symbol
// has rejoined the ordinary provider path under its own steam.
//
// This test exists to stop a well-meaning future guard that
// preserves `.external` across a successful fetch. That would make
// the label a one-way door: a symbol that gained coverage would be
// pinned to a tier only one machine can populate, which is the
// latch bug that routing off `provider == .yahoo` already caused.
const was_external = cache.Store.CandleMeta{
.last_close = 25,
.last_date = Date.fromYmd(2026, 8, 14),
.provider = .external,
};
// Tiingo started carrying it.
const via_tiingo = DataService.applyTiingoCoverage(was_external, "XTRN", 1_787_000_000, .tiingo, .covered);
try std.testing.expectEqual(cache.Store.CandleProvider.tiingo, via_tiingo.provider);
// Yahoo did, with Tiingo still disclaiming it. Provenance follows
// whoever actually answered, same as for any other symbol.
const via_yahoo = DataService.applyTiingoCoverage(was_external, "XTRN", 1_787_000_000, .yahoo, .not_found);
try std.testing.expectEqual(cache.Store.CandleProvider.yahoo, via_yahoo.provider);
// And once relabelled, the negative-cache veto no longer applies -
// the symbol is an ordinary provider-sourced series again, so a
// later unanimous 404 is a real verdict worth remembering.
try std.testing.expect(DataService.shouldNegativeCache(via_tiingo));
try std.testing.expect(DataService.shouldNegativeCache(via_yahoo));
try std.testing.expect(!DataService.shouldNegativeCache(was_external));
}
test "getCandles serves an external-provider cache without touching the network" {
// `provider::external` must be inert on the serve path. In
// particular it must not fall into the `.twelvedata` carve-out,
// which treats a cache as unusable and forces a full re-fetch - for
// a symbol no provider carries, that re-fetch 404s and the
// cold-start path writes a negative marker over the only copy of
// the series. `panic_on_network_attempt` fires if any of that
// regresses.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var store = svc.store();
var bars = [_]Candle{
.{ .date = Date.fromYmd(2026, 8, 13), .open = 24, .high = 24, .low = 24, .close = 24, .adj_close = 24, .volume = 0 },
.{ .date = Date.fromYmd(2026, 8, 14), .open = 25, .high = 25, .low = 25, .close = 25, .adj_close = 25, .volume = 0 },
};
store.cacheCandles("XTRN", bars[0..], .{ .provider = .external }, 9_999_999_999);
svc.panic_on_network_attempt = true;
const result = try svc.getCandles("XTRN", .{});
defer result.deinit();
try std.testing.expectEqual(Source.cached, result.source);
try std.testing.expectEqual(@as(usize, 2), result.data.len);
try std.testing.expect(result.data[1].date.eql(Date.fromYmd(2026, 8, 14)));
// Label survived the read, and no marker was written over the file.
const after = (store.readCandleMeta("XTRN") orelse return error.NoCache).meta;
try std.testing.expectEqual(cache.Store.CandleProvider.external, after.provider);
try std.testing.expect(!store.isNegative("XTRN", .candles_daily));
}
test "getCandles offline serves an external-provider cache" {
// The skip_network path has its own `.twelvedata` check, distinct
// from the one on the online path. `.external` must miss that one
// too, otherwise offline mode reports an externally-managed holding
// as unavailable despite a perfectly good local series.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var store = svc.store();
var bars = [_]Candle{
.{ .date = Date.fromYmd(2026, 8, 14), .open = 25, .high = 25, .low = 25, .close = 25, .adj_close = 25, .volume = 0 },
};
// Expiry in the past: stale on purpose, since offline mode is meant
// to serve a stale externally-managed series rather than fail.
store.cacheCandles("XTRN", bars[0..], .{ .provider = .external }, 1);
svc.panic_on_network_attempt = true;
const result = try svc.getCandles("XTRN", .{ .skip_network = true });
defer result.deinit();
try std.testing.expectEqual(Source.cached, result.source);
try std.testing.expectEqual(@as(usize, 1), result.data.len);
try std.testing.expect(!store.isNegative("XTRN", .candles_daily));
}
// ── Tiingo coverage bookkeeping ──────────────────────────────//
// Regression suite for the one-way drift to Yahoo. The old code
// routed on `CandleMeta.provider == .yahoo`, so any non-transient
// Tiingo failure latched permanently: Yahoo was tried first,
// succeeded, rewrote `provider = .yahoo`, and Tiingo was never asked
// again. 22 of 32 cached symbols had drifted off Tiingo that way.
test "applyTiingoCoverage: .covered clears an armed Tiingo backoff" {
const armed = cache.Store.CandleMeta{
.last_close = 100,
.last_date = Date.fromYmd(2026, 8, 14),
.provider = .yahoo,
.fail_count = 2,
.tiingo_retry_after_s = 1_790_000_000,
};
const next = DataService.applyTiingoCoverage(armed, "SMPL", 1_787_000_000, .tiingo, .covered);
// Tiingo just served the symbol, so the prior 404 is stale news.
try std.testing.expectEqual(@as(i64, 0), next.tiingo_retry_after_s);
try std.testing.expectEqual(cache.Store.CandleProvider.tiingo, next.provider);
// A successful fetch also resets the transient-failure counter.
try std.testing.expectEqual(@as(u8, 0), next.fail_count);
}
test "applyTiingoCoverage: .not_found arms a jittered backoff ~30 days out" {
const now_s: i64 = 1_787_000_000;
const meta = cache.Store.CandleMeta{
.last_close = 100,
.last_date = Date.fromYmd(2026, 8, 14),
.provider = .tiingo,
};
const next = DataService.applyTiingoCoverage(meta, "SMPL", now_s, .yahoo, .not_found);
// Yahoo answered, so provenance records Yahoo...
try std.testing.expectEqual(cache.Store.CandleProvider.yahoo, next.provider);
// ...and the backoff lands 30 days out, +/- the jitter window.
const base = now_s + cache.Ttl.tiingo_backoff;
const max_offset = @divFloor(cache.Ttl.tiingo_backoff * @as(i64, tiingo_backoff_jitter_pct), 100);
try std.testing.expect(next.tiingo_retry_after_s >= base - max_offset);
try std.testing.expect(next.tiingo_retry_after_s <= base + max_offset);
// The spread must be meaningful but bounded: roughly +/-2 days.
try std.testing.expect(max_offset >= 2 * std.time.s_per_day);
try std.testing.expect(max_offset <= 3 * std.time.s_per_day);
}
test "applyTiingoCoverage: .unknown leaves the backoff untouched" {
// A 400 / 402 / malformed body says nothing about coverage. It
// must neither arm a backoff (that's the drift bug) nor clear an
// existing one (that would defeat the 404 we already recorded).
const now_s: i64 = 1_787_000_000;
const unarmed = cache.Store.CandleMeta{
.last_close = 100,
.last_date = Date.fromYmd(2026, 8, 14),
.provider = .tiingo,
};
const a = DataService.applyTiingoCoverage(unarmed, "SMPL", now_s, .yahoo, .unknown);
try std.testing.expectEqual(@as(i64, 0), a.tiingo_retry_after_s);
var armed = unarmed;
armed.tiingo_retry_after_s = 1_790_000_000;
const b = DataService.applyTiingoCoverage(armed, "SMPL", now_s, .yahoo, .unknown);
try std.testing.expectEqual(@as(i64, 1_790_000_000), b.tiingo_retry_after_s);
}
test "applyTiingoCoverage: backoff jitter is deterministic per symbol" {
// Deterministic-by-symbol (Wyhash via computeExpires), NOT random:
// repeated writes for the same symbol must not drift the deadline,
// and distinct symbols must land on different days so a batch
// demoted in one window does not all re-probe together.
const now_s: i64 = 1_787_000_000;
const meta = cache.Store.CandleMeta{
.last_close = 100,
.last_date = Date.fromYmd(2026, 8, 14),
.provider = .tiingo,
};
const a1 = DataService.applyTiingoCoverage(meta, "SMPLA", now_s, .yahoo, .not_found);
const a2 = DataService.applyTiingoCoverage(meta, "SMPLA", now_s, .yahoo, .not_found);
try std.testing.expectEqual(a1.tiingo_retry_after_s, a2.tiingo_retry_after_s);
// Spread check across a handful of symbols: they must not all
// collapse onto the same instant.
const syms = [_][]const u8{ "SMPLA", "SMPLB", "SMPLC", "SMPLD", "SMPLE", "SMPLF" };
var seen: [syms.len]i64 = undefined;
for (syms, 0..) |sym, i| {
seen[i] = DataService.applyTiingoCoverage(meta, sym, now_s, .yahoo, .not_found).tiingo_retry_after_s;
}
var distinct: usize = 0;
for (seen, 0..) |v, i| {
var is_new = true;
for (seen[0..i]) |prev| {
if (prev == v) is_new = false;
}
if (is_new) distinct += 1;
}
try std.testing.expect(distinct > 1);
}
test "isPermanentProviderFailure gates which Tiingo errors are remembered" {
// The rule that Commit 1 reuses for the Yahoo-demotion decision:
// only a genuine 404 is a statement about the symbol. Everything
// else describes one HTTP call.
try std.testing.expect(isPermanentProviderFailure(error.NotFound));
try std.testing.expect(!isPermanentProviderFailure(error.InvalidResponse));
try std.testing.expect(!isPermanentProviderFailure(error.PaymentRequired));
try std.testing.expect(!isPermanentProviderFailure(error.ParseError));
try std.testing.expect(!isPermanentProviderFailure(error.RateLimited));
try std.testing.expect(!isPermanentProviderFailure(error.Unauthorized));
}
test "loadAllPrices force_refresh tops up without wiping the candle cache" {
// Regression: force_refresh must mean "ignore TTL + incremental
// top-up", NOT "delete the cache and re-download from scratch".
// The old behavior invalidated (deleted) candles_daily before the
// fetch, which forced a full network re-download. With the cache
// already covering through today, force_refresh must serve from
// the surviving cache and touch no network.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var store = svc.store();
// Dated far in the future so getCandles' "last cached date is
// today-or-later" branch fires deterministically regardless of the
// test clock - an incremental fetch would have nothing to pull and
// never reaches the network.
var candles = [_]Candle{
.{ .date = Date.fromYmd(2099, 12, 31), .open = 100, .high = 105, .low = 99, .close = 104, .adj_close = 104, .volume = 1000 },
};
store.cacheCandles("HELD", candles[0..], .{ .provider = .tiingo }, market.nextCandleExpiry(std.Io.Timestamp.now(io, .real).toSeconds(), .equity));
// Any provider/network attempt now panics. If force_refresh wiped
// the cache (old behavior), getCandles would fall through to a full
// re-fetch and trip this.
svc.panic_on_network_attempt = true;
const symbols = [_][]const u8{"HELD"};
var result = svc.loadAllPrices(
symbols[0..],
&.{},
.{ .force_refresh = true },
null,
null,
);
defer result.prices.deinit();
// Served from the (un-wiped) cache.
try std.testing.expect(result.prices.contains("HELD"));
try std.testing.expectEqual(@as(f64, 104), result.prices.get("HELD").?);
// The candle cache survived the force-refresh.
try std.testing.expect(svc.getCachedLastClose("HELD") != null);
}
test "getClassification: skip_network with no cache returns FetchFailed" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
svc.panic_on_network_attempt = true;
const err = svc.getClassification("NEVERHEARDOFIT", .{ .skip_network = true });
try std.testing.expectError(DataError.FetchFailed, err);
}
test "getClassification: cache hit returns cached data without network" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
// Pre-populate the classification cache.
var s = svc.store();
var records = [_]Wikidata.ClassificationRecord{.{
.symbol = "AAPL",
.name = "Apple Inc.",
.country = "US",
.as_of = "2026-05-25",
.source = "wikidata",
}};
s.write(Wikidata.ClassificationRecord, "AAPL", records[0..], .{ .seconds = cache.Ttl.classification });
// Network guard on - must return from cache without touching network.
svc.panic_on_network_attempt = true;
const result = try svc.getClassification("AAPL", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 1), result.data.len);
try std.testing.expectEqualStrings("AAPL", result.data[0].symbol);
try std.testing.expectEqualStrings("Apple Inc.", result.data[0].name.?);
try std.testing.expectEqual(Source.cached, result.source);
}
test "populateGeo: country US -> geo US" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var record: Wikidata.ClassificationRecord = .{
.symbol = try allocator.dupe(u8, "TEST"),
.country = try allocator.dupe(u8, "US"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "wikidata"),
};
defer record.deinit(allocator);
try svc.populateGeo(&record);
try std.testing.expect(record.geo != null);
try std.testing.expectEqualStrings("US", record.geo.?);
}
test "populateGeo: country GB -> geo International Developed" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var record: Wikidata.ClassificationRecord = .{
.symbol = try allocator.dupe(u8, "TEST"),
.country = try allocator.dupe(u8, "GB"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "wikidata"),
};
defer record.deinit(allocator);
try svc.populateGeo(&record);
try std.testing.expect(record.geo != null);
try std.testing.expectEqualStrings("International Developed", record.geo.?);
}
test "populateGeo: null country -> noop" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var record: Wikidata.ClassificationRecord = .{
.symbol = try allocator.dupe(u8, "TEST"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "wikidata"),
};
defer record.deinit(allocator);
try svc.populateGeo(&record);
try std.testing.expectEqual(@as(?[]const u8, null), record.geo);
}
test "populateGeo: existing geo not overwritten" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var record: Wikidata.ClassificationRecord = .{
.symbol = try allocator.dupe(u8, "TEST"),
.country = try allocator.dupe(u8, "US"),
.geo = try allocator.dupe(u8, "Already Set"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "wikidata"),
};
defer record.deinit(allocator);
try svc.populateGeo(&record);
try std.testing.expectEqualStrings("Already Set", record.geo.?);
}
test "getClassification: sparse Wikidata + EDGAR managed_fund hit produces merged record" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
// Seed both EDGAR ticker map caches with at least one entry
// each so the synthesizeClassification path doesn't try to
// fetch them (the load helpers treat empty cached slices as
// "miss" and fall through to a network fetch).
var s = svc.store();
var mf_entries = [_]Edgar.MutualFundTickerEntry{.{
.symbol = "FAGIX",
.cik = "0000275309",
}};
s.write(Edgar.MutualFundTickerEntry, "_edgar", mf_entries[0..], cache.DataType.tickers_funds.ttl());
var co_entries = [_]Edgar.CompanyTickerEntry{.{
.symbol = "DUMMY",
.cik = "0000000001",
}};
s.write(Edgar.CompanyTickerEntry, "_edgar", co_entries[0..], cache.DataType.tickers_companies.ttl());
// Seed an etf_metrics negative cache so getEtfMetrics doesn't
// try to fetch from the network.
s.writeNegative("FAGIX", .etf_metrics);
// Sparse Wikidata records (length 1, only name set -- not useful).
var sparse = try allocator.alloc(Wikidata.ClassificationRecord, 1);
sparse[0] = .{
.symbol = try allocator.dupe(u8, "FAGIX"),
.name = try allocator.dupe(u8, "Test Fund"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "wikidata"),
};
// Drive directly through synthesizeClassification (skip the
// Wikidata fetch). It takes ownership of `sparse`.
svc.panic_on_network_attempt = true; // any provider call -> panic
const merged = try svc.synthesizeClassification("FAGIX", sparse, .{ .skip_network = true });
defer Wikidata.ClassificationRecord.freeSlice(allocator, merged);
try std.testing.expectEqual(@as(usize, 1), merged.len);
const c = merged[0];
try std.testing.expectEqualStrings("FAGIX", c.symbol);
try std.testing.expect(c.is_etf);
try std.testing.expectEqualStrings("Fund", c.asset_class.?);
try std.testing.expectEqualStrings("US", c.country.?);
try std.testing.expectEqualStrings("US", c.geo.?);
try std.testing.expectEqualStrings("edgar_fallback", c.source);
// Wikidata's name preserved on merge.
try std.testing.expectEqualStrings("Test Fund", c.name.?);
}
test "synthesizeClassification: no EDGAR hit returns NotFound" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
// Seed both ticker maps with throwaway entries so the
// EDGAR lookup returns .none for our test symbol but doesn't
// try to fetch the maps from the network.
var s = svc.store();
var mf_entries = [_]Edgar.MutualFundTickerEntry{.{
.symbol = "DUMMY1",
.cik = "0000000001",
}};
s.write(Edgar.MutualFundTickerEntry, "_edgar", mf_entries[0..], cache.DataType.tickers_funds.ttl());
var co_entries = [_]Edgar.CompanyTickerEntry{.{
.symbol = "DUMMY2",
.cik = "0000000002",
}};
s.write(Edgar.CompanyTickerEntry, "_edgar", co_entries[0..], cache.DataType.tickers_companies.ttl());
var sparse = try allocator.alloc(Wikidata.ClassificationRecord, 1);
sparse[0] = .{
.symbol = try allocator.dupe(u8, "NEVERHEARDOFIT"),
.name = try allocator.dupe(u8, "ghost"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "wikidata"),
};
svc.panic_on_network_attempt = true;
try std.testing.expectError(error.NotFound, svc.synthesizeClassification("NEVERHEARDOFIT", sparse, .{ .skip_network = true }));
}
test "synthesizeClassification: company_or_uit without ETF/TRUST keyword still routes to multi-row" {
// PTY shape: closed-end fund whose company_tickers title is
// "PIMCO CORPORATE & INCOME OPPORTUNITY FUND" -- no "ETF" or
// "TRUST" in the title, so lookupInTickerMaps returns
// .company_or_uit{is_etf=false}. But it's still fund-shaped
// and should produce multi-row metadata in enrich.
//
// The downstream signal for "fund-like, emit multi-row" is
// ClassificationRecord.is_etf. Set it to true for any
// EDGAR-found .company_or_uit hit (even when the title
// doesn't carry the ETF/TRUST keyword), so PTY-shape
// closed-end funds get the same treatment as ETFs.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var s = svc.store();
// Throwaway MF entry so the MF lookup returns null.
var mf_entries = [_]Edgar.MutualFundTickerEntry{.{
.symbol = "DUMMY",
.cik = "0000000001",
}};
s.write(Edgar.MutualFundTickerEntry, "_edgar", mf_entries[0..], cache.DataType.tickers_funds.ttl());
// PTY in the company map with NO ETF/TRUST in title.
var co_entries = [_]Edgar.CompanyTickerEntry{.{
.symbol = "PTY",
.cik = "0001202604",
.title = "PIMCO CORPORATE & INCOME OPPORTUNITY FUND",
}};
s.write(Edgar.CompanyTickerEntry, "_edgar", co_entries[0..], cache.DataType.tickers_companies.ttl());
s.writeNegative("PTY", .etf_metrics);
var sparse = try allocator.alloc(Wikidata.ClassificationRecord, 1);
sparse[0] = .{
.symbol = try allocator.dupe(u8, "PTY"),
.name = try allocator.dupe(u8, "PIMCO Corporate & Income Opportunity Fund"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "wikidata"),
};
svc.panic_on_network_attempt = true;
const merged = try svc.synthesizeClassification("PTY", sparse, .{ .skip_network = true });
defer Wikidata.ClassificationRecord.freeSlice(allocator, merged);
try std.testing.expectEqual(@as(usize, 1), merged.len);
const c = merged[0];
// is_etf MUST be true so enrich routes through emitEtfRows
// (multi-row sleeve breakdown). The asset_class stays "Fund"
// because no ETF/TRUST keyword in title.
try std.testing.expect(c.is_etf);
try std.testing.expectEqualStrings("Fund", c.asset_class.?);
}
test "synthesizeClassification: NPORT-P series_name beats Wikidata's index name for funds" {
// SOXX shape: Wikidata returns the underlying INDEX name
// ("PHLX Semiconductor Sector") which is technically what the
// ticker symbol is for, but downstream consumers want the
// FUND name ("iShares Semiconductor ETF") that NPORT-P
// <seriesName> carries. Series_name is more authoritative
// for the fund itself.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var s = svc.store();
var mf_entries = [_]Edgar.MutualFundTickerEntry{.{
.symbol = "DUMMY",
.cik = "0000000001",
}};
s.write(Edgar.MutualFundTickerEntry, "_edgar", mf_entries[0..], cache.DataType.tickers_funds.ttl());
var co_entries = [_]Edgar.CompanyTickerEntry{.{
.symbol = "SOXX",
.cik = "0001100663",
.title = "iShares Trust",
}};
s.write(Edgar.CompanyTickerEntry, "_edgar", co_entries[0..], cache.DataType.tickers_companies.ttl());
// Pre-seed etf_metrics with a profile row carrying the
// NPORT-P seriesName.
var etf_records = [_]Edgar.EtfMetricRecord{
.{ .profile = .{
.symbol = try allocator.dupe(u8, "SOXX"),
.series_name = try allocator.dupe(u8, "iShares Semiconductor ETF"),
.cik = try allocator.dupe(u8, "0001100663"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "edgar"),
} },
};
defer for (etf_records) |r| r.deinit(allocator);
s.write(Edgar.EtfMetricRecord, "SOXX", etf_records[0..], cache.DataType.etf_metrics.ttl());
// Wikidata returned only the index name (sparse).
var sparse = try allocator.alloc(Wikidata.ClassificationRecord, 1);
sparse[0] = .{
.symbol = try allocator.dupe(u8, "SOXX"),
.name = try allocator.dupe(u8, "PHLX Semiconductor Sector"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "wikidata"),
};
svc.panic_on_network_attempt = true;
const merged = try svc.synthesizeClassification("SOXX", sparse, .{ .skip_network = true });
defer Wikidata.ClassificationRecord.freeSlice(allocator, merged);
try std.testing.expectEqual(@as(usize, 1), merged.len);
const c = merged[0];
// Series_name from NPORT-P wins -- not Wikidata's index name.
try std.testing.expectEqualStrings("iShares Semiconductor ETF", c.name.?);
}
test "getEntityFacts: skip_network with no cache returns FetchFailed" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
svc.panic_on_network_attempt = true;
const err = svc.getEntityFacts("0000999999", .{ .skip_network = true });
try std.testing.expectError(DataError.FetchFailed, err);
}
test "getEntityFacts: cache hit returns cached shares-outstanding" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var s = svc.store();
var records = [_]Edgar.EntityFactRecord{
.{ .shares_outstanding = .{
.symbol = "",
.shares_outstanding = 14687356000,
.period_end = "2026-04-17",
.form = "10-Q",
.cik = "0000320193",
.as_of = "2026-05-25",
.source = "edgar_xbrl",
} },
};
s.write(Edgar.EntityFactRecord, "0000320193", records[0..], .{ .seconds = cache.Ttl.entity_facts });
svc.panic_on_network_attempt = true;
const result = try svc.getEntityFacts("0000320193", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 1), result.data.len);
switch (result.data[0]) {
.shares_outstanding => |so| {
try std.testing.expectEqual(@as(u64, 14687356000), so.shares_outstanding);
try std.testing.expectEqualStrings("0000320193", so.cik);
},
}
try std.testing.expectEqual(Source.cached, result.source);
}
test "getEtfMetrics: skip_network with no cache returns FetchFailed" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
svc.panic_on_network_attempt = true;
const err = svc.getEtfMetrics("NEVERHEARDOFIT", .{ .skip_network = true });
try std.testing.expectError(DataError.FetchFailed, err);
}
test "getEtfMetrics: cache hit returns cached profile + sectors + holdings" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
const config = Config{ .cache_dir = dir_path };
var svc = DataService.init(io, allocator, config);
defer svc.deinit();
var s = svc.store();
var records = [_]Edgar.EtfMetricRecord{
.{ .profile = .{
.symbol = "VTI",
.cik = "0000036405",
.as_of = "2026-05-25",
.source = "edgar",
} },
.{ .sector = .{
.symbol = "VTI",
.code = "EC/CORP",
.description = "Equity / Corporate",
.pct_of_portfolio = 99.7,
.as_of = "2026-05-25",
.source = "edgar",
} },
.{ .holding = .{
.symbol = "VTI",
.name = "NVIDIA Corp",
.pct_of_portfolio = 6.57,
.as_of = "2026-05-25",
.source = "edgar",
} },
};
s.write(Edgar.EtfMetricRecord, "VTI", records[0..], .{ .seconds = cache.Ttl.etf_metrics });
svc.panic_on_network_attempt = true;
const result = try svc.getEtfMetrics("VTI", .{});
defer result.deinit();
try std.testing.expectEqual(@as(usize, 3), result.data.len);
try std.testing.expect(result.data[0] == .profile);
try std.testing.expect(result.data[1] == .sector);
try std.testing.expect(result.data[2] == .holding);
try std.testing.expectEqualStrings("VTI", result.data[0].profile.symbol);
try std.testing.expectEqual(Source.cached, result.source);
}
test "DataService getProvider initializes Wikidata with user_email" {
const allocator = std.testing.allocator;
const config = Config{
.cache_dir = "/tmp/zfin-test-cache",
.user_email = "test@example.com",
};
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
const wd1 = try svc.getProvider(Wikidata);
try std.testing.expect(svc.wikidata != null);
try std.testing.expectEqualStrings("test@example.com", wd1.user_email);
// Second call returns same instance.
const wd2 = try svc.getProvider(Wikidata);
try std.testing.expect(wd1 == wd2);
}
test "DataService getProvider returns NoApiKey for Wikidata without user_email" {
const allocator = std.testing.allocator;
const config = Config{ .cache_dir = "/tmp/zfin-test-cache" };
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
const wd_result = svc.getProvider(Wikidata);
try std.testing.expectError(DataError.NoApiKey, wd_result);
const ed_result = svc.getProvider(Edgar);
try std.testing.expectError(DataError.NoApiKey, ed_result);
}
test "estimateWaitSeconds returns null when relevant provider not instantiated" {
const allocator = std.testing.allocator;
const config = Config{ .cache_dir = "/tmp/zfin-test-cache" };
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
// No providers initialized yet (lazy). Each rate-limited data
// type returns null because its provider is missing.
try std.testing.expectEqual(@as(?u64, null), svc.estimateWaitSeconds(.dividends));
try std.testing.expectEqual(@as(?u64, null), svc.estimateWaitSeconds(.splits));
try std.testing.expectEqual(@as(?u64, null), svc.estimateWaitSeconds(.earnings));
try std.testing.expectEqual(@as(?u64, null), svc.estimateWaitSeconds(.options));
try std.testing.expectEqual(@as(?u64, null), svc.estimateWaitSeconds(.etf_metrics));
try std.testing.expectEqual(@as(?u64, null), svc.estimateWaitSeconds(.entity_facts));
}
test "estimateWaitSeconds returns 0 for types without rate limiters" {
// candles_daily, classification, etc. are served by providers
// that don't have a rate limiter (Tiingo, Wikidata). The
// function returns 0 for these regardless of provider state --
// there's nothing to wait for.
const allocator = std.testing.allocator;
const config = Config{ .cache_dir = "/tmp/zfin-test-cache" };
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
try std.testing.expectEqual(@as(?u64, 0), svc.estimateWaitSeconds(.candles_daily));
try std.testing.expectEqual(@as(?u64, 0), svc.estimateWaitSeconds(.candles_meta));
try std.testing.expectEqual(@as(?u64, 0), svc.estimateWaitSeconds(.classification));
try std.testing.expectEqual(@as(?u64, 0), svc.estimateWaitSeconds(.meta));
}
test "estimateWaitSeconds returns 0 for fresh rate-limited providers" {
// Once the provider is instantiated, an unused rate limiter
// returns 0 (no wait). This is the steady-state happy path
// for the call at the top of each refresh iteration.
const allocator = std.testing.allocator;
const config = Config{
.cache_dir = "/tmp/zfin-test-cache",
.polygon_key = "test-polygon-key",
.fmp_key = "test-fmp-key",
};
var svc = DataService.init(std.testing.io, allocator, config);
defer svc.deinit();
// Touch each provider to lazy-init it. We don't care about the
// returned pointer; just need svc.pg / svc.fmp to be non-null.
_ = try svc.getProvider(Polygon);
_ = try svc.getProvider(Fmp);
// Fresh limiters have full token bucket -> 0 wait.
try std.testing.expectEqual(@as(?u64, 0), svc.estimateWaitSeconds(.dividends));
try std.testing.expectEqual(@as(?u64, 0), svc.estimateWaitSeconds(.splits));
try std.testing.expectEqual(@as(?u64, 0), svc.estimateWaitSeconds(.earnings));
}
// ── lookupInTickerMaps ────────────────────────────────────────
//
// Pure function - no I/O. Consumed by `lookupEdgarFallback`,
// which loads the maps then calls this. Tests construct
// synthetic ticker-map data directly to exercise every branch
// without touching the cache or network.
fn testNewMfEntry(allocator: std.mem.Allocator, symbol: []const u8, cik: []const u8) !Edgar.MutualFundTickerEntry {
return .{
.symbol = try allocator.dupe(u8, symbol),
.cik = try allocator.dupe(u8, cik),
};
}
fn testNewCoEntry(allocator: std.mem.Allocator, symbol: []const u8, cik: []const u8, title: ?[]const u8) !Edgar.CompanyTickerEntry {
return .{
.symbol = try allocator.dupe(u8, symbol),
.cik = try allocator.dupe(u8, cik),
.title = if (title) |t| try allocator.dupe(u8, t) else null,
};
}
test "lookupInTickerMaps: both maps null -> .none" {
const allocator = std.testing.allocator;
const result = lookupInTickerMaps(allocator, "ANY", null, null);
defer freeEdgarLookup(allocator, result);
try std.testing.expect(result == .none);
}
test "lookupInTickerMaps: symbol in MF map -> .managed_fund" {
const allocator = std.testing.allocator;
const entries = try allocator.alloc(Edgar.MutualFundTickerEntry, 1);
entries[0] = try testNewMfEntry(allocator, "FAGIX", "0000225322");
var map = try Edgar.TickerMap(Edgar.MutualFundTickerEntry).fromEntries(allocator, entries);
defer map.deinit();
const result = lookupInTickerMaps(allocator, "FAGIX", &map, null);
defer freeEdgarLookup(allocator, result);
try std.testing.expect(result == .managed_fund);
}
test "lookupInTickerMaps: symbol in company map with TRUST title -> ETF hint" {
const allocator = std.testing.allocator;
const entries = try allocator.alloc(Edgar.CompanyTickerEntry, 1);
entries[0] = try testNewCoEntry(allocator, "SPY", "0000884394", "SPDR S&P 500 ETF TRUST");
var map = try Edgar.TickerMap(Edgar.CompanyTickerEntry).fromEntries(allocator, entries);
defer map.deinit();
const result = lookupInTickerMaps(allocator, "SPY", null, &map);
defer freeEdgarLookup(allocator, result);
try std.testing.expect(result == .company_or_uit);
try std.testing.expect(result.company_or_uit.is_etf);
try std.testing.expectEqualStrings("SPDR S&P 500 ETF TRUST", result.company_or_uit.title.?);
}
test "lookupInTickerMaps: company map with operating-company title -> not ETF" {
const allocator = std.testing.allocator;
const entries = try allocator.alloc(Edgar.CompanyTickerEntry, 1);
entries[0] = try testNewCoEntry(allocator, "AAPL", "0000320193", "Apple Inc.");
var map = try Edgar.TickerMap(Edgar.CompanyTickerEntry).fromEntries(allocator, entries);
defer map.deinit();
const result = lookupInTickerMaps(allocator, "AAPL", null, &map);
defer freeEdgarLookup(allocator, result);
try std.testing.expect(result == .company_or_uit);
try std.testing.expect(!result.company_or_uit.is_etf);
}
test "lookupInTickerMaps: not in either map -> .none" {
const allocator = std.testing.allocator;
const mf_entries = try allocator.alloc(Edgar.MutualFundTickerEntry, 1);
mf_entries[0] = try testNewMfEntry(allocator, "FAGIX", "0000225322");
var mf_map = try Edgar.TickerMap(Edgar.MutualFundTickerEntry).fromEntries(allocator, mf_entries);
defer mf_map.deinit();
const result = lookupInTickerMaps(allocator, "MISSING", &mf_map, null);
defer freeEdgarLookup(allocator, result);
try std.testing.expect(result == .none);
}
test "lookupInTickerMaps: MF map takes precedence over company map" {
// If a symbol appears in both (rare but possible - class
// shares of an open-end fund vs the fund's parent company),
// we prefer the MF answer. Lock in the contract.
const allocator = std.testing.allocator;
const mf_entries = try allocator.alloc(Edgar.MutualFundTickerEntry, 1);
mf_entries[0] = try testNewMfEntry(allocator, "DUP", "0000000001");
const co_entries = try allocator.alloc(Edgar.CompanyTickerEntry, 1);
co_entries[0] = try testNewCoEntry(allocator, "DUP", "0000000002", "DUP TRUST");
var mf_map = try Edgar.TickerMap(Edgar.MutualFundTickerEntry).fromEntries(allocator, mf_entries);
defer mf_map.deinit();
var co_map = try Edgar.TickerMap(Edgar.CompanyTickerEntry).fromEntries(allocator, co_entries);
defer co_map.deinit();
const result = lookupInTickerMaps(allocator, "DUP", &mf_map, &co_map);
defer freeEdgarLookup(allocator, result);
try std.testing.expect(result == .managed_fund);
}
test "lookupInTickerMaps: company map with null title -> .company_or_uit, no ETF" {
// Defensive: if EDGAR's company file has a row with no
// title, we still return the lookup but can't infer ETF
// status from a missing string.
const allocator = std.testing.allocator;
const entries = try allocator.alloc(Edgar.CompanyTickerEntry, 1);
entries[0] = try testNewCoEntry(allocator, "BARE", "0000000001", null);
var map = try Edgar.TickerMap(Edgar.CompanyTickerEntry).fromEntries(allocator, entries);
defer map.deinit();
const result = lookupInTickerMaps(allocator, "BARE", null, &map);
defer freeEdgarLookup(allocator, result);
try std.testing.expect(result == .company_or_uit);
try std.testing.expect(!result.company_or_uit.is_etf);
try std.testing.expect(result.company_or_uit.title == null);
}
test "lookupInTickerMaps: returned title is owned (survives map deinit)" {
// Critical for the service.lookupEdgarFallback contract:
// the maps get freed before the EdgarLookup is returned to
// the caller. The title must survive that.
const allocator = std.testing.allocator;
const entries = try allocator.alloc(Edgar.CompanyTickerEntry, 1);
entries[0] = try testNewCoEntry(allocator, "VTI", "0000884394", "VANGUARD TOTAL STOCK MARKET ETF");
const result = blk: {
var map = try Edgar.TickerMap(Edgar.CompanyTickerEntry).fromEntries(allocator, entries);
defer map.deinit();
break :blk lookupInTickerMaps(allocator, "VTI", null, &map);
};
defer freeEdgarLookup(allocator, result);
// Map is gone. Title must still be readable.
try std.testing.expect(result == .company_or_uit);
try std.testing.expectEqualStrings("VANGUARD TOTAL STOCK MARKET ETF", result.company_or_uit.title.?);
try std.testing.expect(result.company_or_uit.is_etf);
}
test "freeEdgarLookup: handles all three union variants without leak" {
const allocator = std.testing.allocator;
// .managed_fund - no-op
freeEdgarLookup(allocator, .managed_fund);
// .none - no-op
freeEdgarLookup(allocator, .none);
// .company_or_uit with null title - no-op
freeEdgarLookup(allocator, .{ .company_or_uit = .{ .title = null, .is_etf = false } });
// .company_or_uit with non-null title - frees the title.
const owned = try allocator.dupe(u8, "Some Title");
freeEdgarLookup(allocator, .{ .company_or_uit = .{ .title = owned, .is_etf = true } });
// testing.allocator panics on leak - passing this test means
// the title was freed.
}
// ── CUSIP->ticker cache (loadCusipTickerMap / cacheCusipTicker) ──
test "loadCusipTickerMap: missing file returns empty map" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
var map = svc.loadCusipTickerMap(allocator);
defer map.deinit();
try std.testing.expectEqual(@as(usize, 0), map.count());
}
test "cacheCusipTicker + loadCusipTickerMap: write/read round-trip" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
// Placeholder CUSIPs/tickers - never real PII.
svc.cacheCusipTicker("111111111", "AAA");
svc.cacheCusipTicker("222222222", "BBB");
var map = svc.loadCusipTickerMap(allocator);
defer map.deinit();
try std.testing.expectEqual(@as(usize, 2), map.count());
try std.testing.expectEqualStrings("AAA", map.get("111111111").?);
try std.testing.expectEqualStrings("BBB", map.get("222222222").?);
}
test "cacheCusipTicker: dedups repeated CUSIP (the historical bug)" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
// Write the same CUSIP three times - must collapse to one row.
svc.cacheCusipTicker("111111111", "AAA");
svc.cacheCusipTicker("111111111", "AAA");
svc.cacheCusipTicker("111111111", "AAA");
var map = svc.loadCusipTickerMap(allocator);
defer map.deinit();
try std.testing.expectEqual(@as(usize, 1), map.count());
try std.testing.expectEqualStrings("AAA", map.get("111111111").?);
// The on-disk file should physically contain exactly one data
// row (plus the directive header), proving dedup at the writer.
const path = try std.fs.path.join(allocator, &.{ dir_path, "cusip_tickers.srf" });
defer allocator.free(path);
const data = try std.Io.Dir.cwd().readFileAlloc(io, path, allocator, .limited(64 * 1024));
defer allocator.free(data);
var row_count: usize = 0;
var lines = std.mem.splitScalar(u8, data, '\n');
while (lines.next()) |line| {
if (std.mem.indexOf(u8, line, "cusip::") != null) row_count += 1;
}
try std.testing.expectEqual(@as(usize, 1), row_count);
}
test "loadCusipTickerMap: first occurrence wins on duplicate rows" {
// Tolerate a pre-existing file written by the buggy appender
// (duplicate rows). The reader must not crash and must keep the
// first mapping.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
// Hand-write a file with a duplicate row (as the old bug did).
const path = try std.fs.path.join(allocator, &.{ dir_path, "cusip_tickers.srf" });
defer allocator.free(path);
try std.Io.Dir.cwd().writeFile(io, .{
.sub_path = path,
.data = "#!srfv1\ncusip::111111111,ticker::AAA\ncusip::111111111,ticker::AAA\n",
});
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
var map = svc.loadCusipTickerMap(allocator);
defer map.deinit();
try std.testing.expectEqual(@as(usize, 1), map.count());
try std.testing.expectEqualStrings("AAA", map.get("111111111").?);
}
// ── CUSIP resolution cascade (resolveCusips / appendCusipEntries) ──
test "appendCusipEntries: batches, dedups vs file and within batch" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
// Seed one entry on disk.
svc.cacheCusipTicker("111111111", "AAA");
// Batch: 111 already on disk (skip), 222 + 333 new, 222 repeated
// within the batch (skip the second).
const batch = [_]DataService.CusipEntry{
.{ .cusip = "111111111", .ticker = "ZZZ" },
.{ .cusip = "222222222", .ticker = "BBB" },
.{ .cusip = "333333333", .ticker = "CCC" },
.{ .cusip = "222222222", .ticker = "BBB" },
};
svc.appendCusipEntries(batch[0..]);
var map = svc.loadCusipTickerMap(allocator);
defer map.deinit();
try std.testing.expectEqual(@as(u32, 3), map.count());
try std.testing.expectEqualStrings("AAA", map.get("111111111").?); // file wins
try std.testing.expectEqualStrings("BBB", map.get("222222222").?);
try std.testing.expectEqualStrings("CCC", map.get("333333333").?);
// Physically exactly 3 data rows (plus the directive header).
const path = try std.fs.path.join(allocator, &.{ dir_path, "cusip_tickers.srf" });
defer allocator.free(path);
const data = try std.Io.Dir.cwd().readFileAlloc(io, path, allocator, .limited(64 * 1024));
defer allocator.free(data);
var rows: usize = 0;
var lines = std.mem.splitScalar(u8, data, '\n');
while (lines.next()) |line| {
if (std.mem.indexOf(u8, line, "cusip::") != null) rows += 1;
}
try std.testing.expectEqual(@as(usize, 3), rows);
}
test "mergeCusipBody: merges new entries, skips those already in `have` or the batch" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
// `have` already maps 111 -> AAA (local is authoritative).
svc.cacheCusipTicker("111111111", "AAA");
var have = svc.loadCusipTickerMap(allocator);
defer have.deinit();
var arena = std.heap.ArenaAllocator.init(allocator);
defer arena.deinit();
var out = std.StringHashMap([]const u8).init(arena.allocator());
// Server body: 111 conflicts with `have` (ignored), 222 + 333 are
// new, 222 repeated (the second is skipped).
const body =
"#!srfv1\n" ++
"cusip::111111111,ticker::ZZZ\n" ++
"cusip::222222222,ticker::BBB\n" ++
"cusip::333333333,ticker::CCC\n" ++
"cusip::222222222,ticker::BBB\n";
DataService.mergeCusipBody(arena.allocator(), &out, have, body);
try std.testing.expectEqual(@as(u32, 2), out.count());
try std.testing.expectEqualStrings("BBB", out.get("222222222").?);
try std.testing.expectEqualStrings("CCC", out.get("333333333").?);
try std.testing.expect(out.get("111111111") == null); // have wins
}
test "resolveCusips: warm cache resolves without touching the network" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
// No server_url; assert L2/L3 are never reached for an all-hit set.
svc.panic_on_network_attempt = true;
svc.cacheCusipTicker("111111111", "AAA");
svc.cacheCusipTicker("222222222", "BBB");
// Duplicate + empty CUSIP in the request must be tolerated.
const want = [_][]const u8{ "111111111", "222222222", "111111111", "" };
var map = svc.resolveCusips(allocator, want[0..], false);
defer map.deinit();
try std.testing.expectEqualStrings("AAA", map.get("111111111").?);
try std.testing.expectEqualStrings("BBB", map.get("222222222").?);
}
test "resolveCusips: skip_network serves L1 only, never hits the network" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
// A miss would normally fall through to L2/L3; skip_network must
// prevent any network attempt even so.
svc.panic_on_network_attempt = true;
svc.cacheCusipTicker("111111111", "AAA");
// "999999999" is absent from L1 - with skip_network it stays
// unresolved rather than triggering a server/OpenFIGI lookup.
const want = [_][]const u8{ "111111111", "999999999" };
var map = svc.resolveCusips(allocator, want[0..], true);
defer map.deinit();
try std.testing.expectEqualStrings("AAA", map.get("111111111").?);
try std.testing.expect(map.get("999999999") == null);
}
test "getEtfProfile: carries holding CUSIP through the model boundary" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, Config{ .cache_dir = dir_path });
defer svc.deinit();
// Seed etf_metrics: a profile row + a holding carrying a CUSIP but
// no ticker (the common NPORT-P shape - placeholder values only).
var etf_records = [_]Edgar.EtfMetricRecord{
.{ .profile = .{
.symbol = try allocator.dupe(u8, "TESTF"),
.series_name = try allocator.dupe(u8, "Test Fund"),
.cik = try allocator.dupe(u8, "0000000002"),
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "edgar"),
} },
.{ .holding = .{
.symbol = try allocator.dupe(u8, "TESTF"),
.name = try allocator.dupe(u8, "Placeholder Corp"),
.cusip = try allocator.dupe(u8, "999999999"),
.pct_of_portfolio = 12.5,
.as_of = try allocator.dupe(u8, "2026-06-01"),
.source = try allocator.dupe(u8, "edgar"),
} },
};
defer for (etf_records) |r| r.deinit(allocator);
var s = svc.store();
s.write(Edgar.EtfMetricRecord, "TESTF", etf_records[0..], cache.DataType.etf_metrics.ttl());
svc.panic_on_network_attempt = true;
const result = try svc.getEtfProfile("TESTF", .{ .skip_network = true });
defer result.deinit();
const holdings = result.data.holdings orelse return error.NoHoldings;
try std.testing.expectEqual(@as(usize, 1), holdings.len);
try std.testing.expectEqualStrings("999999999", holdings[0].cusip orelse return error.NoCusip);
try std.testing.expect(holdings[0].symbol == null); // filing had no ticker
}
test "earningsNeedsRefresh: recent missing actual triggers a re-fetch" {
const today = Date.fromYmd(2026, 6, 26);
const events = [_]EarningsEvent{
.{ .date = Date.fromYmd(2026, 6, 20), .estimate = 1.0 }, // 6 days ago, actual not posted yet
};
try std.testing.expect(DataService.earningsNeedsRefresh(&events, today, 14));
}
test "earningsNeedsRefresh: stale missing actual is NOT chased (the SPY case)" {
const today = Date.fromYmd(2026, 6, 26);
const events = [_]EarningsEvent{
.{ .date = Date.fromYmd(2006, 5, 15), .estimate = 2.11 }, // ~20y old, FMP never backfills
.{ .date = Date.fromYmd(2005, 2, 15), .actual = 1.81 },
};
try std.testing.expect(!DataService.earningsNeedsRefresh(&events, today, 14));
}
test "earningsNeedsRefresh: all actuals present -> no re-fetch" {
const today = Date.fromYmd(2026, 6, 26);
const events = [_]EarningsEvent{
.{ .date = Date.fromYmd(2026, 5, 20), .actual = 1.5 },
.{ .date = Date.fromYmd(2026, 2, 20), .actual = 1.2 },
};
try std.testing.expect(!DataService.earningsNeedsRefresh(&events, today, 14));
}
test "earningsNeedsRefresh: upcoming event without actual does not trigger" {
const today = Date.fromYmd(2026, 6, 26);
const events = [_]EarningsEvent{
.{ .date = Date.fromYmd(2026, 8, 26), .estimate = 2.0 }, // future report, no actual yet
};
try std.testing.expect(!DataService.earningsNeedsRefresh(&events, today, 14));
}
test "earningsNeedsRefresh: chase window is inclusive at the boundary" {
const today = Date.fromYmd(2026, 6, 26);
// Exactly 14 days ago -> still chased.
const at_window = [_]EarningsEvent{.{ .date = Date.fromYmd(2026, 6, 12), .estimate = 1.0 }};
try std.testing.expect(DataService.earningsNeedsRefresh(&at_window, today, 14));
// 15 days ago -> past the window, left alone.
const past_window = [_]EarningsEvent{.{ .date = Date.fromYmd(2026, 6, 11), .estimate = 1.0 }};
try std.testing.expect(!DataService.earningsNeedsRefresh(&past_window, today, 14));
}
test "newestDateIn: picks the maximum across candles_meta and candles_daily shapes" {
// `candles_meta` carries one `last_date`; `candles_daily` carries a `date`
// per bar, in an order this check must not assume.
const meta = "#!srfv1\n#!expires=1\nlast_close:num:100.00,last_date::2026-08-07,provider::tiingo\n";
try std.testing.expect(DataService.newestDateIn(meta).?.eql(Date.fromYmd(2026, 8, 7)));
// Deliberately out of order.
const daily = "#!srfv1\ndate::2026-08-05,close:num:1\ndate::2026-08-07,close:num:3\ndate::2026-08-06,close:num:2\n";
try std.testing.expect(DataService.newestDateIn(daily).?.eql(Date.fromYmd(2026, 8, 7)));
// Nothing parseable -> null, so the guard cannot fire on garbage.
try std.testing.expectEqual(@as(?Date, null), DataService.newestDateIn("#!srfv1\n"));
try std.testing.expectEqual(@as(?Date, null), DataService.newestDateIn("last_date::not-a-date\n"));
}
test "isCandleType: only candle data is age-comparable" {
try std.testing.expect(DataService.isCandleType(.candles_daily));
try std.testing.expect(DataService.isCandleType(.candles_meta));
// Dividends, splits, options and the rest have no single newest-bar date,
// so they sync unguarded.
try std.testing.expect(!DataService.isCandleType(.dividends));
try std.testing.expect(!DataService.isCandleType(.splits));
try std.testing.expect(!DataService.isCandleType(.classification));
}
test "serverBarRegression: blocks a regression and reports both dates" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var s = cache.Store.init(io, allocator, dir_path);
// Local copy holds Monday's bar.
s.updateCandleMeta("AAPL", .{ .last_close = 100.0, .last_date = Date.fromYmd(2026, 8, 10), .provider = .tiingo }, 9999999999);
const older = "#!srfv1\nlast_close:num:99.00,last_date::2026-08-07,provider::tiingo\n";
const same = "#!srfv1\nlast_close:num:99.00,last_date::2026-08-10,provider::tiingo\n";
const newer = "#!srfv1\nlast_close:num:99.00,last_date::2026-08-11,provider::tiingo\n";
// THE REGRESSION THIS GUARDS. The shared cache's cron fetched at 17:00 ET;
// some symbols got that session's bar and some did not, and all were
// stamped fresh until the next boundary. Written unconditionally, the older
// body replaces good local data and is then believed for a full day.
const reg = DataService.serverBarRegression(&s, "AAPL", older) orelse
return error.ExpectedRegression;
// Both dates come back, because the warning is only actionable with them.
try std.testing.expect(reg.local.eql(Date.fromYmd(2026, 8, 10)));
try std.testing.expect(reg.incoming.eql(Date.fromYmd(2026, 8, 7)));
try std.testing.expectEqual(@as(?@TypeOf(reg), null), DataService.serverBarRegression(&s, "AAPL", same));
try std.testing.expectEqual(@as(?@TypeOf(reg), null), DataService.serverBarRegression(&s, "AAPL", newer));
// Unknowns never block: no local copy, and an unparseable body.
try std.testing.expectEqual(@as(?@TypeOf(reg), null), DataService.serverBarRegression(&s, "NOLOCAL", older));
try std.testing.expectEqual(@as(?@TypeOf(reg), null), DataService.serverBarRegression(&s, "AAPL", "#!srfv1\n"));
}
test "expiryAfterFetch: a partial catch-up keeps retrying instead of writing off the session" {
// THE REGRESSION THIS GUARDS, with Monday's real numbers. AMZN was fetched
// Mon 2026-08-10 at 17:00 ET holding Thursday's bar, and the incremental
// request returned FRIDAY's. One new candle - a success - so the old code
// stamped the next-boundary TTL and stopped looking until Tue 16:55, while
// Monday's bar showed up at the provider minutes later. Repeat daily and the
// symbol is permanently one session behind.
const kind = market.InstrumentKind.equity;
// Mon 2026-08-10 17:00 ET, five minutes past the 16:55 equity target.
const mon_1700 = Date.fromYmd(2026, 8, 10).toEpoch() + 21 * std.time.s_per_hour;
var friday_only = [_]Candle{.{ .date = Date.fromYmd(2026, 8, 7), .open = 1, .high = 1, .low = 1, .close = 1, .adj_close = 1, .volume = 1 }};
const partial = DataService.expiryAfterFetch(mon_1700, kind, &friday_only);
try std.testing.expectEqual(mon_1700 + market.short_retry_s, partial);
try std.testing.expect(partial != market.nextCandleExpiry(mon_1700, kind));
// And the case that must NOT change: a fetch that actually caught up gets
// the full next-boundary TTL exactly as before.
var through_monday = [_]Candle{
.{ .date = Date.fromYmd(2026, 8, 7), .open = 1, .high = 1, .low = 1, .close = 1, .adj_close = 1, .volume = 1 },
.{ .date = Date.fromYmd(2026, 8, 10), .open = 1, .high = 1, .low = 1, .close = 1, .adj_close = 1, .volume = 1 },
};
try std.testing.expectEqual(
market.nextCandleExpiry(mon_1700, kind),
DataService.expiryAfterFetch(mon_1700, kind, &through_monday),
);
}
test "expiryAfterFetch: takes the maximum, not the last element" {
// Provider ordering is not something this decision should depend on.
const kind = market.InstrumentKind.equity;
const mon_1700 = Date.fromYmd(2026, 8, 10).toEpoch() + 21 * std.time.s_per_hour;
var descending = [_]Candle{
.{ .date = Date.fromYmd(2026, 8, 10), .open = 1, .high = 1, .low = 1, .close = 1, .adj_close = 1, .volume = 1 },
.{ .date = Date.fromYmd(2026, 8, 7), .open = 1, .high = 1, .low = 1, .close = 1, .adj_close = 1, .volume = 1 },
};
// Newest is 08-10 even though it is first, so this is caught up.
try std.testing.expectEqual(
market.nextCandleExpiry(mon_1700, kind),
DataService.expiryAfterFetch(mon_1700, kind, &descending),
);
}
test "expiryAfterFetch: an empty slice falls back to the next boundary" {
// Both call sites are guarded by a non-empty check; this keeps the helper
// safe if a third one is ever added.
const kind = market.InstrumentKind.equity;
const mon_1700 = Date.fromYmd(2026, 8, 10).toEpoch() + 21 * std.time.s_per_hour;
try std.testing.expectEqual(
market.nextCandleExpiry(mon_1700, kind),
DataService.expiryAfterFetch(mon_1700, kind, &.{}),
);
}
test "effectiveOptions: the default policy is a no-op, which is what keeps tests offline" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var svc = DataService.init(io, allocator, .{ .cache_dir = "unused" });
defer svc.deinit();
// Nothing set: identical in, identical out. Test Configs are keyless so a
// provider fetch already fails with NoApiKey before any HTTP, and this field
// must not be able to loosen that by accident.
try std.testing.expectEqual(FetchOptions{}, svc.effectiveOptions(.{}));
try std.testing.expectEqual(
FetchOptions{ .force_refresh = true },
svc.effectiveOptions(.{ .force_refresh = true }),
);
try std.testing.expectEqual(
FetchOptions{ .skip_network = true },
svc.effectiveOptions(.{ .skip_network = true }),
);
}
test "effectiveOptions: the service policy reaches a caller with no opinion" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var svc = DataService.init(io, allocator, .{ .cache_dir = "unused" });
defer svc.deinit();
// THE BUG THIS FIXES. `views/projections.zig` fetches the benchmark pair
// with a hardcoded `.{}`, so `--refresh-data=force` never reached SPY/AGG.
svc.default_options = .{ .force_refresh = true };
try std.testing.expect(svc.effectiveOptions(.{}).force_refresh);
// `--refresh-data=never` likewise, so an offline session stays offline even
// through a call site that never threaded the option.
svc.default_options = .{ .skip_network = true };
try std.testing.expect(svc.effectiveOptions(.{}).skip_network);
try std.testing.expect(!svc.effectiveOptions(.{}).force_refresh);
}
test "effectiveOptions: a union, so an explicit request survives .auto" {
const allocator = std.testing.allocator;
const io = std.testing.io;
var svc = DataService.init(io, allocator, .{ .cache_dir = "unused" });
defer svc.deinit();
// Under `.auto` a caller that explicitly asks for force still gets it -
// `zfin cache refresh` depends on exactly this.
svc.default_options = .{};
try std.testing.expect(svc.effectiveOptions(.{ .force_refresh = true }).force_refresh);
// Both set: skip_network continues to win downstream, which is the existing
// documented precedence - this only unions the flags, it does not reorder them.
svc.default_options = .{ .skip_network = true };
const both = svc.effectiveOptions(.{ .force_refresh = true });
try std.testing.expect(both.force_refresh and both.skip_network);
}
test "effectiveOptions: --refresh-data=never survives a hardcoded call site" {
// The offline guarantee, end to end: with the service policy set to never,
// a fetch through a call site that passed `.{}` must not attempt network.
// `panic_on_network_attempt` turns any attempt into a hard failure.
const allocator = std.testing.allocator;
const io = std.testing.io;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
const dir_path = try tmp.dir.realPathFileAlloc(io, ".", allocator);
defer allocator.free(dir_path);
var svc = DataService.init(io, allocator, .{ .cache_dir = dir_path });
defer svc.deinit();
svc.default_options = .{ .skip_network = true };
svc.panic_on_network_attempt = true;
// Nothing cached and no network permitted -> a clean error, not a panic and
// not a request.
try std.testing.expectError(DataError.FetchFailed, svc.getCandles("NOSUCH", .{}));
}