"""The data-foundation acceptance criteria, as queries rather than assertions. Spec: docs/superpowers/specs/2026-09-01-data-foundation-design.md Three criteria deliberately permit a shortfall provided it is enumerated. A filer that does not tag a concept is a fact about the filer, and recording it is the correct outcome; reporting a quietly lower number is not. So every criterion that can fall short prints the exact ticker, period and concept. Usage:: python -m scripts.data_foundation_report """ from __future__ import annotations import json import sqlite3 import sys DB_PATH = "data/metrics.db" NEW_STATEMENT_LINES = ( "deferred_revenue", "research_development", "sga", "goodwill", "intangibles", "cash_and_st_investments", "debt_current", "operating_lease_liabilities", ) CORE_STATEMENT_LINES = ( "operating_cash_flow", "net_income", "total_assets", "receivables", "payables", "cogs", ) GUIDANCE_FLOOR = 40 def _rows() -> list[dict]: conn = sqlite3.connect(DB_PATH) conn.row_factory = sqlite3.Row try: return [dict(r) for r in conn.execute("SELECT * FROM metrics")] finally: conn.close() def _segment_tickers() -> set[str]: conn = sqlite3.connect(DB_PATH) try: return {r[0] for r in conn.execute("SELECT DISTINCT ticker FROM segments")} except sqlite3.OperationalError: return set() finally: conn.close() def build_report() -> list[dict]: rows = _rows() total = len(rows) out: list[dict] = [] legacy = [f"{r['ticker']} {r['period']}" for r in rows if r.get("data_quality_status") == "LEGACY_UNVERIFIED"] out.append({ "criterion": "no_legacy_unverified_rows", "target": "0", "actual": str(len(legacy)), "passed": not legacy, "detail": legacy, }) no_url = [f"{r['ticker']} {r['period']}" for r in rows if not r.get("source_url")] out.append({ "criterion": "every_row_carries_source_url", "target": "0 missing", "actual": str(len(no_url)), "passed": not no_url, "detail": no_url, }) parsed = [r for r in rows if r.get("guidance_status") == "PARSED"] silent = [f"{r['ticker']} {r['period']}" for r in rows if r.get("guidance_status") == "NO_GUIDANCE_ISSUED"] errored = [f"{r['ticker']} {r['period']}" for r in rows if r.get("guidance_status") == "PARSE_ERROR"] unquoted = [f"{r['ticker']} {r['period']}" for r in parsed if not r.get("guidance_source_quote")] out.append({ "criterion": "guidance_parsed_above_floor", "target": f">= {GUIDANCE_FLOOR}", "actual": str(len(parsed)), "passed": len(parsed) >= GUIDANCE_FLOOR, "detail": [f"no guidance issued: {len(silent)}", f"parse errors: {len(errored)}"] + errored, }) out.append({ "criterion": "every_parsed_guidance_has_a_quote", "target": "0 unquoted", "actual": str(len(unquoted)), "passed": not unquoted, "detail": unquoted, }) core_gaps = [ f"{r['ticker']} {r['period']} {line}" for r in rows for line in CORE_STATEMENT_LINES if r.get(line) is None ] out.append({ "criterion": "core_statement_lines_populated", "target": f"{total} rows each", "actual": f"{len(core_gaps)} gaps", "passed": not core_gaps, "detail": core_gaps, }) new_gaps = [ f"{r['ticker']} {r['period']} {line}" for r in rows for line in NEW_STATEMENT_LINES if r.get(line) is None ] out.append({ "criterion": "new_statement_lines_populated", "target": "populated wherever the filer tags the concept", "actual": f"{len(new_gaps)} absences", "passed": True, "detail": new_gaps, }) fallback = [] for r in rows: try: warnings = json.loads(r.get("quality_warnings") or "[]") except (TypeError, ValueError): warnings = [] for w in warnings: if "fallback_non_sec" in w: fallback.append(f"{r['ticker']} {r['period']} {w}") out.append({ "criterion": "no_fallback_non_sec", "target": "0", "actual": str(len(fallback)), "passed": not fallback, "detail": fallback, }) covered = {r["ticker"] for r in rows} with_segments = _segment_tickers() missing = sorted(covered - with_segments) out.append({ "criterion": "segments_on_every_covered_ticker", "target": f"{len(covered)} tickers", "actual": str(len(with_segments)), "passed": not missing, "detail": missing, }) return out def main() -> None: report = build_report() for r in report: mark = "PASS" if r["passed"] else "FAIL" print(f"[{mark}] {r['criterion']}: {r['actual']} (target {r['target']})") for line in r["detail"][:40]: print(f" {line}") if len(r["detail"]) > 40: print(f" ... and {len(r['detail']) - 40} more") failed = [r for r in report if not r["passed"]] sys.exit(1 if failed else 0) if __name__ == "__main__": main()