Spaces:
Sleeping
Sleeping
Download scripts/data_foundation_report.py from Viney/readthrough: direct link, hf CLI and curl.
- Browser
- Download file 5.09 kB
-
https://huggingface.co/spaces/Viney/readthrough/resolve/main/scripts/data_foundation_report.py
- Command line
-
hf download hf://spaces/Viney/readthrough/scripts/data_foundation_report.py
-
curl -L -o data_foundation_report.py https://huggingface.co/spaces/Viney/readthrough/resolve/main/scripts/data_foundation_report.py
5.09 kB
| """The data-foundation acceptance criteria, as queries rather than assertions. | |
| Spec: docs/superpowers/specs/2026-09-01-data-foundation-design.md | |
| Three criteria deliberately permit a shortfall provided it is enumerated. A | |
| filer that does not tag a concept is a fact about the filer, and recording it | |
| is the correct outcome; reporting a quietly lower number is not. So every | |
| criterion that can fall short prints the exact ticker, period and concept. | |
| Usage:: | |
| python -m scripts.data_foundation_report | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import sqlite3 | |
| import sys | |
| DB_PATH = "data/metrics.db" | |
| NEW_STATEMENT_LINES = ( | |
| "deferred_revenue", "research_development", "sga", "goodwill", | |
| "intangibles", "cash_and_st_investments", "debt_current", | |
| "operating_lease_liabilities", | |
| ) | |
| CORE_STATEMENT_LINES = ( | |
| "operating_cash_flow", "net_income", "total_assets", | |
| "receivables", "payables", "cogs", | |
| ) | |
| GUIDANCE_FLOOR = 40 | |
| def _rows() -> list[dict]: | |
| conn = sqlite3.connect(DB_PATH) | |
| conn.row_factory = sqlite3.Row | |
| try: | |
| return [dict(r) for r in conn.execute("SELECT * FROM metrics")] | |
| finally: | |
| conn.close() | |
| def _segment_tickers() -> set[str]: | |
| conn = sqlite3.connect(DB_PATH) | |
| try: | |
| return {r[0] for r in conn.execute("SELECT DISTINCT ticker FROM segments")} | |
| except sqlite3.OperationalError: | |
| return set() | |
| finally: | |
| conn.close() | |
| def build_report() -> list[dict]: | |
| rows = _rows() | |
| total = len(rows) | |
| out: list[dict] = [] | |
| legacy = [f"{r['ticker']} {r['period']}" for r in rows | |
| if r.get("data_quality_status") == "LEGACY_UNVERIFIED"] | |
| out.append({ | |
| "criterion": "no_legacy_unverified_rows", "target": "0", | |
| "actual": str(len(legacy)), "passed": not legacy, "detail": legacy, | |
| }) | |
| no_url = [f"{r['ticker']} {r['period']}" for r in rows if not r.get("source_url")] | |
| out.append({ | |
| "criterion": "every_row_carries_source_url", "target": "0 missing", | |
| "actual": str(len(no_url)), "passed": not no_url, "detail": no_url, | |
| }) | |
| parsed = [r for r in rows if r.get("guidance_status") == "PARSED"] | |
| silent = [f"{r['ticker']} {r['period']}" for r in rows | |
| if r.get("guidance_status") == "NO_GUIDANCE_ISSUED"] | |
| errored = [f"{r['ticker']} {r['period']}" for r in rows | |
| if r.get("guidance_status") == "PARSE_ERROR"] | |
| unquoted = [f"{r['ticker']} {r['period']}" for r in parsed | |
| if not r.get("guidance_source_quote")] | |
| out.append({ | |
| "criterion": "guidance_parsed_above_floor", "target": f">= {GUIDANCE_FLOOR}", | |
| "actual": str(len(parsed)), "passed": len(parsed) >= GUIDANCE_FLOOR, | |
| "detail": [f"no guidance issued: {len(silent)}", f"parse errors: {len(errored)}"] | |
| + errored, | |
| }) | |
| out.append({ | |
| "criterion": "every_parsed_guidance_has_a_quote", "target": "0 unquoted", | |
| "actual": str(len(unquoted)), "passed": not unquoted, "detail": unquoted, | |
| }) | |
| core_gaps = [ | |
| f"{r['ticker']} {r['period']} {line}" | |
| for r in rows for line in CORE_STATEMENT_LINES if r.get(line) is None | |
| ] | |
| out.append({ | |
| "criterion": "core_statement_lines_populated", "target": f"{total} rows each", | |
| "actual": f"{len(core_gaps)} gaps", "passed": not core_gaps, "detail": core_gaps, | |
| }) | |
| new_gaps = [ | |
| f"{r['ticker']} {r['period']} {line}" | |
| for r in rows for line in NEW_STATEMENT_LINES if r.get(line) is None | |
| ] | |
| out.append({ | |
| "criterion": "new_statement_lines_populated", | |
| "target": "populated wherever the filer tags the concept", | |
| "actual": f"{len(new_gaps)} absences", "passed": True, "detail": new_gaps, | |
| }) | |
| fallback = [] | |
| for r in rows: | |
| try: | |
| warnings = json.loads(r.get("quality_warnings") or "[]") | |
| except (TypeError, ValueError): | |
| warnings = [] | |
| for w in warnings: | |
| if "fallback_non_sec" in w: | |
| fallback.append(f"{r['ticker']} {r['period']} {w}") | |
| out.append({ | |
| "criterion": "no_fallback_non_sec", "target": "0", | |
| "actual": str(len(fallback)), "passed": not fallback, "detail": fallback, | |
| }) | |
| covered = {r["ticker"] for r in rows} | |
| with_segments = _segment_tickers() | |
| missing = sorted(covered - with_segments) | |
| out.append({ | |
| "criterion": "segments_on_every_covered_ticker", | |
| "target": f"{len(covered)} tickers", | |
| "actual": str(len(with_segments)), "passed": not missing, "detail": missing, | |
| }) | |
| return out | |
| def main() -> None: | |
| report = build_report() | |
| for r in report: | |
| mark = "PASS" if r["passed"] else "FAIL" | |
| print(f"[{mark}] {r['criterion']}: {r['actual']} (target {r['target']})") | |
| for line in r["detail"][:40]: | |
| print(f" {line}") | |
| if len(r["detail"]) > 40: | |
| print(f" ... and {len(r['detail']) - 40} more") | |
| failed = [r for r in report if not r["passed"]] | |
| sys.exit(1 if failed else 0) | |
| if __name__ == "__main__": | |
| main() | |