"""Backfill per-product / per-region results for filings already ingested. ``ingest.py`` now pulls the segment detail as it goes, but every filing ingested before that lives in ``metrics.db`` with only its consolidated figures. This walks those rows, fetches each filing's XBRL instance once and fills the ``segments`` table. One HTTP request per filing, ~1 MB each, so it is rate-limited to stay inside the SEC's fair-access policy. Usage:: python -m scripts.backfill_segments AAPL python -m scripts.backfill_segments --all python -m scripts.backfill_segments AAPL --dry-run """ from __future__ import annotations import argparse import sys import time from ingestion.segments import PRODUCT_AXIS, SEGMENT_AXIS, fetch_segments from storage import metrics_db, segments_db # The SEC asks for no more than 10 requests a second; one filing per second is # far inside that and keeps a full backfill polite. _PAUSE_SECONDS = 1.0 def _cik_and_doc(source_url: str) -> tuple[str, str]: """Pull the CIK and primary document out of a stored filing URL.""" try: cik = source_url.split("/data/")[1].split("/")[0] except (IndexError, AttributeError): return "", "" return cik, source_url.rsplit("/", 1)[-1] def backfill(ticker: str, *, dry_run: bool = False) -> int: rows = metrics_db.get_all_metrics(ticker.upper()) filings = [ r for r in rows if r.get("accession") and r.get("source_url") and r.get("period_basis") != "annual" ] if not filings: print(f"{ticker}: no ingested filings carrying an accession") return 1 total = failures = 0 for row in filings: cik, primary_doc = _cik_and_doc(row["source_url"]) if not cik: print(f" {row['period']}: unparseable source_url, skipped") failures += 1 continue try: facts, url = fetch_segments(cik, row["accession"], primary_doc) except Exception as exc: # noqa: BLE001 - one bad filing must not stop the run print(f" {row['period']}: fetch failed ({exc})") failures += 1 continue products = sum(1 for f in facts if f.axis == PRODUCT_AXIS) regions = sum(1 for f in facts if f.axis == SEGMENT_AXIS) print(f" {row['period']}: {len(facts)} line(s) — {products} product, {regions} region") if not dry_run: total += segments_db.upsert_segments( ticker, facts, row["accession"], url ) time.sleep(_PAUSE_SECONDS) if dry_run: print(f"{ticker}: dry run, nothing written") else: print(f"{ticker}: {total} row(s) written") return 1 if failures and not total else 0 def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("ticker", nargs="?", help="Ticker to backfill.") parser.add_argument("--all", action="store_true", help="Every ingested ticker.") parser.add_argument("--dry-run", action="store_true", help="Fetch and report without writing.") args = parser.parse_args() if args.all: tickers = metrics_db.list_tickers() elif args.ticker: tickers = [args.ticker.upper()] else: parser.error("give a ticker or --all") failures = 0 for ticker in tickers: print(f"\n== {ticker} ==") failures += backfill(ticker, dry_run=args.dry_run) return 1 if failures else 0 if __name__ == "__main__": raise SystemExit(main())