readthrough / scripts /backfill_segments.py
Viney's picture
Claude Opus 5 (1M context)
feat: read per-product and per-region results from the filing's XBRL instance
232b6f9
Raw History Blame Contribute Delete
3.51 kB
"""Backfill per-product / per-region results for filings already ingested.
``ingest.py`` now pulls the segment detail as it goes, but every filing
ingested before that lives in ``metrics.db`` with only its consolidated
figures. This walks those rows, fetches each filing's XBRL instance once and
fills the ``segments`` table.
One HTTP request per filing, ~1 MB each, so it is rate-limited to stay inside
the SEC's fair-access policy.
Usage::
python -m scripts.backfill_segments AAPL
python -m scripts.backfill_segments --all
python -m scripts.backfill_segments AAPL --dry-run
"""
from __future__ import annotations
import argparse
import sys
import time
from ingestion.segments import PRODUCT_AXIS, SEGMENT_AXIS, fetch_segments
from storage import metrics_db, segments_db
# The SEC asks for no more than 10 requests a second; one filing per second is
# far inside that and keeps a full backfill polite.
_PAUSE_SECONDS = 1.0
def _cik_and_doc(source_url: str) -> tuple[str, str]:
"""Pull the CIK and primary document out of a stored filing URL."""
try:
cik = source_url.split("/data/")[1].split("/")[0]
except (IndexError, AttributeError):
return "", ""
return cik, source_url.rsplit("/", 1)[-1]
def backfill(ticker: str, *, dry_run: bool = False) -> int:
rows = metrics_db.get_all_metrics(ticker.upper())
filings = [
r for r in rows
if r.get("accession") and r.get("source_url") and r.get("period_basis") != "annual"
]
if not filings:
print(f"{ticker}: no ingested filings carrying an accession")
return 1
total = failures = 0
for row in filings:
cik, primary_doc = _cik_and_doc(row["source_url"])
if not cik:
print(f" {row['period']}: unparseable source_url, skipped")
failures += 1
continue
try:
facts, url = fetch_segments(cik, row["accession"], primary_doc)
except Exception as exc: # noqa: BLE001 - one bad filing must not stop the run
print(f" {row['period']}: fetch failed ({exc})")
failures += 1
continue
products = sum(1 for f in facts if f.axis == PRODUCT_AXIS)
regions = sum(1 for f in facts if f.axis == SEGMENT_AXIS)
print(f" {row['period']}: {len(facts)} line(s) — {products} product, {regions} region")
if not dry_run:
total += segments_db.upsert_segments(
ticker, facts, row["accession"], url
)
time.sleep(_PAUSE_SECONDS)
if dry_run:
print(f"{ticker}: dry run, nothing written")
else:
print(f"{ticker}: {total} row(s) written")
return 1 if failures and not total else 0
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("ticker", nargs="?", help="Ticker to backfill.")
parser.add_argument("--all", action="store_true", help="Every ingested ticker.")
parser.add_argument("--dry-run", action="store_true", help="Fetch and report without writing.")
args = parser.parse_args()
if args.all:
tickers = metrics_db.list_tickers()
elif args.ticker:
tickers = [args.ticker.upper()]
else:
parser.error("give a ticker or --all")
failures = 0
for ticker in tickers:
print(f"\n== {ticker} ==")
failures += backfill(ticker, dry_run=args.dry_run)
return 1 if failures else 0
if __name__ == "__main__":
raise SystemExit(main())