Spaces:
Sleeping
Sleeping
File size: 5,898 Bytes
5b9b478 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 | """Backfill earnings-call transcripts that ``ingest.py`` can no longer reach.
Two gaps leave a stored period without its transcript, and neither is repaired
by a normal delta run:
1. **The filing aged out of EDGAR's "recent" index.** ``ingest.py`` walks
``submissions/CIK.json``'s ``filings.recent`` array to decide which periods
to process. A heavy filer (Alphabet files hundreds of Form 4s a year) pushes
its own older 10-Qs out of that array into the paginated archive, so the
period stays in ``metrics.db`` but is never revisited β and its missing
transcript is never fetched.
2. **A stale "unavailable" marker.** ``sections.db`` records an empty string to
mean "asked the provider, it had nothing", which correctly stops a delta run
from re-asking every time. But providers backfill their own archives, so a
marker written months ago can be wrong today.
This script targets exactly the periods that are missing a transcript, asks the
provider again, and stores anything it gets through the same code path
ingestion uses (``sections.db`` plus the Chroma transcript collection).
Usage::
python scripts/backfill_transcripts.py # report only
python scripts/backfill_transcripts.py --apply # every ticker
python scripts/backfill_transcripts.py --apply GOOGL TSLA
"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from dotenv import load_dotenv
load_dotenv()
from ingestion.embedder import embed_and_store_transcript
from ingestion.transcript import fetch_transcript
from storage.metrics_db import get_all_metrics
from storage.sections_db import get_section, init_sections_db, upsert_section
def _period_to_av_quarter(period: str) -> str:
"""EDGAR period β provider quarter. 'Q12026' β '2026Q1'; 'FY2024' β '2024Q4'."""
period = (period or "").strip()
if len(period) < 6:
return ""
if period.startswith("FY"):
year = period[2:]
return f"{year}Q4" if len(year) == 4 and year.isdigit() else ""
if period[0] == "Q" and period[1] in "1234":
year = period[2:]
return f"{year}{period[:2]}" if len(year) == 4 and year.isdigit() else ""
return ""
def _known_tickers() -> list[str]:
import sqlite3
from storage.metrics_db import DB_PATH
if not DB_PATH.exists():
return []
with sqlite3.connect(DB_PATH) as conn:
return [row[0] for row in conn.execute(
"SELECT DISTINCT ticker FROM metrics ORDER BY ticker"
)]
def find_gaps(ticker: str) -> list[dict]:
"""Stored periods whose transcript is absent or recorded as unavailable."""
gaps: list[dict] = []
for row in get_all_metrics(ticker) or []:
period = str(row.get("period") or "")
form_type = str(row.get("form_type") or "")
if not period or form_type not in ("10-Q", "10-K"):
continue
stored = get_section(ticker, period, "transcript")
if stored:
continue
gaps.append({
"ticker": ticker,
"period": period,
"form_type": form_type,
"company_name": str(row.get("company_name") or ticker),
"filing_date": str(row.get("filing_date") or ""),
# None = never asked; "" = asked, provider had nothing at the time.
"reason": "never_attempted" if stored is None else "marked_unavailable",
})
return gaps
def backfill(ticker: str, apply: bool) -> tuple[int, int]:
"""Return (gaps found, transcripts stored)."""
gaps = find_gaps(ticker)
if not gaps:
print(f"[{ticker}] complete β every stored period has a transcript.")
return 0, 0
stored_count = 0
for gap in gaps:
quarter = _period_to_av_quarter(gap["period"])
if not quarter:
print(f"[{ticker}] {gap['period']}: unparseable period, skipped.")
continue
if not apply:
print(f"[{ticker}] {gap['period']} ({quarter}) β missing ({gap['reason']})")
continue
try:
text = fetch_transcript(ticker, quarter) or ""
except Exception as exc:
print(f"[{ticker}] {gap['period']}: provider error β {exc}")
continue
# Writing the empty string back is deliberate: it records that the
# provider was asked and had nothing, which is what stops a delta run
# from re-asking on every ingestion.
upsert_section(ticker, gap["period"], gap["form_type"], "transcript", text)
if not text:
print(f"[{ticker}] {gap['period']} ({quarter}): provider still has nothing.")
continue
embed_and_store_transcript(
ticker=ticker,
company_name=gap["company_name"],
transcript_text=text,
transcript_date=gap["filing_date"],
period=gap["period"],
source_url="",
provider="alphavantage",
)
stored_count += 1
print(f"[{ticker}] {gap['period']} ({quarter}): stored {len(text):,} chars.")
return len(gaps), stored_count
def main(argv: list[str]) -> int:
apply = "--apply" in argv
tickers = [a.upper() for a in argv if not a.startswith("--")] or _known_tickers()
if not tickers:
print("No ingested tickers found. Run ingest.py first.")
return 1
init_sections_db()
total_gaps = total_stored = 0
for ticker in tickers:
gaps, stored = backfill(ticker, apply)
total_gaps += gaps
total_stored += stored
if apply:
print(f"\nDone. {total_stored} transcript(s) stored across {len(tickers)} ticker(s).")
else:
print(f"\n{total_gaps} gap(s) found. Re-run with --apply to fetch them.")
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))
|