"""Backfill earnings-call transcripts that ``ingest.py`` can no longer reach. Two gaps leave a stored period without its transcript, and neither is repaired by a normal delta run: 1. **The filing aged out of EDGAR's "recent" index.** ``ingest.py`` walks ``submissions/CIK.json``'s ``filings.recent`` array to decide which periods to process. A heavy filer (Alphabet files hundreds of Form 4s a year) pushes its own older 10-Qs out of that array into the paginated archive, so the period stays in ``metrics.db`` but is never revisited — and its missing transcript is never fetched. 2. **A stale "unavailable" marker.** ``sections.db`` records an empty string to mean "asked the provider, it had nothing", which correctly stops a delta run from re-asking every time. But providers backfill their own archives, so a marker written months ago can be wrong today. This script targets exactly the periods that are missing a transcript, asks the provider again, and stores anything it gets through the same code path ingestion uses (``sections.db`` plus the Chroma transcript collection). Usage:: python scripts/backfill_transcripts.py # report only python scripts/backfill_transcripts.py --apply # every ticker python scripts/backfill_transcripts.py --apply GOOGL TSLA """ from __future__ import annotations import sys from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[1])) from dotenv import load_dotenv load_dotenv() from ingestion.embedder import embed_and_store_transcript from ingestion.transcript import fetch_transcript from storage.metrics_db import get_all_metrics from storage.sections_db import get_section, init_sections_db, upsert_section def _period_to_av_quarter(period: str) -> str: """EDGAR period → provider quarter. 'Q12026' → '2026Q1'; 'FY2024' → '2024Q4'.""" period = (period or "").strip() if len(period) < 6: return "" if period.startswith("FY"): year = period[2:] return f"{year}Q4" if len(year) == 4 and year.isdigit() else "" if period[0] == "Q" and period[1] in "1234": year = period[2:] return f"{year}{period[:2]}" if len(year) == 4 and year.isdigit() else "" return "" def _known_tickers() -> list[str]: import sqlite3 from storage.metrics_db import DB_PATH if not DB_PATH.exists(): return [] with sqlite3.connect(DB_PATH) as conn: return [row[0] for row in conn.execute( "SELECT DISTINCT ticker FROM metrics ORDER BY ticker" )] def find_gaps(ticker: str) -> list[dict]: """Stored periods whose transcript is absent or recorded as unavailable.""" gaps: list[dict] = [] for row in get_all_metrics(ticker) or []: period = str(row.get("period") or "") form_type = str(row.get("form_type") or "") if not period or form_type not in ("10-Q", "10-K"): continue stored = get_section(ticker, period, "transcript") if stored: continue gaps.append({ "ticker": ticker, "period": period, "form_type": form_type, "company_name": str(row.get("company_name") or ticker), "filing_date": str(row.get("filing_date") or ""), # None = never asked; "" = asked, provider had nothing at the time. "reason": "never_attempted" if stored is None else "marked_unavailable", }) return gaps def backfill(ticker: str, apply: bool) -> tuple[int, int]: """Return (gaps found, transcripts stored).""" gaps = find_gaps(ticker) if not gaps: print(f"[{ticker}] complete — every stored period has a transcript.") return 0, 0 stored_count = 0 for gap in gaps: quarter = _period_to_av_quarter(gap["period"]) if not quarter: print(f"[{ticker}] {gap['period']}: unparseable period, skipped.") continue if not apply: print(f"[{ticker}] {gap['period']} ({quarter}) — missing ({gap['reason']})") continue try: text = fetch_transcript(ticker, quarter) or "" except Exception as exc: print(f"[{ticker}] {gap['period']}: provider error — {exc}") continue # Writing the empty string back is deliberate: it records that the # provider was asked and had nothing, which is what stops a delta run # from re-asking on every ingestion. upsert_section(ticker, gap["period"], gap["form_type"], "transcript", text) if not text: print(f"[{ticker}] {gap['period']} ({quarter}): provider still has nothing.") continue embed_and_store_transcript( ticker=ticker, company_name=gap["company_name"], transcript_text=text, transcript_date=gap["filing_date"], period=gap["period"], source_url="", provider="alphavantage", ) stored_count += 1 print(f"[{ticker}] {gap['period']} ({quarter}): stored {len(text):,} chars.") return len(gaps), stored_count def main(argv: list[str]) -> int: apply = "--apply" in argv tickers = [a.upper() for a in argv if not a.startswith("--")] or _known_tickers() if not tickers: print("No ingested tickers found. Run ingest.py first.") return 1 init_sections_db() total_gaps = total_stored = 0 for ticker in tickers: gaps, stored = backfill(ticker, apply) total_gaps += gaps total_stored += stored if apply: print(f"\nDone. {total_stored} transcript(s) stored across {len(tickers)} ticker(s).") else: print(f"\n{total_gaps} gap(s) found. Re-run with --apply to fetch them.") return 0 if __name__ == "__main__": sys.exit(main(sys.argv[1:]))