"""Backfill the supplementary statement lines for filings already ingested. ``ingest.py`` now stores operating cash flow, net income, total assets, receivables, inventory, payables, cost of revenue, share-based compensation and depreciation alongside the headline metrics. Every filing ingested before that has those columns empty, and a full re-ingest would re-download the filing text, re-chunk it and re-embed it to obtain nine numbers that are already in a file we can fetch once per company. This walks the stored rows instead: one ``companyfacts`` request per ticker, then a recompute per accession. It writes **only** the new columns. The headline metrics, their selection contexts and their quality status are left exactly as ingested -- a backfill that silently re-derived a verified revenue figure would be indistinguishable from a corruption. Usage:: python -m scripts.backfill_statement_lines AAPL python -m scripts.backfill_statement_lines --all python -m scripts.backfill_statement_lines AAPL --dry-run """ from __future__ import annotations import argparse import sqlite3 import time from ingestion.edgar import STATEMENT_LINES, compute_metrics_for_accn, get_all_xbrl_facts, get_cik from storage import metrics_db # One companyfacts document per ticker, so the pause is per company rather # than per filing. Well inside the SEC's ten-requests-a-second policy. _PAUSE_SECONDS = 0.5 _UPDATE_SQL = ( "UPDATE metrics SET " + ", ".join(f"{name}=:{name}" for name in STATEMENT_LINES) + " WHERE ticker=:ticker AND period=:period" ) def _write(values: dict) -> None: with sqlite3.connect(metrics_db.DB_PATH) as conn: conn.execute(_UPDATE_SQL, values) def backfill(ticker: str, *, dry_run: bool = False) -> int: ticker = ticker.upper() rows = [r for r in metrics_db.get_all_metrics(ticker) if r.get("accession")] if not rows: print(f"{ticker}: no ingested filing carries an accession") return 1 cik = get_cik(ticker) if not cik: print(f"{ticker}: no CIK") return 1 facts = get_all_xbrl_facts(cik) filled = 0 for row in rows: metrics = compute_metrics_for_accn( facts, row["accession"], row.get("form_type") or "10-Q", report_date=row.get("report_date") or "", filing_date=row.get("filing_date") or "", ) values = {name: metrics.get(name) for name in STATEMENT_LINES} found = [name for name, value in values.items() if value is not None] print(f" {row['period']}: {len(found)}/{len(STATEMENT_LINES)} line(s)" + (f" — missing {', '.join(n for n in STATEMENT_LINES if n not in found)}" if len(found) < len(STATEMENT_LINES) else "")) if not dry_run and found: _write({**values, "ticker": ticker, "period": row["period"]}) filled += 1 print(f"{ticker}: {'dry run, nothing written' if dry_run else f'{filled} row(s) updated'}") return 0 def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("ticker", nargs="?", help="Ticker to backfill.") parser.add_argument("--all", action="store_true", help="Every ingested ticker.") parser.add_argument("--dry-run", action="store_true", help="Fetch and report without writing.") args = parser.parse_args() if args.all: tickers = metrics_db.list_tickers() elif args.ticker: tickers = [args.ticker.upper()] else: parser.error("give a ticker or --all") metrics_db.init_db() failures = 0 for ticker in tickers: print(f"\n== {ticker} ==") failures += backfill(ticker, dry_run=args.dry_run) time.sleep(_PAUSE_SECONDS) return 1 if failures else 0 if __name__ == "__main__": raise SystemExit(main())