readthrough / scripts /backfill_statement_lines.py
Viney's picture
Claude Opus 5 (1M context)
feat: decompose the EPS headline and read the statements behind it
bb3ad6b
Raw History Blame Contribute Delete
3.86 kB
"""Backfill the supplementary statement lines for filings already ingested.
``ingest.py`` now stores operating cash flow, net income, total assets,
receivables, inventory, payables, cost of revenue, share-based compensation
and depreciation alongside the headline metrics. Every filing ingested before
that has those columns empty, and a full re-ingest would re-download the
filing text, re-chunk it and re-embed it to obtain nine numbers that are
already in a file we can fetch once per company.
This walks the stored rows instead: one ``companyfacts`` request per ticker,
then a recompute per accession. It writes **only** the new columns. The
headline metrics, their selection contexts and their quality status are left
exactly as ingested -- a backfill that silently re-derived a verified revenue
figure would be indistinguishable from a corruption.
Usage::
python -m scripts.backfill_statement_lines AAPL
python -m scripts.backfill_statement_lines --all
python -m scripts.backfill_statement_lines AAPL --dry-run
"""
from __future__ import annotations
import argparse
import sqlite3
import time
from ingestion.edgar import STATEMENT_LINES, compute_metrics_for_accn, get_all_xbrl_facts, get_cik
from storage import metrics_db
# One companyfacts document per ticker, so the pause is per company rather
# than per filing. Well inside the SEC's ten-requests-a-second policy.
_PAUSE_SECONDS = 0.5
_UPDATE_SQL = (
"UPDATE metrics SET "
+ ", ".join(f"{name}=:{name}" for name in STATEMENT_LINES)
+ " WHERE ticker=:ticker AND period=:period"
)
def _write(values: dict) -> None:
with sqlite3.connect(metrics_db.DB_PATH) as conn:
conn.execute(_UPDATE_SQL, values)
def backfill(ticker: str, *, dry_run: bool = False) -> int:
ticker = ticker.upper()
rows = [r for r in metrics_db.get_all_metrics(ticker) if r.get("accession")]
if not rows:
print(f"{ticker}: no ingested filing carries an accession")
return 1
cik = get_cik(ticker)
if not cik:
print(f"{ticker}: no CIK")
return 1
facts = get_all_xbrl_facts(cik)
filled = 0
for row in rows:
metrics = compute_metrics_for_accn(
facts,
row["accession"],
row.get("form_type") or "10-Q",
report_date=row.get("report_date") or "",
filing_date=row.get("filing_date") or "",
)
values = {name: metrics.get(name) for name in STATEMENT_LINES}
found = [name for name, value in values.items() if value is not None]
print(f" {row['period']}: {len(found)}/{len(STATEMENT_LINES)} line(s)"
+ (f" — missing {', '.join(n for n in STATEMENT_LINES if n not in found)}"
if len(found) < len(STATEMENT_LINES) else ""))
if not dry_run and found:
_write({**values, "ticker": ticker, "period": row["period"]})
filled += 1
print(f"{ticker}: {'dry run, nothing written' if dry_run else f'{filled} row(s) updated'}")
return 0
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("ticker", nargs="?", help="Ticker to backfill.")
parser.add_argument("--all", action="store_true", help="Every ingested ticker.")
parser.add_argument("--dry-run", action="store_true", help="Fetch and report without writing.")
args = parser.parse_args()
if args.all:
tickers = metrics_db.list_tickers()
elif args.ticker:
tickers = [args.ticker.upper()]
else:
parser.error("give a ticker or --all")
metrics_db.init_db()
failures = 0
for ticker in tickers:
print(f"\n== {ticker} ==")
failures += backfill(ticker, dry_run=args.dry_run)
time.sleep(_PAUSE_SECONDS)
return 1 if failures else 0
if __name__ == "__main__":
raise SystemExit(main())