File size: 5,898 Bytes
5b9b478
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
"""Backfill earnings-call transcripts that ``ingest.py`` can no longer reach.

Two gaps leave a stored period without its transcript, and neither is repaired
by a normal delta run:

1. **The filing aged out of EDGAR's "recent" index.** ``ingest.py`` walks
   ``submissions/CIK.json``'s ``filings.recent`` array to decide which periods
   to process. A heavy filer (Alphabet files hundreds of Form 4s a year) pushes
   its own older 10-Qs out of that array into the paginated archive, so the
   period stays in ``metrics.db`` but is never revisited β€” and its missing
   transcript is never fetched.

2. **A stale "unavailable" marker.** ``sections.db`` records an empty string to
   mean "asked the provider, it had nothing", which correctly stops a delta run
   from re-asking every time. But providers backfill their own archives, so a
   marker written months ago can be wrong today.

This script targets exactly the periods that are missing a transcript, asks the
provider again, and stores anything it gets through the same code path
ingestion uses (``sections.db`` plus the Chroma transcript collection).

Usage::

    python scripts/backfill_transcripts.py                # report only
    python scripts/backfill_transcripts.py --apply        # every ticker
    python scripts/backfill_transcripts.py --apply GOOGL TSLA
"""
from __future__ import annotations

import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))

from dotenv import load_dotenv

load_dotenv()

from ingestion.embedder import embed_and_store_transcript
from ingestion.transcript import fetch_transcript
from storage.metrics_db import get_all_metrics
from storage.sections_db import get_section, init_sections_db, upsert_section


def _period_to_av_quarter(period: str) -> str:
    """EDGAR period β†’ provider quarter. 'Q12026' β†’ '2026Q1'; 'FY2024' β†’ '2024Q4'."""
    period = (period or "").strip()
    if len(period) < 6:
        return ""
    if period.startswith("FY"):
        year = period[2:]
        return f"{year}Q4" if len(year) == 4 and year.isdigit() else ""
    if period[0] == "Q" and period[1] in "1234":
        year = period[2:]
        return f"{year}{period[:2]}" if len(year) == 4 and year.isdigit() else ""
    return ""


def _known_tickers() -> list[str]:
    import sqlite3

    from storage.metrics_db import DB_PATH

    if not DB_PATH.exists():
        return []
    with sqlite3.connect(DB_PATH) as conn:
        return [row[0] for row in conn.execute(
            "SELECT DISTINCT ticker FROM metrics ORDER BY ticker"
        )]


def find_gaps(ticker: str) -> list[dict]:
    """Stored periods whose transcript is absent or recorded as unavailable."""
    gaps: list[dict] = []
    for row in get_all_metrics(ticker) or []:
        period = str(row.get("period") or "")
        form_type = str(row.get("form_type") or "")
        if not period or form_type not in ("10-Q", "10-K"):
            continue
        stored = get_section(ticker, period, "transcript")
        if stored:
            continue
        gaps.append({
            "ticker": ticker,
            "period": period,
            "form_type": form_type,
            "company_name": str(row.get("company_name") or ticker),
            "filing_date": str(row.get("filing_date") or ""),
            # None = never asked; "" = asked, provider had nothing at the time.
            "reason": "never_attempted" if stored is None else "marked_unavailable",
        })
    return gaps


def backfill(ticker: str, apply: bool) -> tuple[int, int]:
    """Return (gaps found, transcripts stored)."""
    gaps = find_gaps(ticker)
    if not gaps:
        print(f"[{ticker}] complete β€” every stored period has a transcript.")
        return 0, 0

    stored_count = 0
    for gap in gaps:
        quarter = _period_to_av_quarter(gap["period"])
        if not quarter:
            print(f"[{ticker}] {gap['period']}: unparseable period, skipped.")
            continue

        if not apply:
            print(f"[{ticker}] {gap['period']} ({quarter}) β€” missing ({gap['reason']})")
            continue

        try:
            text = fetch_transcript(ticker, quarter) or ""
        except Exception as exc:
            print(f"[{ticker}] {gap['period']}: provider error β€” {exc}")
            continue

        # Writing the empty string back is deliberate: it records that the
        # provider was asked and had nothing, which is what stops a delta run
        # from re-asking on every ingestion.
        upsert_section(ticker, gap["period"], gap["form_type"], "transcript", text)
        if not text:
            print(f"[{ticker}] {gap['period']} ({quarter}): provider still has nothing.")
            continue

        embed_and_store_transcript(
            ticker=ticker,
            company_name=gap["company_name"],
            transcript_text=text,
            transcript_date=gap["filing_date"],
            period=gap["period"],
            source_url="",
            provider="alphavantage",
        )
        stored_count += 1
        print(f"[{ticker}] {gap['period']} ({quarter}): stored {len(text):,} chars.")

    return len(gaps), stored_count


def main(argv: list[str]) -> int:
    apply = "--apply" in argv
    tickers = [a.upper() for a in argv if not a.startswith("--")] or _known_tickers()
    if not tickers:
        print("No ingested tickers found. Run ingest.py first.")
        return 1

    init_sections_db()
    total_gaps = total_stored = 0
    for ticker in tickers:
        gaps, stored = backfill(ticker, apply)
        total_gaps += gaps
        total_stored += stored

    if apply:
        print(f"\nDone. {total_stored} transcript(s) stored across {len(tickers)} ticker(s).")
    else:
        print(f"\n{total_gaps} gap(s) found. Re-run with --apply to fetch them.")
    return 0


if __name__ == "__main__":
    sys.exit(main(sys.argv[1:]))