File size: 9,077 Bytes
b2931f4
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
import json
import os
import time
from collections.abc import Iterator
from datetime import date
from pathlib import Path

import httpx
from pydantic import BaseModel

# ── Paths ─────────────────────────────────────────────────────────────────
# ← why: same trick as config.py β€” resolve to repo root so the script works
#   regardless of CWD. edgar.py β†’ ingestion/ β†’ finrag/ β†’ src/ β†’ backend/ β†’ ROOT
REPO_ROOT = Path(__file__).resolve().parents[4]
DATA_DIR = REPO_ROOT / "data" / "raw"

# ── HTTP client ───────────────────────────────────────────────────────────
# ← why: SEC requires a real contactable email. Fall back to yours so the
#   script never accidentally runs with a placeholder; override via env in CI.
USER_AGENT = os.getenv("SEC_USER_AGENT", "Aryan Sharma aryan250403@gmail.com")
HTTP_HEADERS = {
    "User-Agent": USER_AGENT,
    "Accept-Encoding": "gzip, deflate",
}
client = httpx.Client(headers=HTTP_HEADERS, timeout=30.0)

# ── Corpus config ─────────────────────────────────────────────────────────
TARGET_TICKERS = ["AAPL", "TSLA", "JPM"]
TARGET_YEARS = [2022, 2023, 2024]
REQUEST_SLEEP_SECONDS = 0.2  # ← why: SEC allows 10 req/s; 5 req/s is polite.


# ── Models ────────────────────────────────────────────────────────────────
class Filing(BaseModel):
    # ← why: every field here exists because some downstream stage needs it.
    #   ticker/cik/accession give three independent identifiers.
    #   filing_date vs period_of_report are *different things* and both matter.
    #   sec_url is for citation rendering in the UI later.
    ticker: str
    company_name: str
    cik: str
    form: str
    filing_date: date
    period_of_report: date
    fiscal_year: int
    accession_number: str
    accession_clean: str
    primary_document: str
    sec_url: str
    out_dir: str  # relative to DATA_DIR

    def metadata_dict(self) -> dict:
        # ← why: pydantic's model_dump() emits dates as date objects; JSON
        #   can't serialize those, so coerce to ISO strings explicitly.
        d = self.model_dump()
        d["filing_date"] = self.filing_date.isoformat()
        d["period_of_report"] = self.period_of_report.isoformat()
        return d


# ── EDGAR API calls ───────────────────────────────────────────────────────
def resolve_cik_map() -> dict[str, tuple[str, str]]:
    """Return {ticker: (cik_str, company_name)} for the whole market.

    ← why: SEC publishes the entire tickerβ†’CIK map as one JSON file. Fetching
       it once and resolving locally is cheaper than per-ticker lookups.
    """
    url = "https://www.sec.gov/files/company_tickers.json"
    response = client.get(url)
    response.raise_for_status()
    data = response.json()
    return {
        entry["ticker"].upper(): (str(entry["cik_str"]), entry["title"])
        for entry in data.values()
    }


def _iter_submission_pages(padded_cik: str) -> Iterator[dict]:
    """Yield each 'recent'-shaped page from the submissions endpoint.

    First yields filings.recent, then walks filings.files for older pages.
    High-volume filers (banks, frequent 8-K issuers) overflow `recent` and
    require pulling the paginated files to find 10-Ks more than ~1-2 yrs old.
    """
    url = f"https://data.sec.gov/submissions/CIK{padded_cik}.json"
    response = client.get(url)
    response.raise_for_status()
    data = response.json()

    yield data["filings"]["recent"]

    for file_entry in data["filings"].get("files", []):
        time.sleep(REQUEST_SLEEP_SECONDS)
        file_url = f"https://data.sec.gov/submissions/{file_entry['name']}"
        resp = client.get(file_url)
        resp.raise_for_status()
        yield resp.json()


def list_10k_filings(
    ticker: str, cik: str, company_name: str, target_years: list[int]
) -> list[Filing]:
    """Hit EDGAR's submissions endpoint and pull 10-Ks for the target years.

    Walks paginated submission pages until every target year is found or
    pages are exhausted.
    """
    padded_cik = cik.zfill(10)
    remaining_years = set(target_years)
    filings: list[Filing] = []

    for page in _iter_submission_pages(padded_cik):
        if not remaining_years:
            break
        for acc_num, form, filing_date_str, period_str, prim_doc in zip(
            page["accessionNumber"],
            page["form"],
            page["filingDate"],
            page["reportDate"],
            page["primaryDocument"],
        ):
            if form != "10-K":
                continue
            if not period_str:
                continue
            period_of_report = date.fromisoformat(period_str)
            fiscal_year = period_of_report.year
            if fiscal_year not in remaining_years:
                continue

            accession_clean = acc_num.replace("-", "")
            sec_url = (
                f"https://www.sec.gov/Archives/edgar/data/{int(cik)}/"
                f"{accession_clean}/{prim_doc}"
            )
            filings.append(
                Filing(
                    ticker=ticker,
                    company_name=company_name,
                    cik=cik,
                    form=form,
                    filing_date=date.fromisoformat(filing_date_str),
                    period_of_report=period_of_report,
                    fiscal_year=fiscal_year,
                    accession_number=acc_num,
                    accession_clean=accession_clean,
                    primary_document=prim_doc,
                    sec_url=sec_url,
                    out_dir=f"{ticker}_{fiscal_year}",
                )
            )
            remaining_years.discard(fiscal_year)
    return filings


def download_filing(filing: Filing, base_out_dir: Path) -> None:
    """Download the primary 10-K HTML and write metadata.json alongside it."""
    dir_path = base_out_dir / filing.out_dir
    dir_path.mkdir(parents=True, exist_ok=True)

    htm_path = dir_path / "filing.htm"
    meta_path = dir_path / "metadata.json"

    # ← why: idempotency. Existence of *both* artifacts means a clean prior run.
    #   If only one exists, we redo to repair partial state.
    if htm_path.exists() and meta_path.exists():
        print(f"  ↳ skip   {filing.ticker} FY{filing.fiscal_year} (already on disk)")
        return

    print(f"  ↳ fetch  {filing.ticker} FY{filing.fiscal_year}  β†’  {filing.sec_url}")
    response = client.get(filing.sec_url)
    response.raise_for_status()
    htm_path.write_bytes(response.content)
    meta_path.write_text(json.dumps(filing.metadata_dict(), indent=2))


def write_manifest(filings: list[Filing], base_out_dir: Path) -> None:
    """Top-level index of everything we've downloaded.

    ← why: lets later stages (parser, embedder) load the corpus by reading one
       file instead of walking the tree.
    """
    manifest_path = base_out_dir / "manifest.json"
    manifest = {
        "filings": [
            {
                "ticker": f.ticker,
                "fiscal_year": f.fiscal_year,
                "period_of_report": f.period_of_report.isoformat(),
                "path": f.out_dir,
            }
            for f in filings
        ]
    }
    manifest_path.write_text(json.dumps(manifest, indent=2))


# ── CLI entrypoint ────────────────────────────────────────────────────────
def main() -> None:
    DATA_DIR.mkdir(parents=True, exist_ok=True)
    print(f"Downloading to: {DATA_DIR}")
    print(f"User-Agent: {USER_AGENT}\n")

    ticker_map = resolve_cik_map()
    all_filings: list[Filing] = []

    for ticker in TARGET_TICKERS:
        if ticker not in ticker_map:
            raise ValueError(f"Ticker {ticker} not found in SEC ticker map")
        cik, company_name = ticker_map[ticker]
        print(f"[{ticker}] {company_name} (CIK {cik})")

        filings = list_10k_filings(ticker, cik, company_name, TARGET_YEARS)
        if len(filings) < len(TARGET_YEARS):
            missing = set(TARGET_YEARS) - {f.fiscal_year for f in filings}
            print(f"  ⚠ missing fiscal years: {sorted(missing)}")

        for filing in filings:
            download_filing(filing, DATA_DIR)
            time.sleep(REQUEST_SLEEP_SECONDS)
        all_filings.extend(filings)
        print()

    write_manifest(all_filings, DATA_DIR)
    print(f"Done. {len(all_filings)} filings on disk.")


if __name__ == "__main__":
    main()