| """Build the FinChat vector store from recent SEC 10-K filings (EDGAR). |
| |
| Pipeline: |
| fetch each target 10-K (edgartools) -> split into chunks |
| -> embed locally -> store in Chroma (persisted to config.VECTORSTORE_DIR) |
| |
| Run once, from the project root: |
| python -m src.ingest |
| """ |
| from __future__ import annotations |
|
|
| import shutil |
| import sys |
| import time |
| from pathlib import Path |
|
|
| |
| sys.path.insert(0, str(Path(__file__).resolve().parents[1])) |
|
|
| from edgar import Company, set_identity |
| from langchain_core.documents import Document |
| from langchain_text_splitters import RecursiveCharacterTextSplitter |
| from langchain_huggingface import HuggingFaceEmbeddings |
| from langchain_chroma import Chroma |
|
|
| from src import config |
| from src.financials import financial_documents |
|
|
| |
| |
| try: |
| sys.stdout.reconfigure(encoding="utf-8", errors="replace") |
| except Exception: |
| pass |
|
|
|
|
| def get_filing(ticker: str, fiscal_year: int): |
| """Return the 10-K filing whose fiscal period matches fiscal_year, else None. |
| |
| Matches on the filing's period_of_report year, so an offset fiscal year |
| (e.g. Amcor's June close) still resolves to the right filing. |
| """ |
| for f in Company(ticker).get_filings(form="10-K"): |
| period = getattr(f, "period_of_report", None) |
| if period and str(period)[:4] == str(fiscal_year): |
| return f |
| return None |
|
|
|
|
| def fetch_documents() -> list[Document]: |
| """Fetch every target filing -> text chunks + structured financial facts.""" |
| set_identity(config.EDGAR_IDENTITY) |
| splitter = RecursiveCharacterTextSplitter( |
| chunk_size=config.CHUNK_SIZE, |
| chunk_overlap=config.CHUNK_OVERLAP, |
| ) |
|
|
| docs: list[Document] = [] |
| for ticker, name, year in config.TARGET_FILINGS: |
| print(f"Fetching {ticker} FY{year} 10-K ...", end=" ", flush=True) |
| filing = get_filing(ticker, year) |
| if filing is None: |
| print("NOT FOUND -- skipping") |
| continue |
|
|
| |
| text = filing.text() |
| source = f"{name} 10-K (FY{year})" |
| for chunk in splitter.split_text(text): |
| docs.append( |
| Document( |
| page_content=chunk, |
| metadata={ |
| "ticker": ticker, |
| "company": name, |
| "year": str(year), |
| "accession": filing.accession_no, |
| "source": source, |
| "type": "text", |
| }, |
| ) |
| ) |
|
|
| |
| fin_docs = financial_documents(filing, ticker, name, year) |
| docs.extend(fin_docs) |
|
|
| print(f"{len(text):,} chars + {len(fin_docs)} financial statements " |
| f"-> {len(docs)} docs so far") |
| time.sleep(0.5) |
| return docs |
|
|
|
|
| def _safe_rmtree(path: Path, retries: int = 3) -> None: |
| """Delete a directory, retrying briefly through transient file locks.""" |
| for _ in range(retries): |
| try: |
| shutil.rmtree(path) |
| return |
| except (PermissionError, OSError): |
| time.sleep(1.0) |
| shutil.rmtree(path) |
|
|
|
|
| def build_vectorstore(chunks: list[Document]) -> None: |
| if config.VECTORSTORE_DIR.exists(): |
| print("Removing existing vector store ...") |
| _safe_rmtree(config.VECTORSTORE_DIR) |
| config.VECTORSTORE_DIR.parent.mkdir(parents=True, exist_ok=True) |
|
|
| print(f"Loading embedding model {config.EMBEDDING_MODEL} ...") |
| embeddings = HuggingFaceEmbeddings(model_name=config.EMBEDDING_MODEL) |
|
|
| print(f"Embedding & storing {len(chunks)} chunks (a few minutes) ...") |
| Chroma.from_documents( |
| documents=chunks, |
| embedding=embeddings, |
| collection_name=config.CHROMA_COLLECTION, |
| persist_directory=str(config.VECTORSTORE_DIR), |
| ) |
| print(f"Done. Vector store saved to: {config.VECTORSTORE_DIR}") |
|
|
|
|
| def build_index() -> None: |
| """Full pipeline: fetch filings -> chunk -> embed -> store. |
| |
| Importable so the app can bootstrap the store on first run. |
| """ |
| docs = fetch_documents() |
| if not docs: |
| raise SystemExit("No filings were fetched — check tickers/years in config.py.") |
| print(f"Total: {len(docs)} chunks from {len(config.TARGET_FILINGS)} target filings.") |
| build_vectorstore(docs) |
|
|
|
|
| def main() -> None: |
| build_index() |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|