| |
| """ |
| Build the demo index at Docker-image build time. |
| |
| Parses every PDF in data/uploads (the committed public demo filing) into |
| the graph + Chroma store so a fresh container boots ready to answer — |
| no 3-minute first-upload wait in front of a recruiter, and no 60MB+ of |
| binary index files in git. |
| |
| Run from the repo root: python -m backend.seed |
| Idempotent: skips work if the graph file already exists. |
| """ |
|
|
| from pathlib import Path |
|
|
| from backend.graph import FinancialGraph |
| from backend.parser import parse_document |
|
|
| DATA_DIR = Path("data") |
| GRAPH_PATH = DATA_DIR / "graph.json" |
| UPLOADS_DIR = DATA_DIR / "uploads" |
|
|
| |
| |
| KNOWN_FILINGS = { |
| "HDFC_Bank_Annual_Report_2024_25-310202.pdf": ("HDFC Bank", "2025"), |
| } |
|
|
|
|
| def main(): |
| if GRAPH_PATH.exists(): |
| print(f"seed: {GRAPH_PATH} already exists, skipping") |
| return |
|
|
| pdfs = sorted(UPLOADS_DIR.glob("*.pdf")) if UPLOADS_DIR.is_dir() else [] |
| if not pdfs: |
| print("seed: no PDFs in data/uploads, nothing to do") |
| return |
|
|
| fg = FinancialGraph() |
| for pdf in pdfs: |
| company, year = KNOWN_FILINGS.get( |
| pdf.name, (pdf.stem.replace("_", " ")[:40], "2025") |
| ) |
| print(f"seed: parsing {pdf.name} as {company} {year}") |
| parsed = parse_document(str(pdf), company, year) |
| |
| fg.add_document(parsed, use_llm_fallback=False) |
|
|
| DATA_DIR.mkdir(exist_ok=True) |
| fg.save(str(GRAPH_PATH)) |
| print("seed: done") |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|