#!/usr/bin/env python3 """Build eval.json + corpus_index.json for the FinLongDocQA viewer. - eval.json one entry per QA example (7,527); question, gold numeric answer, type, reasoning trace, executable python, evidence page numbers, and the source report (company/year). - corpus_index.json one entry per annual-report markdown file in the corpus (1,456 == every reports//.md). Each has a title ("TICKER · YEAR"), the count of questions that cite it, and `md_url` pointing at the report on the HF dataset CDN (streamed + rendered client-side, never bundled). Run from the viewer repo root: python scripts/build_data.py [--reports-dir DIR] [--qa FILE] Reads: dataset_qa.jsonl, and the reports/ tree (for the full corpus listing). Writes: eval.json, corpus_index.json """ import argparse import json import os from collections import Counter ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) DEFAULT_QA = os.path.join(ROOT, "dataset_qa.jsonl") DEFAULT_REPORTS = "/mnt/ramdisk/blobstore/timchen0618/data/FinLongDocQA/reports" # The markdown corpus lives on a Hugging Face dataset; the viewer fetches each # report on demand from the CDN so the Space never bundles the ~1 GB corpus. MD_URL = "https://huggingface.co/datasets/timchen0618/finlongdocqa-reports/resolve/main/reports/{company}/{year}.md" def enumerate_reports(reports_dir): """Return sorted set of (company, year) from the reports/ tree, if present.""" docs = set() if not os.path.isdir(reports_dir): return docs for company in os.listdir(reports_dir): cdir = os.path.join(reports_dir, company) if not os.path.isdir(cdir): continue for fn in os.listdir(cdir): if fn.endswith(".md"): docs.add((company, fn[:-3])) return docs def main(): ap = argparse.ArgumentParser() ap.add_argument("--qa", default=DEFAULT_QA) ap.add_argument("--reports-dir", default=DEFAULT_REPORTS) args = ap.parse_args() with open(args.qa, encoding="utf-8") as f: qa = [json.loads(l) for l in f if l.strip()] eval_rows = [] per_doc_q = Counter() for r in qa: company = r.get("company") year = str(r.get("year")) doc_id = company + "/" + year per_doc_q[(company, year)] += 1 pages = r.get("page_numbers") or [] eval_rows.append({ "id": r.get("id"), "company": company, "year": year, "doc_id": doc_id, "title": company + " · " + year, "question": r.get("question", ""), "type": r.get("type", ""), "answer": r.get("answer"), "thoughts": r.get("thoughts", ""), "python_code": r.get("python_code", ""), "page_numbers": pages, }) # corpus = every report file on disk (falls back to QA-referenced docs) report_docs = enumerate_reports(args.reports_dir) if not report_docs: report_docs = set(per_doc_q.keys()) print("WARN: reports dir not found; corpus limited to QA-referenced docs") corpus_rows = [] for company, year in sorted(report_docs): corpus_rows.append({ "doc_id": company + "/" + year, "company": company, "year": year, "title": company + " · " + year, "n_questions": per_doc_q.get((company, year), 0), "md_url": MD_URL.format(company=company, year=year), }) with open(os.path.join(ROOT, "eval.json"), "w", encoding="utf-8") as f: json.dump(eval_rows, f, ensure_ascii=False, indent=0) with open(os.path.join(ROOT, "corpus_index.json"), "w", encoding="utf-8") as f: json.dump(corpus_rows, f, ensure_ascii=False, indent=0) print(f"wrote eval.json: {len(eval_rows)} questions") print(f"wrote corpus_index.json: {len(corpus_rows)} documents " f"({sum(1 for c in corpus_rows if c['n_questions'] == 0)} with no questions)") print("types:", dict(Counter(r["type"] for r in eval_rows))) print("years:", dict(Counter(r["year"] for r in eval_rows))) if __name__ == "__main__": main()