finlongdocqa-viewer / scripts /build_data.py
finlongdocqa-viewer bot
FinLongDocQA viewer: Corpus (1456 markdown reports) + Eval (7527 Q) static Space
8415278
Raw
History Blame Contribute Delete
4.3 kB
#!/usr/bin/env python3
"""Build eval.json + corpus_index.json for the FinLongDocQA viewer.
- eval.json one entry per QA example (7,527); question, gold numeric
answer, type, reasoning trace, executable python, evidence
page numbers, and the source report (company/year).
- corpus_index.json one entry per annual-report markdown file in the corpus
(1,456 == every reports/<TICKER>/<YEAR>.md). Each has a
title ("TICKER 路 YEAR"), the count of questions that cite
it, and `md_url` pointing at the report on the HF dataset
CDN (streamed + rendered client-side, never bundled).
Run from the viewer repo root:
python scripts/build_data.py [--reports-dir DIR] [--qa FILE]
Reads: dataset_qa.jsonl, and the reports/ tree (for the full corpus listing).
Writes: eval.json, corpus_index.json
"""
import argparse
import json
import os
from collections import Counter
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DEFAULT_QA = os.path.join(ROOT, "dataset_qa.jsonl")
DEFAULT_REPORTS = "/mnt/ramdisk/blobstore/timchen0618/data/FinLongDocQA/reports"
# The markdown corpus lives on a Hugging Face dataset; the viewer fetches each
# report on demand from the CDN so the Space never bundles the ~1 GB corpus.
MD_URL = "https://huggingface.co/datasets/timchen0618/finlongdocqa-reports/resolve/main/reports/{company}/{year}.md"
def enumerate_reports(reports_dir):
"""Return sorted set of (company, year) from the reports/ tree, if present."""
docs = set()
if not os.path.isdir(reports_dir):
return docs
for company in os.listdir(reports_dir):
cdir = os.path.join(reports_dir, company)
if not os.path.isdir(cdir):
continue
for fn in os.listdir(cdir):
if fn.endswith(".md"):
docs.add((company, fn[:-3]))
return docs
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--qa", default=DEFAULT_QA)
ap.add_argument("--reports-dir", default=DEFAULT_REPORTS)
args = ap.parse_args()
with open(args.qa, encoding="utf-8") as f:
qa = [json.loads(l) for l in f if l.strip()]
eval_rows = []
per_doc_q = Counter()
for r in qa:
company = r.get("company")
year = str(r.get("year"))
doc_id = company + "/" + year
per_doc_q[(company, year)] += 1
pages = r.get("page_numbers") or []
eval_rows.append({
"id": r.get("id"),
"company": company,
"year": year,
"doc_id": doc_id,
"title": company + " 路 " + year,
"question": r.get("question", ""),
"type": r.get("type", ""),
"answer": r.get("answer"),
"thoughts": r.get("thoughts", ""),
"python_code": r.get("python_code", ""),
"page_numbers": pages,
})
# corpus = every report file on disk (falls back to QA-referenced docs)
report_docs = enumerate_reports(args.reports_dir)
if not report_docs:
report_docs = set(per_doc_q.keys())
print("WARN: reports dir not found; corpus limited to QA-referenced docs")
corpus_rows = []
for company, year in sorted(report_docs):
corpus_rows.append({
"doc_id": company + "/" + year,
"company": company,
"year": year,
"title": company + " 路 " + year,
"n_questions": per_doc_q.get((company, year), 0),
"md_url": MD_URL.format(company=company, year=year),
})
with open(os.path.join(ROOT, "eval.json"), "w", encoding="utf-8") as f:
json.dump(eval_rows, f, ensure_ascii=False, indent=0)
with open(os.path.join(ROOT, "corpus_index.json"), "w", encoding="utf-8") as f:
json.dump(corpus_rows, f, ensure_ascii=False, indent=0)
print(f"wrote eval.json: {len(eval_rows)} questions")
print(f"wrote corpus_index.json: {len(corpus_rows)} documents "
f"({sum(1 for c in corpus_rows if c['n_questions'] == 0)} with no questions)")
print("types:", dict(Counter(r["type"] for r in eval_rows)))
print("years:", dict(Counter(r["year"] for r in eval_rows)))
if __name__ == "__main__":
main()