#!/usr/bin/env python3 """Build static FinanceBench run and comparison shards. Run from the viewer repository root: python3 scripts/build_runs.py The defaults point at the retained canonical full runs. Every output run is left-joined to the 150-question open-source eval set, so absent or empty responses remain visible and count against the full denominator. """ from __future__ import annotations import argparse import json import shutil from pathlib import Path from typing import Any, Iterator ROOT = Path(__file__).resolve().parent.parent DEFAULT_EVAL = ROOT / "financebench_open_source.jsonl" EVENTS_LIMIT_BYTES = 256 * 1024 STRING_LIMIT = 8192 RUN_DEFINITIONS = { "c1": { "label": "c1 Closed-book", "accent": "#38bdf8", "response": Path("/mnt/tmp/financebench-c1-predictions.jsonl/predictions"), "judge": Path("/mnt/tmp/financebench-c1-evaluated.jsonl/evaluated"), "provenance": "Baseline c1 closed-book full run", "expected": {"correct": 56, "answered": 134}, }, "c2": { "label": "c2 With-docs", "accent": "#a78bfa", "response": Path("/mnt/tmp/financebench-c2-predictions.jsonl/predictions"), "judge": Path("/mnt/tmp/financebench-c2-evaluated.jsonl/evaluated"), "provenance": "Baseline c2 with-docs full run", "expected": {"correct": 113, "answered": 137}, }, "c6": { "label": "c6 Agentic-DCI", "accent": "#34d399", "response": Path( "/tmp/baseline-pipeline-runs/gifted_helmet_6vkx7rqq1f/" "c6_predictions/predictions" ), "judge": Path( "/tmp/baseline-pipeline-runs/gifted_helmet_6vkx7rqq1f/" "c6_evaluated/evaluated" ), "provenance": ( "Corrected FinanceBench c6 rawtext full run " "(AML gifted_helmet_6vkx7rqq1f)" ), "expected": {"correct": 122, "answered": 150}, }, "naive": { "label": "Naive-search", "accent": "#f59e0b", "response": Path("/mnt/tmp/financebench-naive-predictions.jsonl/predictions"), "judge": Path("/mnt/tmp/financebench-naive-evaluated.jsonl/evaluated"), "provenance": "Naive-search full run", "expected": {"correct": 102, "answered": 124}, }, "e2e": { "label": "E2E v3", "accent": "#22c55e", "response": Path( "/home/azureuser/projects/information-scaffolds/outputs/e2e_runs/" "new-datasets-full-20260711/financebench/named-outputs/" "predictions/predictions" ), "judge": Path( "/home/azureuser/projects/information-scaffolds/outputs/e2e_runs/" "new-datasets-full-20260711/judges/financebench/named-outputs/" "judged/judged" ), "provenance": "E2E v3 new-datasets full run (2026-07-11)", "expected": {"correct": 121, "answered": 149}, }, "e2e_rawtext": { "label": "E2E v3 + rawtext", "accent": "#fb7185", "response": Path( "/tmp/viewer-overlay-finance/predictions/named-outputs/" "predictions/predictions" ), "judge": Path( "/tmp/viewer-overlay-finance/evaluated/named-outputs/" "canonical_evaluated/evaluated" ), "provenance": ( "Native E2E v3 + rawtext overlay full run " "(AML silly_salt_1vr6cy7bw8)" ), "expected": {"correct": 123, "answered": 150}, }, } SCOPE_NOTE = ( "FinanceBench open-source 150-question eval subset. Scores use all 150 " "questions; missing or unanswered responses count as incorrect. c6 uses the " "corrected rawtext asset layout from AML job gifted_helmet_6vkx7rqq1f." ) def iter_jsonl(path: Path) -> Iterator[dict[str, Any]]: """Yield non-empty JSONL objects without loading the source file at once.""" with path.open(encoding="utf-8") as handle: for line_number, line in enumerate(handle, 1): if not line.strip(): continue try: value = json.loads(line) except json.JSONDecodeError as exc: raise ValueError(f"{path}:{line_number}: invalid JSON: {exc}") from exc if not isinstance(value, dict): raise ValueError(f"{path}:{line_number}: expected a JSON object") yield value def keyed_rows(path: Path, key: str) -> dict[str, dict[str, Any]]: rows: dict[str, dict[str, Any]] = {} for row in iter_jsonl(path): value = row.get(key) if not isinstance(value, str) or not value: raise ValueError(f"{path}: row missing non-empty {key!r}") if value in rows: raise ValueError(f"{path}: duplicate {key} {value!r}") rows[value] = row return rows def cap_string(value: str, limit: int = STRING_LIMIT) -> str: if len(value) <= limit: return value omitted = len(value) - limit suffix = f"\n… [truncated {omitted} characters]" return value[: limit - len(suffix)] + suffix def cap_nested(value: Any) -> Any: """Recursively cap strings retained from judge data or trajectory events.""" if isinstance(value, str): return cap_string(value) if isinstance(value, list): return [cap_nested(item) for item in value] if isinstance(value, dict): return {str(key): cap_nested(item) for key, item in value.items()} if value is None or isinstance(value, (bool, int, float)): return value return cap_string(str(value)) def serialized_size(value: Any) -> int: return len( json.dumps(value, ensure_ascii=False, separators=(",", ":")).encode("utf-8") ) def compact_events(events: Any) -> list[dict[str, Any]]: if not isinstance(events, list): return [] compacted: list[dict[str, Any]] = [] for event in events: if not isinstance(event, dict): compacted.append({"type": "event", "content": cap_nested(event)}) continue item = { key: cap_nested(event.get(key)) for key in ("type", "name", "input", "content") if event.get(key) is not None } compacted.append(item) if serialized_size(compacted) <= EVENTS_LIMIT_BYTES: return compacted kept: list[dict[str, Any]] = [] reserve = 512 for event in compacted: if serialized_size(kept + [event]) > EVENTS_LIMIT_BYTES - reserve: break kept.append(event) marker = { "type": "truncated", "content": ( f"Trajectory truncated after {len(kept)} of {len(compacted)} events " f"to stay near the {EVENTS_LIMIT_BYTES}-byte serialized limit." ), } kept.append(marker) return kept def normalized_failure(response: dict[str, Any], answered: bool) -> str | None: failure = ( response.get("failure_reason") or response.get("error") or response.get("exception") ) if failure: return cap_string(str(failure)) if answered: return None return ( response.get("stop_reason") or response.get("finish_reason") or ((response.get("finish_reasons") or [None])[-1]) or "missing_response" ) def normalize_record( eval_row: dict[str, Any], response: dict[str, Any] | None, judge: dict[str, Any] | None, ) -> dict[str, Any]: response = response or {} parsed = judge.get("parsed", {}) if judge else {} prediction = response.get("answer") if prediction is not None: prediction = str(prediction) answered = bool(prediction and prediction.strip()) correct = parsed.get("correct") if judge else None if not isinstance(correct, bool): correct = None finish_reasons = response.get("finish_reasons") finish_reason = response.get("finish_reason") if not finish_reason and isinstance(finish_reasons, list) and finish_reasons: finish_reason = finish_reasons[-1] return { "qid": eval_row["financebench_id"], "question": eval_row.get("question"), "gold": eval_row.get("answer"), "prediction": prediction, "extracted_answer": cap_nested(parsed.get("extracted_final_answer")), "answered": answered, "correct": correct, "judge_text": cap_nested(judge.get("judge_text")) if judge else None, "judge_confidence": cap_nested(parsed.get("confidence")) if judge else None, "stop_reason": response.get("stop_reason"), "finish_reason": finish_reason, "failure_reason": normalized_failure(response, answered), "token_usage": cap_nested(response.get("tokens") or response.get("usage")), "turns": response.get("turns"), "tool_counts": cap_nested(response.get("tool_call_counts") or {}), "events": compact_events(response.get("events")), "metadata": { "company": eval_row.get("company"), "doc_name": eval_row.get("doc_name"), "question_type": eval_row.get("question_type"), "question_reasoning": eval_row.get("question_reasoning"), "domain_question_num": eval_row.get("domain_question_num"), }, } def write_json(path: Path, value: Any) -> None: path.parent.mkdir(parents=True, exist_ok=True) with path.open("w", encoding="utf-8") as handle: json.dump(value, handle, ensure_ascii=False, indent=2) handle.write("\n") def clean_records(path: Path) -> None: if path.exists(): shutil.rmtree(path) path.mkdir(parents=True) def index_projection(record: dict[str, Any], path: str) -> dict[str, Any]: status = ( "missing" if not record["answered"] else ("correct" if record["correct"] is True else "incorrect") ) return { "qid": record["qid"], "question": record["question"], "gold": record["gold"], "prediction": record["prediction"], "company": record["metadata"]["company"], "doc_name": record["metadata"]["doc_name"], "answered": record["answered"], "correct": record["correct"], "status": status, "path": path, } def compare_projection(record: dict[str, Any]) -> dict[str, Any]: return { key: record[key] for key in ( "prediction", "extracted_answer", "answered", "correct", "judge_confidence", "stop_reason", "finish_reason", "failure_reason", "token_usage", "turns", "tool_counts", ) } def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--eval", type=Path, default=DEFAULT_EVAL) for slot, definition in RUN_DEFINITIONS.items(): parser.add_argument( f"--{slot}-response", type=Path, default=definition["response"] ) parser.add_argument(f"--{slot}-judge", type=Path, default=definition["judge"]) return parser.parse_args() def main() -> None: args = parse_args() eval_rows = list(iter_jsonl(args.eval)) eval_by_qid = { row["financebench_id"]: row for row in eval_rows if isinstance(row.get("financebench_id"), str) } if len(eval_rows) != 150 or len(eval_by_qid) != 150: raise ValueError( f"Expected 150 unique eval qids, found {len(eval_rows)} rows and " f"{len(eval_by_qid)} unique qids" ) manifest_runs: list[dict[str, Any]] = [] normalized_runs: dict[str, dict[str, dict[str, Any]]] = {} eval_qids = set(eval_by_qid) for slot, definition in RUN_DEFINITIONS.items(): response_path = getattr(args, f"{slot}_response") judge_path = getattr(args, f"{slot}_judge") responses = keyed_rows(response_path, "qid") judges = keyed_rows(judge_path, "qid") unknown = (set(responses) | set(judges)) - eval_qids if unknown: raise ValueError(f"{slot}: {len(unknown)} qids are absent from eval") records_dir = ROOT / "runs" / slot / "records" clean_records(records_dir) run_records: dict[str, dict[str, Any]] = {} index_records: list[dict[str, Any]] = [] for eval_row in eval_rows: qid = eval_row["financebench_id"] record = normalize_record(eval_row, responses.get(qid), judges.get(qid)) run_records[qid] = record relative_path = f"runs/{slot}/records/{qid}.json" write_json(ROOT / relative_path, record) index_records.append(index_projection(record, relative_path)) answered = sum(record["answered"] for record in run_records.values()) correct = sum(record["correct"] is True for record in run_records.values()) expected = definition["expected"] if answered != expected["answered"] or correct != expected["correct"]: raise ValueError( f"{slot}: got {correct} correct and {answered} answered; expected " f"{expected['correct']} correct and {expected['answered']} answered" ) denominator = len(eval_rows) run_manifest = { "slot": slot, "label": definition["label"], "score": { "numerator": correct, "denominator": denominator, "percent": round(correct * 100 / denominator, 2), }, "answered": answered, "missing": denominator - answered, "accent": definition["accent"], "source": { "eval": str(args.eval), "response": str(response_path), "judge": str(judge_path), }, "provenance": definition["provenance"], "scope_note": SCOPE_NOTE, "index": f"runs/{slot}/index.json", } manifest_runs.append(run_manifest) normalized_runs[slot] = run_records write_json( ROOT / "runs" / slot / "index.json", { "slot": slot, "label": definition["label"], "score": run_manifest["score"], "answered": answered, "missing": denominator - answered, "records": index_records, }, ) manifest = { "schema_version": 1, "dataset": "financebench", "denominator": len(eval_rows), "scope_note": SCOPE_NOTE, "runs": manifest_runs, "compare": {"label": "Compare", "index": "compare/index.json"}, } write_json(ROOT / "runs" / "manifest.json", manifest) compare_dir = ROOT / "compare" / "records" clean_records(compare_dir) compare_index: list[dict[str, Any]] = [] slots = list(RUN_DEFINITIONS) for eval_row in eval_rows: qid = eval_row["financebench_id"] run_values = { slot: compare_projection(normalized_runs[slot][qid]) for slot in slots } correct_values = [run_values[slot]["correct"] is True for slot in slots] missing_values = [not run_values[slot]["answered"] for slot in slots] flags = { "disagreement": len(set(correct_values)) > 1, "any_missing": any(missing_values), "only_e2e_correct": ( run_values["e2e"]["correct"] is True and all( run_values[slot]["correct"] is not True for slot in slots if slot != "e2e" ) ), } record = { "qid": qid, "question": eval_row.get("question"), "gold": eval_row.get("answer"), "metadata": { "company": eval_row.get("company"), "doc_name": eval_row.get("doc_name"), "question_type": eval_row.get("question_type"), "question_reasoning": eval_row.get("question_reasoning"), }, "runs": run_values, "flags": flags, } relative_path = f"compare/records/{qid}.json" write_json(ROOT / relative_path, record) compare_index.append( { "qid": qid, "question": eval_row.get("question"), "company": eval_row.get("company"), "flags": flags, "path": relative_path, } ) write_json( ROOT / "compare" / "index.json", { "label": "Compare", "runs": [ { "slot": run["slot"], "label": run["label"], "accent": run["accent"], "score": run["score"], "answered": run["answered"], "missing": run["missing"], } for run in manifest_runs ], "records": compare_index, }, ) print(f"Wrote {len(manifest_runs)} runs × {len(eval_rows)} records") for run in manifest_runs: score = run["score"] print( f" {run['slot']}: {score['numerator']}/{score['denominator']} " f"({score['percent']:.2f}%), answered {run['answered']}, " f"missing {run['missing']}" ) print(f"Wrote {len(compare_index)} prejoined compare records") if __name__ == "__main__": main()