#!/usr/bin/env python3 """CLI wrapper for :func:`mitointeract_recovery.bindingdb_benchmark.prepare_benchmark`. Builds the assay- and citation-stratified BindingDB exact-Kd benchmark (``sample.jsonl``, four pair-level split manifests, deterministic ``audit.json``) from the pinned ``gold_exact_kd.jsonl`` and its source audit. All validation, aggregation, and split logic lives in ``mitointeract_recovery.bindingdb_benchmark``; this script only parses arguments and prints the resulting audit. """ from __future__ import annotations import argparse import json from pathlib import Path from mitointeract_recovery.bindingdb_benchmark import prepare_benchmark def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument( "--gold-jsonl", type=Path, required=True, help="path to gold_exact_kd.jsonl from prepare_bindingdb_gold_kd.py", ) parser.add_argument( "--source-audit", type=Path, required=True, help="path to bindingdb_audit.json pinning the gold file checksums", ) parser.add_argument("--output-dir", type=Path, required=True) parser.add_argument("--seed", type=int, default=42) args = parser.parse_args() audit = prepare_benchmark( gold_jsonl=args.gold_jsonl, source_audit=args.source_audit, output_dir=args.output_dir, seed=args.seed, ) print(json.dumps(audit, indent=2, sort_keys=True)) if __name__ == "__main__": main()