# Configuration for the scripts in this repo. Copy to .env and edit: # cp example.env .env # The eval scripts read .env automatically; build_corpus.py reads # NYSENATE_API_KEY from it (use --data-dir or export DATA_DIR for the rest). # Every value is optional -- the defaults shown are what the scripts assume. # OpenLegislation API key for the statute scrape in build_corpus.py # (free key: https://legislation.nysenate.gov) NYSENATE_API_KEY= # Folder holding the corpus documents listed in sources.csv DATA_DIR=RAG Data # Where results/ and emb/ (embedding caches) are written SCRATCH=. # The 75-item project eval set EVAL_PATH=eval_set_v1.jsonl # Generator model and a tag appended to output filenames GEN_MODEL=Qwen/Qwen3-4B-Instruct-2507 MODEL_TAG= # External benchmark data locations # git clone https://github.com/IBM/mt-rag-benchmark (then unzip corpora/passage_level/govt.jsonl.zip) MTRAG_REPO=mt-rag-benchmark # LegalBench-RAG bundle from the link in https://github.com/ZeroEntropy-AI/legalbenchrag LEGALBENCHRAG_ROOT=legalbenchrag # HuggingFace model cache location # HF_HOME=/path/to/hf-cache