| # Configuration for the scripts in this repo. Copy to .env and edit: | |
| # cp example.env .env | |
| # The eval scripts read .env automatically; build_corpus.py reads | |
| # NYSENATE_API_KEY from it (use --data-dir or export DATA_DIR for the rest). | |
| # Every value is optional -- the defaults shown are what the scripts assume. | |
| # OpenLegislation API key for the statute scrape in build_corpus.py | |
| # (free key: https://legislation.nysenate.gov) | |
| NYSENATE_API_KEY= | |
| # Folder holding the corpus documents listed in sources.csv | |
| DATA_DIR=RAG Data | |
| # Where results/ and emb/ (embedding caches) are written | |
| SCRATCH=. | |
| # The 75-item project eval set | |
| EVAL_PATH=eval_set_v1.jsonl | |
| # Generator model and a tag appended to output filenames | |
| GEN_MODEL=Qwen/Qwen3-4B-Instruct-2507 | |
| MODEL_TAG= | |
| # External benchmark data locations | |
| # git clone https://github.com/IBM/mt-rag-benchmark (then unzip corpora/passage_level/govt.jsonl.zip) | |
| MTRAG_REPO=mt-rag-benchmark | |
| # LegalBench-RAG bundle from the link in https://github.com/ZeroEntropy-AI/legalbenchrag | |
| LEGALBENCHRAG_ROOT=legalbenchrag | |
| # HuggingFace model cache location | |
| # HF_HOME=/path/to/hf-cache | |