Spaces:
Running
Running
| """ | |
| inspect-ai Task definition for L2-Bench. | |
| Pairs the generate() solver with the LLM-as-judge scorer, sourcing task data | |
| from the gated L2-Bench dataset on the Hugging Face Hub. | |
| Version | Date | Author | Change comment | |
| --------|------------|-----------|--------------- | |
| 1.0.0 | 2026-07-29 | M. Ku | Initial open-source release | |
| """ | |
| import os | |
| from pathlib import Path | |
| from huggingface_hub import snapshot_download | |
| from huggingface_hub.errors import GatedRepoError, HfHubHTTPError | |
| from inspect_ai import Task | |
| from inspect_ai.dataset import MemoryDataset | |
| from inspect_ai.solver import generate | |
| from l2_bench_eval import config | |
| from l2_bench_eval.dataset import create_inspect_dataset | |
| from l2_bench_eval.score import ScorerSetting, l2_bench_scorer | |
| def _fetch_tasks_from_hf() -> tuple[Path, Path]: | |
| """Download (or reuse cached) task CSV + resources from the HF dataset repo.""" | |
| try: | |
| snapshot_dir = Path( | |
| snapshot_download( | |
| repo_id=config.TASKS_REPO, | |
| repo_type="dataset", | |
| token=os.environ.get("HF_TOKEN"), | |
| ) | |
| ) | |
| except (GatedRepoError, HfHubHTTPError) as exc: | |
| raise RuntimeError( | |
| f"Could not download the gated dataset '{config.TASKS_REPO}'.\n" | |
| f" 1. Accept the terms at https://huggingface.co/datasets/{config.TASKS_REPO} " | |
| "(browser, one time, instant approval).\n" | |
| " 2. Authenticate: `hf auth login`, or set HF_TOKEN in .env.\n" | |
| "Alternatively pass both --csv-path and --resources-dir to use local files." | |
| ) from exc | |
| return snapshot_dir / config.TASKS_CSV_FILE, snapshot_dir / config.TASKS_RESOURCES_DIR | |
| def create_l2_bench_eval_task( | |
| scorer_setting: ScorerSetting | None = None, | |
| csv_path: Path | None = None, | |
| resources_dir: Path | None = None, | |
| **kwargs # for internal testing params like first_n_samples, sample_range and dataset | |
| ) -> Task: | |
| """Create an inspect-ai Task for L2-Bench evaluation with scoring. | |
| Parameters | |
| ---------- | |
| scorer_setting : ScorerSetting or None | |
| Judge model and generation configuration. Falls back to the production | |
| judge declared in ``config`` when ``None``. | |
| csv_path : Path or None | |
| Path to ``l2-bench_tasks.csv``. Uses the repo default when ``None``. | |
| resources_dir : Path or None | |
| Path to the task resources directory. Uses the repo default when | |
| ``None``. | |
| **kwargs | |
| Internal testing parameters: ``first_n_samples`` (int), | |
| ``sample_range`` (tuple of int), ``dataset`` (MemoryDataset). | |
| Returns | |
| ------- | |
| Task | |
| inspect-ai Task configured for generation and scoring. | |
| """ | |
| if not scorer_setting: | |
| scorer_setting = ScorerSetting(model=config.DEFAULT_JUDGE_MODEL) | |
| if csv_path is None or resources_dir is None: | |
| hf_csv, hf_resources = _fetch_tasks_from_hf() | |
| csv_path = csv_path or hf_csv | |
| resources_dir = resources_dir or hf_resources | |
| clean_dataset = create_inspect_dataset(csv_path, resources_dir) | |
| dataset = clean_dataset | |
| first_n_samples = kwargs.get('first_n_samples') | |
| if first_n_samples and isinstance(first_n_samples, int): | |
| dataset = clean_dataset[:first_n_samples] | |
| sample_range = kwargs.get('sample_range') | |
| if sample_range and isinstance(sample_range, tuple) and len(sample_range) == 2 and all(isinstance(index, int) for index in sample_range): | |
| dataset = clean_dataset[sample_range[0] : sample_range[1]] | |
| task_ids = kwargs.get('task_ids') | |
| if task_ids and isinstance(task_ids, list): | |
| str_task_ids = [str(tid) for tid in task_ids] | |
| clean_ids = {sample.id for sample in clean_dataset.samples} | |
| missing = [tid for tid in str_task_ids if tid not in clean_ids] | |
| if missing and not kwargs.get("dataset"): | |
| raise ValueError(f"Task IDs not found in dataset: {missing}") | |
| dataset = MemoryDataset( | |
| samples=[s for s in clean_dataset.samples if s.id in str_task_ids], | |
| name="l2-bench-samples", | |
| ) | |
| if kwargs.get("dataset") and isinstance(kwargs.get("dataset"), MemoryDataset): | |
| dataset = kwargs.get("dataset") | |
| return Task( | |
| dataset=dataset, | |
| solver=generate(), | |
| scorer=l2_bench_scorer( | |
| csv_path=csv_path, | |
| setting=scorer_setting, | |
| ), | |
| name="l2-bench-eval", | |
| version=1, | |
| metadata={ | |
| "benchmark": "l2-bench", | |
| "purpose": "response_evaluation", | |
| }, | |
| ) | |