jimmyedgell's picture
L2-Bench Pipeline v1.0.0: initial open-source release
224d30c
Raw
History Blame Contribute Delete
4.57 kB
"""
inspect-ai Task definition for L2-Bench.
Pairs the generate() solver with the LLM-as-judge scorer, sourcing task data
from the gated L2-Bench dataset on the Hugging Face Hub.
Version | Date | Author | Change comment
--------|------------|-----------|---------------
1.0.0 | 2026-07-29 | M. Ku | Initial open-source release
"""
import os
from pathlib import Path
from huggingface_hub import snapshot_download
from huggingface_hub.errors import GatedRepoError, HfHubHTTPError
from inspect_ai import Task
from inspect_ai.dataset import MemoryDataset
from inspect_ai.solver import generate
from l2_bench_eval import config
from l2_bench_eval.dataset import create_inspect_dataset
from l2_bench_eval.score import ScorerSetting, l2_bench_scorer
def _fetch_tasks_from_hf() -> tuple[Path, Path]:
"""Download (or reuse cached) task CSV + resources from the HF dataset repo."""
try:
snapshot_dir = Path(
snapshot_download(
repo_id=config.TASKS_REPO,
repo_type="dataset",
token=os.environ.get("HF_TOKEN"),
)
)
except (GatedRepoError, HfHubHTTPError) as exc:
raise RuntimeError(
f"Could not download the gated dataset '{config.TASKS_REPO}'.\n"
f" 1. Accept the terms at https://huggingface.co/datasets/{config.TASKS_REPO} "
"(browser, one time, instant approval).\n"
" 2. Authenticate: `hf auth login`, or set HF_TOKEN in .env.\n"
"Alternatively pass both --csv-path and --resources-dir to use local files."
) from exc
return snapshot_dir / config.TASKS_CSV_FILE, snapshot_dir / config.TASKS_RESOURCES_DIR
def create_l2_bench_eval_task(
scorer_setting: ScorerSetting | None = None,
csv_path: Path | None = None,
resources_dir: Path | None = None,
**kwargs # for internal testing params like first_n_samples, sample_range and dataset
) -> Task:
"""Create an inspect-ai Task for L2-Bench evaluation with scoring.
Parameters
----------
scorer_setting : ScorerSetting or None
Judge model and generation configuration. Falls back to the production
judge declared in ``config`` when ``None``.
csv_path : Path or None
Path to ``l2-bench_tasks.csv``. Uses the repo default when ``None``.
resources_dir : Path or None
Path to the task resources directory. Uses the repo default when
``None``.
**kwargs
Internal testing parameters: ``first_n_samples`` (int),
``sample_range`` (tuple of int), ``dataset`` (MemoryDataset).
Returns
-------
Task
inspect-ai Task configured for generation and scoring.
"""
if not scorer_setting:
scorer_setting = ScorerSetting(model=config.DEFAULT_JUDGE_MODEL)
if csv_path is None or resources_dir is None:
hf_csv, hf_resources = _fetch_tasks_from_hf()
csv_path = csv_path or hf_csv
resources_dir = resources_dir or hf_resources
clean_dataset = create_inspect_dataset(csv_path, resources_dir)
dataset = clean_dataset
first_n_samples = kwargs.get('first_n_samples')
if first_n_samples and isinstance(first_n_samples, int):
dataset = clean_dataset[:first_n_samples]
sample_range = kwargs.get('sample_range')
if sample_range and isinstance(sample_range, tuple) and len(sample_range) == 2 and all(isinstance(index, int) for index in sample_range):
dataset = clean_dataset[sample_range[0] : sample_range[1]]
task_ids = kwargs.get('task_ids')
if task_ids and isinstance(task_ids, list):
str_task_ids = [str(tid) for tid in task_ids]
clean_ids = {sample.id for sample in clean_dataset.samples}
missing = [tid for tid in str_task_ids if tid not in clean_ids]
if missing and not kwargs.get("dataset"):
raise ValueError(f"Task IDs not found in dataset: {missing}")
dataset = MemoryDataset(
samples=[s for s in clean_dataset.samples if s.id in str_task_ids],
name="l2-bench-samples",
)
if kwargs.get("dataset") and isinstance(kwargs.get("dataset"), MemoryDataset):
dataset = kwargs.get("dataset")
return Task(
dataset=dataset,
solver=generate(),
scorer=l2_bench_scorer(
csv_path=csv_path,
setting=scorer_setting,
),
name="l2-bench-eval",
version=1,
metadata={
"benchmark": "l2-bench",
"purpose": "response_evaluation",
},
)