"""Load the JSONL files build_datasets.py produces into HF `datasets.Dataset` objects with the `images` column cast to lazily-loaded PIL images (paths on disk, only opened when a batch actually needs them — matters once the dataset is tens of thousands of rows).""" from __future__ import annotations from datasets import Dataset, Image, Sequence, load_dataset def load_jsonl_dataset(path: str, split: str | None = None) -> Dataset: ds = load_dataset("json", data_files=path)["train"] if split is not None: ds = ds.filter(lambda row: row["split"] == split) # "images" is a list of file-path strings in the JSONL; casting to # Sequence(Image()) makes datasets open them lazily as PIL images. ds = ds.cast_column("images", Sequence(Image())) return ds def dataset_stats(ds: Dataset) -> dict: rounds = sorted(set(ds["round"])) if "round" in ds.column_names else [] return {"n_rows": len(ds), "n_rounds": len(rounds), "rounds": rounds}