Download data/dataset_utils.py from halle01/coder-fake: direct link, hf CLI and curl.
- Browser
- Download file 979 Bytes
-
https://huggingface.co/halle01/coder-fake/resolve/main/data/dataset_utils.py
- Command line
-
hf download hf://halle01/coder-fake/data/dataset_utils.py
-
curl -L -o dataset_utils.py https://huggingface.co/halle01/coder-fake/resolve/main/data/dataset_utils.py
979 Bytes
| """Load the JSONL files build_datasets.py produces into HF `datasets.Dataset` | |
| objects with the `images` column cast to lazily-loaded PIL images (paths on | |
| disk, only opened when a batch actually needs them — matters once the | |
| dataset is tens of thousands of rows).""" | |
| from __future__ import annotations | |
| from datasets import Dataset, Image, Sequence, load_dataset | |
| def load_jsonl_dataset(path: str, split: str | None = None) -> Dataset: | |
| ds = load_dataset("json", data_files=path)["train"] | |
| if split is not None: | |
| ds = ds.filter(lambda row: row["split"] == split) | |
| # "images" is a list of file-path strings in the JSONL; casting to | |
| # Sequence(Image()) makes datasets open them lazily as PIL images. | |
| ds = ds.cast_column("images", Sequence(Image())) | |
| return ds | |
| def dataset_stats(ds: Dataset) -> dict: | |
| rounds = sorted(set(ds["round"])) if "round" in ds.column_names else [] | |
| return {"n_rows": len(ds), "n_rounds": len(rounds), "rounds": rounds} | |