Spaces:
Running
Running
Download shellops_data.py from Hoyant-Su/Agentic-RL: direct link, hf CLI and curl.
- Browser
- Download file 4.02 kB
-
https://huggingface.co/spaces/Hoyant-Su/Agentic-RL/resolve/main/shellops_data.py
- Command line
-
hf download hf://spaces/Hoyant-Su/Agentic-RL/shellops_data.py
-
curl -L -o shellops_data.py https://huggingface.co/spaces/Hoyant-Su/Agentic-RL/resolve/main/shellops_data.py
4.02 kB
| import json | |
| from functools import lru_cache | |
| from pathlib import Path | |
| DATA_DIR = Path(__file__).parent / "data" | |
| def _catalog(): | |
| tasks = json.loads((DATA_DIR / "shellops.json").read_text(encoding="utf-8")) | |
| manifest = json.loads((DATA_DIR / "shellops_manifest.json").read_text(encoding="utf-8")) | |
| by_id = {(task["partition"], task["task_id"]): task for task in tasks} | |
| return tasks, manifest, by_id | |
| def dataset_overview() -> dict: | |
| """Inspect ShellOps and ShellOps-Pro task counts, train/test splits, task types, published schemas, source files, license and citation.""" | |
| manifest = _catalog()[1] | |
| overview = {key: manifest[key] for key in ( | |
| "repo_id", "dataset_url", "dataset_card_url", "metadata_url", "license", | |
| "unique_tasks", "partitions", "split_semantics", "train_subset_rows", | |
| "service_scope", "asset_link_scope", "citation", | |
| )} | |
| overview["files"] = [{key: source[key] for key in ( | |
| "repository_path", "partition", "split", "rows", "size_bytes", "url", "schema", "task_types", | |
| )} for source in manifest["files"]] | |
| return overview | |
| def search_tasks(query: str, partition: str = "all", split: str = "all", limit: int = 10, offset: int = 0) -> dict: | |
| """Find real ShellOps CLI benchmark tasks by case-insensitive literal substring in the complete instruction, task ID or published task type. Empty query lists all tasks. Select partition 'all', 'shellops' or 'shellops_pro'; select published split 'all', 'train_src', 'train' or 'test'. Results are ordered by partition then task ID, with explicit pagination and no relevance scoring. The train subset is not double-counted.""" | |
| tasks, manifest, _ = _catalog() | |
| if partition not in {"all", *manifest["partitions"]}: | |
| raise ValueError("partition must be all, shellops or shellops_pro") | |
| if split not in {"all", *(entry["split"] for entry in manifest["files"])}: | |
| raise ValueError("split must be all, train_src, train or test") | |
| if limit < 1 or offset < 0: | |
| raise ValueError("limit must be positive and offset must be nonnegative") | |
| needle = query.casefold() | |
| matches = [ | |
| task for task in tasks | |
| if (partition == "all" or task["partition"] == partition) | |
| and (split == "all" or split in {source["split"] for source in task["sources"]}) | |
| and any(needle in task[field].casefold() for field in ("instruction", "task_id", "task_type")) | |
| ] | |
| page = matches[offset:offset + limit] | |
| return { | |
| "query": query, | |
| "retrieval": "case-insensitive literal substring; ordered by partition and task ID", | |
| "partition": partition, | |
| "split": split, | |
| "total_matches": len(matches), | |
| "limit": limit, | |
| "offset": offset, | |
| "returned": len(page), | |
| "next_offset": offset + len(page) if offset + len(page) < len(matches) else None, | |
| "results": [ | |
| {key: task[key] for key in ("task_id", "partition", "instruction", "task_type", "sources")} | |
| for task in page | |
| ], | |
| "citation": manifest["citation"], | |
| } | |
| def get_task(task_id: str, partition: str) -> dict: | |
| """Inspect one published ShellOps or ShellOps-Pro task by its exact task_id and partition ('shellops' or 'shellops_pro'). Returns the complete instruction, actual reward specification, published reference answer/command, file-entry metadata, pinned parquet rows and workspace asset links. File content is available at the source links. No shell execution or solution verification is performed.""" | |
| _, manifest, by_id = _catalog() | |
| if (partition, task_id) not in by_id: | |
| raise ValueError("No task matches that exact partition and task_id; use search_tasks to find an existing task.") | |
| return { | |
| **by_id[(partition, task_id)], | |
| "dataset_url": manifest["dataset_url"], | |
| "service_scope": manifest["service_scope"], | |
| "asset_link_scope": manifest["asset_link_scope"], | |
| "citation": manifest["citation"], | |
| } | |