Spaces:
Running on Zero
Running on Zero
Download test_eval_benchmarks.py from Expanded-Repetition/Expanded_Repetition: direct link, hf CLI and curl.
- Browser
- Download file 856 Bytes
-
https://huggingface.co/spaces/Expanded-Repetition/Expanded_Repetition/resolve/main/test_eval_benchmarks.py
- Command line
-
hf download hf://spaces/Expanded-Repetition/Expanded_Repetition/test_eval_benchmarks.py
-
curl -L -o test_eval_benchmarks.py https://huggingface.co/spaces/Expanded-Repetition/Expanded_Repetition/resolve/main/test_eval_benchmarks.py
856 Bytes
| """Offline quality gate for the fixed evaluation benchmark; does not require app.py or a model.""" | |
| import importlib.util | |
| from pathlib import Path | |
| HERE = Path(__file__).resolve().parent | |
| spec = importlib.util.spec_from_file_location("eval_eim", HERE / "eval_eim.py") | |
| module = importlib.util.module_from_spec(spec) | |
| spec.loader.exec_module(module) | |
| problems = module.PROBLEMS | |
| assert len(problems) >= 25, f"benchmark too small: {len(problems)}" | |
| ids = [p["id"] for p in problems] | |
| assert len(ids) == len(set(ids)), "duplicate benchmark IDs" | |
| count = 0 | |
| for problem in problems: | |
| assert len(problem["tests"]) >= 5, f"{problem['id']} has too few tests" | |
| ns = {} | |
| exec(problem["reference"], ns) | |
| for test in problem["tests"]: | |
| exec(test, ns) | |
| count += 1 | |
| print(f"PASS: {len(problems)} unique problems; all {count} reference assertions pass.") | |