Buckets:
| { | |
| "project_id": "data-agent", | |
| "label": "Data-Agent Bench", | |
| "description": "Verified data-analysis tasks over Kaggle datasets, graded exact / numeric / LLM-judge.", | |
| "source": { | |
| "hf_dataset": "AdithyaSK/data_agent_rl_environment_eval" | |
| }, | |
| "support": { | |
| "claude-code": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.12, pass@1 0.0631, 50/50 tasks measured, mean 9.16 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode", | |
| "+2 tokens per prompt re-render \u2014 harmless for eval, forks every turn when training", | |
| "reads os.environ inside run(); concurrent only via the context-local overlay" | |
| ] | |
| }, | |
| "codex": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.08, pass@1 0.0524, 50/50 tasks measured, mean 13.97 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode" | |
| ] | |
| }, | |
| "gemini-cli": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.1, pass@1 0.0288, 50/50 tasks measured, mean 13.32 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode", | |
| "+2 tokens per prompt re-render \u2014 harmless for eval, forks every turn when training", | |
| "reads os.environ inside run(); concurrent only via the context-local overlay" | |
| ] | |
| }, | |
| "goose": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.08, pass@1 0.0433, 50/50 tasks measured, mean 16.79 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode", | |
| "reads os.environ inside run(); concurrent only via the context-local overlay" | |
| ] | |
| }, | |
| "kimi-cli": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.02, pass@1 0.0048, 50/50 tasks measured, mean 13.3 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode", | |
| "-10 tokens per tool call \u2014 the largest measured skew; unsafe to train on" | |
| ] | |
| }, | |
| "mini-swe-agent": { | |
| "tier": "stable", | |
| "evidence": "pass@4 0.32, pass@1 0.1643, 50/50 tasks measured, mean 11.54 turns" | |
| }, | |
| "openclaw": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.08, pass@1 0.0431, 50/50 tasks measured, mean 8.29 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode" | |
| ] | |
| }, | |
| "opencode": { | |
| "tier": "stable", | |
| "evidence": "pass@4 0.1, pass@1 0.0604, 50/50 tasks measured, mean 14.78 turns" | |
| }, | |
| "openhands-sdk": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.04, pass@1 0.0386, 50/50 tasks measured, mean 13.12 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode" | |
| ] | |
| }, | |
| "pi": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.08, pass@1 0.0631, 50/50 tasks measured, mean 12.06 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode" | |
| ] | |
| }, | |
| "qwen-coder": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.06, pass@1 0.0288, 50/50 tasks measured, mean 9.21 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode" | |
| ] | |
| }, | |
| "swe-agent": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.04, pass@1 0.0096, 50/50 tasks measured, mean 37.14 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode" | |
| ] | |
| }, | |
| "terminus-2": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.2941, pass@1 0.1266, 17/50 tasks measured, mean 26.28 turns", | |
| "caveats": [ | |
| "PAUSED mid-sweep: 8 consecutive unscorable rollouts and never once graded" | |
| ] | |
| }, | |
| "trae-agent": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.08, pass@1 0.0481, 50/50 tasks measured, mean 39.68 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode" | |
| ] | |
| }, | |
| "vibe": { | |
| "tier": "experimental", | |
| "evidence": "pass@4 0.1, pass@1 0.0613, 50/50 tasks measured, mean 14.24 turns", | |
| "caveats": [ | |
| "eval measured but no verified training run \u2014 capture working does not prove the trainer can consume it, which is a separate failure mode" | |
| ] | |
| } | |
| }, | |
| "tier_rule": "stable = graded rollouts on >=90% of tasks AND a verified training run. experimental = anything else, with the gap named. A low pass rate is never a downgrade: a harness scoring 0.0 is reporting a real result, and penalising that would rank harnesses by task difficulty.", | |
| "support_source": [ | |
| "phase0_resumed.json", | |
| "phase0_partA.json", | |
| "phase0_b1.json", | |
| "phase0_b2.json", | |
| "phase0_full.json" | |
| ] | |
| } |
Xet Storage Details
- Size:
- 5.32 kB
- Xet hash:
- 0b92496debb6c38a31c38c009a18f6a8e9becf3cbb42fd560520e0e0bfa086cc
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.