Spaces:
Running
Running
Publish Codex Benchmark evaluation evidence
Browse files- README.md +1 -1
- REPORT.md +1 -1
- comparison.json +6 -6
- data.json +283 -32
- episodes.csv +1 -1
- episodes.json +251 -0
- manifest.json +9 -9
- protocol.json +11 -11
- publication-manifest.json +15 -15
- summary.json +16 -16
- task-index.json +4 -4
README.md
CHANGED
|
@@ -10,7 +10,7 @@ pinned: false
|
|
| 10 |
|
| 11 |
# Codex Benchmark
|
| 12 |
|
| 13 |
-
|
| 14 |
|
| 15 |
Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
|
| 16 |
|
|
|
|
| 10 |
|
| 11 |
# Codex Benchmark
|
| 12 |
|
| 13 |
+
39/42 ordinary RoboDojo tasks published: 31 successes, 8 valid native failures, 3 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
|
| 14 |
|
| 15 |
Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
|
| 16 |
|
REPORT.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
# Codex Benchmark
|
| 2 |
|
| 3 |
-
|
| 4 |
|
| 5 |
Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
|
| 6 |
|
|
|
|
| 1 |
# Codex Benchmark
|
| 2 |
|
| 3 |
+
39/42 ordinary RoboDojo tasks published: 31 successes, 8 valid native failures, 3 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
|
| 4 |
|
| 5 |
Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
|
| 6 |
|
comparison.json
CHANGED
|
@@ -1,10 +1,10 @@
|
|
| 1 |
{
|
| 2 |
"baseline_revision": "1a7f1c9f178ba7aeb659e5b527a77978e9136072",
|
| 3 |
"summary": {
|
| 4 |
-
"completed":
|
| 5 |
-
"codex_successes":
|
| 6 |
-
"codex_success_rate_completed_subset": 0.
|
| 7 |
-
"kinex_successes_same_subset":
|
| 8 |
"kinex_successes_all42": 35,
|
| 9 |
"all42_comparison_complete": false
|
| 10 |
},
|
|
@@ -60,8 +60,8 @@
|
|
| 60 |
{
|
| 61 |
"task_key": "task04/07",
|
| 62 |
"native_id": "robodojo/insert-tubes",
|
| 63 |
-
"status": "
|
| 64 |
-
"codex_success":
|
| 65 |
"kinex_success": true,
|
| 66 |
"kinex_version": "0.10.3"
|
| 67 |
},
|
|
|
|
| 1 |
{
|
| 2 |
"baseline_revision": "1a7f1c9f178ba7aeb659e5b527a77978e9136072",
|
| 3 |
"summary": {
|
| 4 |
+
"completed": 39,
|
| 5 |
+
"codex_successes": 31,
|
| 6 |
+
"codex_success_rate_completed_subset": 0.7948717948717948,
|
| 7 |
+
"kinex_successes_same_subset": 34,
|
| 8 |
"kinex_successes_all42": 35,
|
| 9 |
"all42_comparison_complete": false
|
| 10 |
},
|
|
|
|
| 60 |
{
|
| 61 |
"task_key": "task04/07",
|
| 62 |
"native_id": "robodojo/insert-tubes",
|
| 63 |
+
"status": "completed",
|
| 64 |
+
"codex_success": true,
|
| 65 |
"kinex_success": true,
|
| 66 |
"kinex_version": "0.10.3"
|
| 67 |
},
|
data.json
CHANGED
|
@@ -6,7 +6,7 @@
|
|
| 6 |
"benchmark_complete": false,
|
| 7 |
"edition": "plain-codex-astra-high-seed0",
|
| 8 |
"created_at": "2026-10-09T00:09:42.989838+00:00",
|
| 9 |
-
"updated_at": "2026-10-
|
| 10 |
"model": "gpt-6-astra",
|
| 11 |
"effort": "high",
|
| 12 |
"seed": 0,
|
|
@@ -17,38 +17,38 @@
|
|
| 17 |
},
|
| 18 |
"summary": {
|
| 19 |
"planned_tasks": 42,
|
| 20 |
-
"published_results":
|
| 21 |
-
"pending_tasks":
|
| 22 |
"families": [
|
| 23 |
{
|
| 24 |
"id": "task04",
|
| 25 |
"name": "RoboDojo",
|
| 26 |
"total": 42,
|
| 27 |
-
"completed":
|
| 28 |
-
"pending":
|
| 29 |
-
"successes":
|
| 30 |
-
"valid_results":
|
| 31 |
-
"success_rate": 0.
|
| 32 |
-
"input_tokens":
|
| 33 |
-
"cached_input_tokens":
|
| 34 |
-
"output_tokens":
|
| 35 |
-
"usage_complete":
|
| 36 |
"control_frequency_hz": 25,
|
| 37 |
"max_control_steps": 7500,
|
| 38 |
"preflight_results": 0,
|
| 39 |
-
"formal_results":
|
| 40 |
-
"modified_results":
|
| 41 |
"original_results": 8
|
| 42 |
}
|
| 43 |
],
|
| 44 |
"progress": {
|
| 45 |
-
"finished":
|
| 46 |
"interrupted": 2,
|
| 47 |
"native_failures": 8,
|
| 48 |
-
"native_successes":
|
| 49 |
"needs_review": 0,
|
| 50 |
"queued": 0,
|
| 51 |
-
"running":
|
| 52 |
},
|
| 53 |
"interrupted_attempts": 4,
|
| 54 |
"usage_incomplete_tasks": [],
|
|
@@ -59,20 +59,20 @@
|
|
| 59 |
"id": "task04",
|
| 60 |
"name": "RoboDojo",
|
| 61 |
"total": 42,
|
| 62 |
-
"completed":
|
| 63 |
-
"pending":
|
| 64 |
-
"successes":
|
| 65 |
-
"valid_results":
|
| 66 |
-
"success_rate": 0.
|
| 67 |
-
"input_tokens":
|
| 68 |
-
"cached_input_tokens":
|
| 69 |
-
"output_tokens":
|
| 70 |
-
"usage_complete":
|
| 71 |
"control_frequency_hz": 25,
|
| 72 |
"max_control_steps": 7500,
|
| 73 |
"preflight_results": 0,
|
| 74 |
-
"formal_results":
|
| 75 |
-
"modified_results":
|
| 76 |
"original_results": 8
|
| 77 |
}
|
| 78 |
],
|
|
@@ -670,8 +670,8 @@
|
|
| 670 |
"catalog_instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
|
| 671 |
"native_instruction": "Insert the three tubes into the rack one by one.",
|
| 672 |
"instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
|
| 673 |
-
"status": "
|
| 674 |
-
"episode_id":
|
| 675 |
"planned_protocol": {
|
| 676 |
"episodes": 1,
|
| 677 |
"seed": 0,
|
|
@@ -727,8 +727,8 @@
|
|
| 727 |
},
|
| 728 |
"display_slot": "07",
|
| 729 |
"display_key": "task04/07",
|
| 730 |
-
"run_status": "
|
| 731 |
-
"status_note": "
|
| 732 |
"attempt_history": [
|
| 733 |
{
|
| 734 |
"id": "07-robodojo-insert-tubes-codex-seed0-attempt01",
|
|
@@ -5123,6 +5123,257 @@
|
|
| 5123 |
"selected_for_formal_metrics": true,
|
| 5124 |
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/06/seed-0/"
|
| 5125 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5126 |
{
|
| 5127 |
"id": "task04-08-seed0-formal",
|
| 5128 |
"task_key": "task04/08",
|
|
|
|
| 6 |
"benchmark_complete": false,
|
| 7 |
"edition": "plain-codex-astra-high-seed0",
|
| 8 |
"created_at": "2026-10-09T00:09:42.989838+00:00",
|
| 9 |
+
"updated_at": "2026-10-09T05:17:09.534508+00:00",
|
| 10 |
"model": "gpt-6-astra",
|
| 11 |
"effort": "high",
|
| 12 |
"seed": 0,
|
|
|
|
| 17 |
},
|
| 18 |
"summary": {
|
| 19 |
"planned_tasks": 42,
|
| 20 |
+
"published_results": 39,
|
| 21 |
+
"pending_tasks": 3,
|
| 22 |
"families": [
|
| 23 |
{
|
| 24 |
"id": "task04",
|
| 25 |
"name": "RoboDojo",
|
| 26 |
"total": 42,
|
| 27 |
+
"completed": 39,
|
| 28 |
+
"pending": 3,
|
| 29 |
+
"successes": 31,
|
| 30 |
+
"valid_results": 39,
|
| 31 |
+
"success_rate": 0.7948717948717948,
|
| 32 |
+
"input_tokens": 229797708,
|
| 33 |
+
"cached_input_tokens": 225752320,
|
| 34 |
+
"output_tokens": 869279,
|
| 35 |
+
"usage_complete": 39,
|
| 36 |
"control_frequency_hz": 25,
|
| 37 |
"max_control_steps": 7500,
|
| 38 |
"preflight_results": 0,
|
| 39 |
+
"formal_results": 39,
|
| 40 |
+
"modified_results": 31,
|
| 41 |
"original_results": 8
|
| 42 |
}
|
| 43 |
],
|
| 44 |
"progress": {
|
| 45 |
+
"finished": 39,
|
| 46 |
"interrupted": 2,
|
| 47 |
"native_failures": 8,
|
| 48 |
+
"native_successes": 31,
|
| 49 |
"needs_review": 0,
|
| 50 |
"queued": 0,
|
| 51 |
+
"running": 1
|
| 52 |
},
|
| 53 |
"interrupted_attempts": 4,
|
| 54 |
"usage_incomplete_tasks": [],
|
|
|
|
| 59 |
"id": "task04",
|
| 60 |
"name": "RoboDojo",
|
| 61 |
"total": 42,
|
| 62 |
+
"completed": 39,
|
| 63 |
+
"pending": 3,
|
| 64 |
+
"successes": 31,
|
| 65 |
+
"valid_results": 39,
|
| 66 |
+
"success_rate": 0.7948717948717948,
|
| 67 |
+
"input_tokens": 229797708,
|
| 68 |
+
"cached_input_tokens": 225752320,
|
| 69 |
+
"output_tokens": 869279,
|
| 70 |
+
"usage_complete": 39,
|
| 71 |
"control_frequency_hz": 25,
|
| 72 |
"max_control_steps": 7500,
|
| 73 |
"preflight_results": 0,
|
| 74 |
+
"formal_results": 39,
|
| 75 |
+
"modified_results": 31,
|
| 76 |
"original_results": 8
|
| 77 |
}
|
| 78 |
],
|
|
|
|
| 670 |
"catalog_instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
|
| 671 |
"native_instruction": "Insert the three tubes into the rack one by one.",
|
| 672 |
"instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
|
| 673 |
+
"status": "completed",
|
| 674 |
+
"episode_id": "task04-07-seed0-formal",
|
| 675 |
"planned_protocol": {
|
| 676 |
"episodes": 1,
|
| 677 |
"seed": 0,
|
|
|
|
| 727 |
},
|
| 728 |
"display_slot": "07",
|
| 729 |
"display_key": "task04/07",
|
| 730 |
+
"run_status": "finished",
|
| 731 |
+
"status_note": "Queued for evaluation.",
|
| 732 |
"attempt_history": [
|
| 733 |
{
|
| 734 |
"id": "07-robodojo-insert-tubes-codex-seed0-attempt01",
|
|
|
|
| 5123 |
"selected_for_formal_metrics": true,
|
| 5124 |
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/06/seed-0/"
|
| 5125 |
},
|
| 5126 |
+
{
|
| 5127 |
+
"id": "task04-07-seed0-formal",
|
| 5128 |
+
"task_key": "task04/07",
|
| 5129 |
+
"family": "task04",
|
| 5130 |
+
"slot": "07",
|
| 5131 |
+
"seed": 0,
|
| 5132 |
+
"episode": 1,
|
| 5133 |
+
"phase": "formal",
|
| 5134 |
+
"status": "completed",
|
| 5135 |
+
"success": true,
|
| 5136 |
+
"native_reward": 1.0,
|
| 5137 |
+
"valid": true,
|
| 5138 |
+
"execution": {
|
| 5139 |
+
"reason": null,
|
| 5140 |
+
"status": "finished"
|
| 5141 |
+
},
|
| 5142 |
+
"verdict": {
|
| 5143 |
+
"evidence_valid": true,
|
| 5144 |
+
"steps": 4162,
|
| 5145 |
+
"success": true,
|
| 5146 |
+
"termination": "success"
|
| 5147 |
+
},
|
| 5148 |
+
"steps": 4162,
|
| 5149 |
+
"simulation_time_s": null,
|
| 5150 |
+
"wall_time_s": 2428.860172,
|
| 5151 |
+
"model": "gpt-6-astra",
|
| 5152 |
+
"effort": "high",
|
| 5153 |
+
"harness": "codex",
|
| 5154 |
+
"codex_version": "0.160.0",
|
| 5155 |
+
"native_instruction": "Insert the three tubes into the rack one by one.",
|
| 5156 |
+
"instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
|
| 5157 |
+
"instruction_policy": "modified",
|
| 5158 |
+
"usage": {
|
| 5159 |
+
"accounting": "reported-responses",
|
| 5160 |
+
"audit_complete": true,
|
| 5161 |
+
"cache_hit_rate": 0.9885433078393561,
|
| 5162 |
+
"cache_reported_input_tokens": 11924908,
|
| 5163 |
+
"cache_write_input_tokens": 0,
|
| 5164 |
+
"cache_write_reported_input_tokens": 11924908,
|
| 5165 |
+
"cached_input_tokens": 11788288,
|
| 5166 |
+
"completed_turns": 1,
|
| 5167 |
+
"cost_usd": null,
|
| 5168 |
+
"failed_turns": 0,
|
| 5169 |
+
"input_tokens": 11924908,
|
| 5170 |
+
"known_cache_write_input_tokens": 0,
|
| 5171 |
+
"known_cached_input_tokens": 11788288,
|
| 5172 |
+
"known_input_tokens": 11924908,
|
| 5173 |
+
"known_output_tokens": 45245,
|
| 5174 |
+
"known_reasoning_output_tokens": 23793,
|
| 5175 |
+
"output_tokens": 45245,
|
| 5176 |
+
"reasoning_output_tokens": 23793,
|
| 5177 |
+
"reasoning_reported_output_tokens": 45245,
|
| 5178 |
+
"reported_responses": {
|
| 5179 |
+
"cache_reported_input_tokens": 168,
|
| 5180 |
+
"cache_write_input_tokens": 168,
|
| 5181 |
+
"cache_write_reported_input_tokens": 168,
|
| 5182 |
+
"cached_input_tokens": 168,
|
| 5183 |
+
"input_tokens": 168,
|
| 5184 |
+
"output_tokens": 168,
|
| 5185 |
+
"reasoning_output_tokens": 168,
|
| 5186 |
+
"reasoning_reported_output_tokens": 168
|
| 5187 |
+
},
|
| 5188 |
+
"response_count": 168,
|
| 5189 |
+
"response_ids_complete": true,
|
| 5190 |
+
"schema": "rlebench/token-usage/1",
|
| 5191 |
+
"source": "Codex token_usage_record per response",
|
| 5192 |
+
"uncached_input_tokens": 136620,
|
| 5193 |
+
"unidentified_usage_records": 0
|
| 5194 |
+
},
|
| 5195 |
+
"call_activity": {
|
| 5196 |
+
"model_tool_calls": 167,
|
| 5197 |
+
"model_tool_calls_by_name": {
|
| 5198 |
+
"exec": 167
|
| 5199 |
+
},
|
| 5200 |
+
"nested_python_tool_invocations": null,
|
| 5201 |
+
"python_device_rpc_attempts": null,
|
| 5202 |
+
"python_device_rpc_attempts_by_action": null,
|
| 5203 |
+
"python_device_rpc_errors": null,
|
| 5204 |
+
"python_instrumented_model_tool_calls": null,
|
| 5205 |
+
"python_tool_invocations": null,
|
| 5206 |
+
"python_tool_invocations_by_origin": null,
|
| 5207 |
+
"schema": "rlebench/call-activity/1",
|
| 5208 |
+
"source": "Codex native sessions"
|
| 5209 |
+
},
|
| 5210 |
+
"media": {
|
| 5211 |
+
"passed": true,
|
| 5212 |
+
"width": 2880,
|
| 5213 |
+
"height": 720,
|
| 5214 |
+
"duration_s": 41.6,
|
| 5215 |
+
"speed": 4,
|
| 5216 |
+
"source_fps": 10,
|
| 5217 |
+
"output_fps": 20,
|
| 5218 |
+
"recording": {
|
| 5219 |
+
"accepted_samples": 1666,
|
| 5220 |
+
"captured_samples": 1666,
|
| 5221 |
+
"clock": "simulation",
|
| 5222 |
+
"dropped_samples": 0,
|
| 5223 |
+
"encoded_frames": 1665,
|
| 5224 |
+
"end_time_s": 166.48000000000116,
|
| 5225 |
+
"error": null,
|
| 5226 |
+
"experimental": true,
|
| 5227 |
+
"fps": 10,
|
| 5228 |
+
"received_samples": 1666,
|
| 5229 |
+
"schema": "roboenv/recording/1",
|
| 5230 |
+
"state": "closed",
|
| 5231 |
+
"status": "complete",
|
| 5232 |
+
"views": [
|
| 5233 |
+
{
|
| 5234 |
+
"fov_y": 45.0,
|
| 5235 |
+
"height": 720,
|
| 5236 |
+
"name": "third_person",
|
| 5237 |
+
"pose": null,
|
| 5238 |
+
"source": "third_person",
|
| 5239 |
+
"width": 960
|
| 5240 |
+
},
|
| 5241 |
+
{
|
| 5242 |
+
"fov_y": 45.0,
|
| 5243 |
+
"height": 720,
|
| 5244 |
+
"name": "left_wrist",
|
| 5245 |
+
"pose": null,
|
| 5246 |
+
"source": "left_wrist",
|
| 5247 |
+
"width": 960
|
| 5248 |
+
},
|
| 5249 |
+
{
|
| 5250 |
+
"fov_y": 45.0,
|
| 5251 |
+
"height": 720,
|
| 5252 |
+
"name": "right_wrist",
|
| 5253 |
+
"pose": null,
|
| 5254 |
+
"source": "right_wrist",
|
| 5255 |
+
"width": 960
|
| 5256 |
+
}
|
| 5257 |
+
]
|
| 5258 |
+
},
|
| 5259 |
+
"view_names": [
|
| 5260 |
+
"third_person",
|
| 5261 |
+
"left_wrist",
|
| 5262 |
+
"right_wrist"
|
| 5263 |
+
],
|
| 5264 |
+
"sha256": "0eceba672ce4ecdeea98aa372cce1c4d7dd69cf118c2766b39ad4fee0204a3fb",
|
| 5265 |
+
"scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
|
| 5266 |
+
},
|
| 5267 |
+
"analysis": {
|
| 5268 |
+
"outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
|
| 5269 |
+
"duration": "\u4f7f\u7528 4,162 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
|
| 5270 |
+
"execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
|
| 5271 |
+
},
|
| 5272 |
+
"provenance": {
|
| 5273 |
+
"sources": {
|
| 5274 |
+
"RLE-Bench-inhouse": {
|
| 5275 |
+
"build_inputs": [
|
| 5276 |
+
"pyproject.toml",
|
| 5277 |
+
"src",
|
| 5278 |
+
"tasks",
|
| 5279 |
+
"README.md",
|
| 5280 |
+
"Makefile",
|
| 5281 |
+
"tests",
|
| 5282 |
+
"docs"
|
| 5283 |
+
],
|
| 5284 |
+
"content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
|
| 5285 |
+
"revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
|
| 5286 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
|
| 5287 |
+
},
|
| 5288 |
+
"RoboEnv": {
|
| 5289 |
+
"build_inputs": [
|
| 5290 |
+
"pyproject.toml",
|
| 5291 |
+
"README.md",
|
| 5292 |
+
"src",
|
| 5293 |
+
"runtime/pyproject.toml",
|
| 5294 |
+
"runtime/README.md",
|
| 5295 |
+
"runtime/src",
|
| 5296 |
+
"runtime/environments.json",
|
| 5297 |
+
"runtime/locks",
|
| 5298 |
+
"catalog",
|
| 5299 |
+
"upstreams.lock.json",
|
| 5300 |
+
"third_party/patches",
|
| 5301 |
+
"docs/validation"
|
| 5302 |
+
],
|
| 5303 |
+
"content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
|
| 5304 |
+
"revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
|
| 5305 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
|
| 5306 |
+
}
|
| 5307 |
+
},
|
| 5308 |
+
"job": "07-robodojo-insert-tubes-codex-seed0-attempt02",
|
| 5309 |
+
"attempt": 2,
|
| 5310 |
+
"harness": "stock Codex CLI",
|
| 5311 |
+
"codex_version": "0.160.0",
|
| 5312 |
+
"model": "gpt-6-astra",
|
| 5313 |
+
"effort": "high",
|
| 5314 |
+
"service_tier": "default",
|
| 5315 |
+
"fresh_session": true,
|
| 5316 |
+
"source_jobs": [],
|
| 5317 |
+
"resume_trajectory": false,
|
| 5318 |
+
"imported_skills": [],
|
| 5319 |
+
"automatic_harbor_retries": 0,
|
| 5320 |
+
"request_policy": {
|
| 5321 |
+
"max_request_retries": 50,
|
| 5322 |
+
"configuration": "explicit retry50 SSE",
|
| 5323 |
+
"usage_accounting": "reported-responses"
|
| 5324 |
+
},
|
| 5325 |
+
"classification": "formal",
|
| 5326 |
+
"measured_images": {
|
| 5327 |
+
"agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
|
| 5328 |
+
"task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
|
| 5329 |
+
},
|
| 5330 |
+
"session_original_sha256": "dced20541deac3be24265775d5b357aadecb6b53ad8b991fbf38cfcb1a690cfd",
|
| 5331 |
+
"protocol_sha256": "331cfc33f7aa656ce04ffa0d8e7927d625ea36dd825616041760649c239cee4b"
|
| 5332 |
+
},
|
| 5333 |
+
"links": {
|
| 5334 |
+
"native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/session.jsonl",
|
| 5335 |
+
"trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/trajectory.json",
|
| 5336 |
+
"provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/provider-usage.jsonl",
|
| 5337 |
+
"transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/transcript.json",
|
| 5338 |
+
"verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/episode.json",
|
| 5339 |
+
"protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/protocol.json",
|
| 5340 |
+
"native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/instructions.json",
|
| 5341 |
+
"workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/workspace.tar.gz",
|
| 5342 |
+
"workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/workspace.json",
|
| 5343 |
+
"owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
|
| 5344 |
+
"recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/evidence/episode/evidence/recording/manifest.json",
|
| 5345 |
+
"usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/usage.json",
|
| 5346 |
+
"analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/analysis.json",
|
| 5347 |
+
"provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/provenance.json",
|
| 5348 |
+
"video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/video.mp4",
|
| 5349 |
+
"poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/poster.jpg",
|
| 5350 |
+
"media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/media-validation.json"
|
| 5351 |
+
},
|
| 5352 |
+
"resources": [
|
| 5353 |
+
{
|
| 5354 |
+
"name": "tools/robot.py",
|
| 5355 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/tools/robot.py",
|
| 5356 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 5357 |
+
},
|
| 5358 |
+
{
|
| 5359 |
+
"name": "tools/tubes.py",
|
| 5360 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/tools/tubes.py",
|
| 5361 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 5362 |
+
},
|
| 5363 |
+
{
|
| 5364 |
+
"name": "memos/insert-tubes.md",
|
| 5365 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/memos/insert-tubes.md",
|
| 5366 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 5367 |
+
}
|
| 5368 |
+
],
|
| 5369 |
+
"session_counts": {
|
| 5370 |
+
"visible_events": 362,
|
| 5371 |
+
"observed_images": 84,
|
| 5372 |
+
"tool_errors": 4
|
| 5373 |
+
},
|
| 5374 |
+
"selected_for_formal_metrics": true,
|
| 5375 |
+
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/"
|
| 5376 |
+
},
|
| 5377 |
{
|
| 5378 |
"id": "task04-08-seed0-formal",
|
| 5379 |
"task_key": "task04/08",
|
episodes.csv
CHANGED
|
@@ -5,7 +5,7 @@ task04/03,robodojo/store-laptop-and-headphones,completed,finished,False,5696,181
|
|
| 5 |
task04/04,robodojo/cover-blocks,completed,finished,True,2184,2163308,2107392,9584
|
| 6 |
task04/05,robodojo/play-xylophone,completed,finished,True,818,1397747,1321344,9368
|
| 7 |
task04/06,robodojo/store-tools-in-toolbox,completed,finished,False,7421,15593052,15416704,57648
|
| 8 |
-
task04/07,robodojo/insert-tubes,
|
| 9 |
task04/08,robodojo/deposit-coin,completed,finished,True,1432,2957393,2867840,15930
|
| 10 |
task04/09,robodojo/fasten-screws,completed,finished,True,7340,5986459,5888000,23452
|
| 11 |
task04/10,robodojo/play-stacking-toy,completed,finished,True,2460,4468354,4390912,20623
|
|
|
|
| 5 |
task04/04,robodojo/cover-blocks,completed,finished,True,2184,2163308,2107392,9584
|
| 6 |
task04/05,robodojo/play-xylophone,completed,finished,True,818,1397747,1321344,9368
|
| 7 |
task04/06,robodojo/store-tools-in-toolbox,completed,finished,False,7421,15593052,15416704,57648
|
| 8 |
+
task04/07,robodojo/insert-tubes,completed,finished,True,4162,11924908,11788288,45245
|
| 9 |
task04/08,robodojo/deposit-coin,completed,finished,True,1432,2957393,2867840,15930
|
| 10 |
task04/09,robodojo/fasten-screws,completed,finished,True,7340,5986459,5888000,23452
|
| 11 |
task04/10,robodojo/play-stacking-toy,completed,finished,True,2460,4468354,4390912,20623
|
episodes.json
CHANGED
|
@@ -1239,6 +1239,257 @@
|
|
| 1239 |
"selected_for_formal_metrics": true,
|
| 1240 |
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/06/seed-0/"
|
| 1241 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1242 |
{
|
| 1243 |
"id": "task04-08-seed0-formal",
|
| 1244 |
"task_key": "task04/08",
|
|
|
|
| 1239 |
"selected_for_formal_metrics": true,
|
| 1240 |
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/06/seed-0/"
|
| 1241 |
},
|
| 1242 |
+
{
|
| 1243 |
+
"id": "task04-07-seed0-formal",
|
| 1244 |
+
"task_key": "task04/07",
|
| 1245 |
+
"family": "task04",
|
| 1246 |
+
"slot": "07",
|
| 1247 |
+
"seed": 0,
|
| 1248 |
+
"episode": 1,
|
| 1249 |
+
"phase": "formal",
|
| 1250 |
+
"status": "completed",
|
| 1251 |
+
"success": true,
|
| 1252 |
+
"native_reward": 1.0,
|
| 1253 |
+
"valid": true,
|
| 1254 |
+
"execution": {
|
| 1255 |
+
"reason": null,
|
| 1256 |
+
"status": "finished"
|
| 1257 |
+
},
|
| 1258 |
+
"verdict": {
|
| 1259 |
+
"evidence_valid": true,
|
| 1260 |
+
"steps": 4162,
|
| 1261 |
+
"success": true,
|
| 1262 |
+
"termination": "success"
|
| 1263 |
+
},
|
| 1264 |
+
"steps": 4162,
|
| 1265 |
+
"simulation_time_s": null,
|
| 1266 |
+
"wall_time_s": 2428.860172,
|
| 1267 |
+
"model": "gpt-6-astra",
|
| 1268 |
+
"effort": "high",
|
| 1269 |
+
"harness": "codex",
|
| 1270 |
+
"codex_version": "0.160.0",
|
| 1271 |
+
"native_instruction": "Insert the three tubes into the rack one by one.",
|
| 1272 |
+
"instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
|
| 1273 |
+
"instruction_policy": "modified",
|
| 1274 |
+
"usage": {
|
| 1275 |
+
"accounting": "reported-responses",
|
| 1276 |
+
"audit_complete": true,
|
| 1277 |
+
"cache_hit_rate": 0.9885433078393561,
|
| 1278 |
+
"cache_reported_input_tokens": 11924908,
|
| 1279 |
+
"cache_write_input_tokens": 0,
|
| 1280 |
+
"cache_write_reported_input_tokens": 11924908,
|
| 1281 |
+
"cached_input_tokens": 11788288,
|
| 1282 |
+
"completed_turns": 1,
|
| 1283 |
+
"cost_usd": null,
|
| 1284 |
+
"failed_turns": 0,
|
| 1285 |
+
"input_tokens": 11924908,
|
| 1286 |
+
"known_cache_write_input_tokens": 0,
|
| 1287 |
+
"known_cached_input_tokens": 11788288,
|
| 1288 |
+
"known_input_tokens": 11924908,
|
| 1289 |
+
"known_output_tokens": 45245,
|
| 1290 |
+
"known_reasoning_output_tokens": 23793,
|
| 1291 |
+
"output_tokens": 45245,
|
| 1292 |
+
"reasoning_output_tokens": 23793,
|
| 1293 |
+
"reasoning_reported_output_tokens": 45245,
|
| 1294 |
+
"reported_responses": {
|
| 1295 |
+
"cache_reported_input_tokens": 168,
|
| 1296 |
+
"cache_write_input_tokens": 168,
|
| 1297 |
+
"cache_write_reported_input_tokens": 168,
|
| 1298 |
+
"cached_input_tokens": 168,
|
| 1299 |
+
"input_tokens": 168,
|
| 1300 |
+
"output_tokens": 168,
|
| 1301 |
+
"reasoning_output_tokens": 168,
|
| 1302 |
+
"reasoning_reported_output_tokens": 168
|
| 1303 |
+
},
|
| 1304 |
+
"response_count": 168,
|
| 1305 |
+
"response_ids_complete": true,
|
| 1306 |
+
"schema": "rlebench/token-usage/1",
|
| 1307 |
+
"source": "Codex token_usage_record per response",
|
| 1308 |
+
"uncached_input_tokens": 136620,
|
| 1309 |
+
"unidentified_usage_records": 0
|
| 1310 |
+
},
|
| 1311 |
+
"call_activity": {
|
| 1312 |
+
"model_tool_calls": 167,
|
| 1313 |
+
"model_tool_calls_by_name": {
|
| 1314 |
+
"exec": 167
|
| 1315 |
+
},
|
| 1316 |
+
"nested_python_tool_invocations": null,
|
| 1317 |
+
"python_device_rpc_attempts": null,
|
| 1318 |
+
"python_device_rpc_attempts_by_action": null,
|
| 1319 |
+
"python_device_rpc_errors": null,
|
| 1320 |
+
"python_instrumented_model_tool_calls": null,
|
| 1321 |
+
"python_tool_invocations": null,
|
| 1322 |
+
"python_tool_invocations_by_origin": null,
|
| 1323 |
+
"schema": "rlebench/call-activity/1",
|
| 1324 |
+
"source": "Codex native sessions"
|
| 1325 |
+
},
|
| 1326 |
+
"media": {
|
| 1327 |
+
"passed": true,
|
| 1328 |
+
"width": 2880,
|
| 1329 |
+
"height": 720,
|
| 1330 |
+
"duration_s": 41.6,
|
| 1331 |
+
"speed": 4,
|
| 1332 |
+
"source_fps": 10,
|
| 1333 |
+
"output_fps": 20,
|
| 1334 |
+
"recording": {
|
| 1335 |
+
"accepted_samples": 1666,
|
| 1336 |
+
"captured_samples": 1666,
|
| 1337 |
+
"clock": "simulation",
|
| 1338 |
+
"dropped_samples": 0,
|
| 1339 |
+
"encoded_frames": 1665,
|
| 1340 |
+
"end_time_s": 166.48000000000116,
|
| 1341 |
+
"error": null,
|
| 1342 |
+
"experimental": true,
|
| 1343 |
+
"fps": 10,
|
| 1344 |
+
"received_samples": 1666,
|
| 1345 |
+
"schema": "roboenv/recording/1",
|
| 1346 |
+
"state": "closed",
|
| 1347 |
+
"status": "complete",
|
| 1348 |
+
"views": [
|
| 1349 |
+
{
|
| 1350 |
+
"fov_y": 45.0,
|
| 1351 |
+
"height": 720,
|
| 1352 |
+
"name": "third_person",
|
| 1353 |
+
"pose": null,
|
| 1354 |
+
"source": "third_person",
|
| 1355 |
+
"width": 960
|
| 1356 |
+
},
|
| 1357 |
+
{
|
| 1358 |
+
"fov_y": 45.0,
|
| 1359 |
+
"height": 720,
|
| 1360 |
+
"name": "left_wrist",
|
| 1361 |
+
"pose": null,
|
| 1362 |
+
"source": "left_wrist",
|
| 1363 |
+
"width": 960
|
| 1364 |
+
},
|
| 1365 |
+
{
|
| 1366 |
+
"fov_y": 45.0,
|
| 1367 |
+
"height": 720,
|
| 1368 |
+
"name": "right_wrist",
|
| 1369 |
+
"pose": null,
|
| 1370 |
+
"source": "right_wrist",
|
| 1371 |
+
"width": 960
|
| 1372 |
+
}
|
| 1373 |
+
]
|
| 1374 |
+
},
|
| 1375 |
+
"view_names": [
|
| 1376 |
+
"third_person",
|
| 1377 |
+
"left_wrist",
|
| 1378 |
+
"right_wrist"
|
| 1379 |
+
],
|
| 1380 |
+
"sha256": "0eceba672ce4ecdeea98aa372cce1c4d7dd69cf118c2766b39ad4fee0204a3fb",
|
| 1381 |
+
"scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
|
| 1382 |
+
},
|
| 1383 |
+
"analysis": {
|
| 1384 |
+
"outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
|
| 1385 |
+
"duration": "\u4f7f\u7528 4,162 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
|
| 1386 |
+
"execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
|
| 1387 |
+
},
|
| 1388 |
+
"provenance": {
|
| 1389 |
+
"sources": {
|
| 1390 |
+
"RLE-Bench-inhouse": {
|
| 1391 |
+
"build_inputs": [
|
| 1392 |
+
"pyproject.toml",
|
| 1393 |
+
"src",
|
| 1394 |
+
"tasks",
|
| 1395 |
+
"README.md",
|
| 1396 |
+
"Makefile",
|
| 1397 |
+
"tests",
|
| 1398 |
+
"docs"
|
| 1399 |
+
],
|
| 1400 |
+
"content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
|
| 1401 |
+
"revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
|
| 1402 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
|
| 1403 |
+
},
|
| 1404 |
+
"RoboEnv": {
|
| 1405 |
+
"build_inputs": [
|
| 1406 |
+
"pyproject.toml",
|
| 1407 |
+
"README.md",
|
| 1408 |
+
"src",
|
| 1409 |
+
"runtime/pyproject.toml",
|
| 1410 |
+
"runtime/README.md",
|
| 1411 |
+
"runtime/src",
|
| 1412 |
+
"runtime/environments.json",
|
| 1413 |
+
"runtime/locks",
|
| 1414 |
+
"catalog",
|
| 1415 |
+
"upstreams.lock.json",
|
| 1416 |
+
"third_party/patches",
|
| 1417 |
+
"docs/validation"
|
| 1418 |
+
],
|
| 1419 |
+
"content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
|
| 1420 |
+
"revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
|
| 1421 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
|
| 1422 |
+
}
|
| 1423 |
+
},
|
| 1424 |
+
"job": "07-robodojo-insert-tubes-codex-seed0-attempt02",
|
| 1425 |
+
"attempt": 2,
|
| 1426 |
+
"harness": "stock Codex CLI",
|
| 1427 |
+
"codex_version": "0.160.0",
|
| 1428 |
+
"model": "gpt-6-astra",
|
| 1429 |
+
"effort": "high",
|
| 1430 |
+
"service_tier": "default",
|
| 1431 |
+
"fresh_session": true,
|
| 1432 |
+
"source_jobs": [],
|
| 1433 |
+
"resume_trajectory": false,
|
| 1434 |
+
"imported_skills": [],
|
| 1435 |
+
"automatic_harbor_retries": 0,
|
| 1436 |
+
"request_policy": {
|
| 1437 |
+
"max_request_retries": 50,
|
| 1438 |
+
"configuration": "explicit retry50 SSE",
|
| 1439 |
+
"usage_accounting": "reported-responses"
|
| 1440 |
+
},
|
| 1441 |
+
"classification": "formal",
|
| 1442 |
+
"measured_images": {
|
| 1443 |
+
"agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
|
| 1444 |
+
"task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
|
| 1445 |
+
},
|
| 1446 |
+
"session_original_sha256": "dced20541deac3be24265775d5b357aadecb6b53ad8b991fbf38cfcb1a690cfd",
|
| 1447 |
+
"protocol_sha256": "331cfc33f7aa656ce04ffa0d8e7927d625ea36dd825616041760649c239cee4b"
|
| 1448 |
+
},
|
| 1449 |
+
"links": {
|
| 1450 |
+
"native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/session.jsonl",
|
| 1451 |
+
"trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/trajectory.json",
|
| 1452 |
+
"provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/provider-usage.jsonl",
|
| 1453 |
+
"transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/transcript.json",
|
| 1454 |
+
"verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/episode.json",
|
| 1455 |
+
"protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/protocol.json",
|
| 1456 |
+
"native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/instructions.json",
|
| 1457 |
+
"workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/workspace.tar.gz",
|
| 1458 |
+
"workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/workspace.json",
|
| 1459 |
+
"owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
|
| 1460 |
+
"recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/evidence/episode/evidence/recording/manifest.json",
|
| 1461 |
+
"usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/usage.json",
|
| 1462 |
+
"analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/analysis.json",
|
| 1463 |
+
"provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/provenance.json",
|
| 1464 |
+
"video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/video.mp4",
|
| 1465 |
+
"poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/poster.jpg",
|
| 1466 |
+
"media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/media-validation.json"
|
| 1467 |
+
},
|
| 1468 |
+
"resources": [
|
| 1469 |
+
{
|
| 1470 |
+
"name": "tools/robot.py",
|
| 1471 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/tools/robot.py",
|
| 1472 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 1473 |
+
},
|
| 1474 |
+
{
|
| 1475 |
+
"name": "tools/tubes.py",
|
| 1476 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/tools/tubes.py",
|
| 1477 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 1478 |
+
},
|
| 1479 |
+
{
|
| 1480 |
+
"name": "memos/insert-tubes.md",
|
| 1481 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/memos/insert-tubes.md",
|
| 1482 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 1483 |
+
}
|
| 1484 |
+
],
|
| 1485 |
+
"session_counts": {
|
| 1486 |
+
"visible_events": 362,
|
| 1487 |
+
"observed_images": 84,
|
| 1488 |
+
"tool_errors": 4
|
| 1489 |
+
},
|
| 1490 |
+
"selected_for_formal_metrics": true,
|
| 1491 |
+
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/"
|
| 1492 |
+
},
|
| 1493 |
{
|
| 1494 |
"id": "task04-08-seed0-formal",
|
| 1495 |
"task_key": "task04/08",
|
manifest.json
CHANGED
|
@@ -1,21 +1,21 @@
|
|
| 1 |
{
|
| 2 |
".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
|
| 3 |
"DATA_FORMAT.md": "90cde3a7afcec6312f6705d47a959bd2135f6ed599b1ef70944eeca209b96778",
|
| 4 |
-
"README.md": "
|
| 5 |
-
"REPORT.md": "
|
| 6 |
"THIRD_PARTY_NOTICES.md": "b90a0d2dc93b1686caed55e3c68f85a1c761cb95a4935c692a223ab7e19c4c28",
|
| 7 |
"app.js": "3ef085aa517e167f8020eaef3ef26bbf4ca71f759e332c2358ddf3411c25c771",
|
| 8 |
"attempt-history.json": "8555f458c1dd362f5a07447342994fddf265ff23bbc40e9bd6f9984a892ddb4e",
|
| 9 |
-
"comparison.json": "
|
| 10 |
-
"data.json": "
|
| 11 |
-
"episodes.csv": "
|
| 12 |
-
"episodes.json": "
|
| 13 |
"failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
|
| 14 |
"favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
|
| 15 |
"index.html": "da4a91fd226b8792395edbd3d2badda0e1fd69913d9c27480c12bcbdbbc73c37",
|
| 16 |
"licenses/RoboDojo-LICENSE.txt": "7794bb06af8fe5485ca912454ad1665ccd7846c45f9c2b1258658e480948cdbe",
|
| 17 |
-
"protocol.json": "
|
| 18 |
"style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
|
| 19 |
-
"summary.json": "
|
| 20 |
-
"task-index.json": "
|
| 21 |
}
|
|
|
|
| 1 |
{
|
| 2 |
".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
|
| 3 |
"DATA_FORMAT.md": "90cde3a7afcec6312f6705d47a959bd2135f6ed599b1ef70944eeca209b96778",
|
| 4 |
+
"README.md": "e962b0773cc1cddd293ecd3182f6f36161e2396c7652751f19efe08fa992c1f9",
|
| 5 |
+
"REPORT.md": "1d1faa33c849daac0560cd4ede19c59d5fe1f7cd04e7a404f43714a5215006fe",
|
| 6 |
"THIRD_PARTY_NOTICES.md": "b90a0d2dc93b1686caed55e3c68f85a1c761cb95a4935c692a223ab7e19c4c28",
|
| 7 |
"app.js": "3ef085aa517e167f8020eaef3ef26bbf4ca71f759e332c2358ddf3411c25c771",
|
| 8 |
"attempt-history.json": "8555f458c1dd362f5a07447342994fddf265ff23bbc40e9bd6f9984a892ddb4e",
|
| 9 |
+
"comparison.json": "1d92156de449708d2f190642a7247c94b80dab37915174dba8f908a832621c98",
|
| 10 |
+
"data.json": "a8433e92fc2ca0ea0ab8debf5f697a0ee53e3bacb1968573e4b6b9a03b5f3731",
|
| 11 |
+
"episodes.csv": "887754b432854d3e2c05538688ef2e16944fdf08facb001e2591052ae5e35d4f",
|
| 12 |
+
"episodes.json": "f09de90724e941a862a8f716c57d049123a5f0a96252732bcb697feb603ad497",
|
| 13 |
"failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
|
| 14 |
"favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
|
| 15 |
"index.html": "da4a91fd226b8792395edbd3d2badda0e1fd69913d9c27480c12bcbdbbc73c37",
|
| 16 |
"licenses/RoboDojo-LICENSE.txt": "7794bb06af8fe5485ca912454ad1665ccd7846c45f9c2b1258658e480948cdbe",
|
| 17 |
+
"protocol.json": "764850bb861719efbc63bf62ca4b2f2ce8e34797d6fb2d16aec94c1d3dc74894",
|
| 18 |
"style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
|
| 19 |
+
"summary.json": "c784cb052580571e67a596785372ac0f27cb924b19c41eb9d2e235b06f335f13",
|
| 20 |
+
"task-index.json": "db238d1e28b1bf49e140a4bbd190962b20960bc3eae90ca0ac538f9ef3038e96"
|
| 21 |
}
|
protocol.json
CHANGED
|
@@ -16,20 +16,20 @@
|
|
| 16 |
"id": "task04",
|
| 17 |
"name": "RoboDojo",
|
| 18 |
"total": 42,
|
| 19 |
-
"completed":
|
| 20 |
-
"pending":
|
| 21 |
-
"successes":
|
| 22 |
-
"valid_results":
|
| 23 |
-
"success_rate": 0.
|
| 24 |
-
"input_tokens":
|
| 25 |
-
"cached_input_tokens":
|
| 26 |
-
"output_tokens":
|
| 27 |
-
"usage_complete":
|
| 28 |
"control_frequency_hz": 25,
|
| 29 |
"max_control_steps": 7500,
|
| 30 |
"preflight_results": 0,
|
| 31 |
-
"formal_results":
|
| 32 |
-
"modified_results":
|
| 33 |
"original_results": 8
|
| 34 |
}
|
| 35 |
],
|
|
|
|
| 16 |
"id": "task04",
|
| 17 |
"name": "RoboDojo",
|
| 18 |
"total": 42,
|
| 19 |
+
"completed": 39,
|
| 20 |
+
"pending": 3,
|
| 21 |
+
"successes": 31,
|
| 22 |
+
"valid_results": 39,
|
| 23 |
+
"success_rate": 0.7948717948717948,
|
| 24 |
+
"input_tokens": 229797708,
|
| 25 |
+
"cached_input_tokens": 225752320,
|
| 26 |
+
"output_tokens": 869279,
|
| 27 |
+
"usage_complete": 39,
|
| 28 |
"control_frequency_hz": 25,
|
| 29 |
"max_control_steps": 7500,
|
| 30 |
"preflight_results": 0,
|
| 31 |
+
"formal_results": 39,
|
| 32 |
+
"modified_results": 31,
|
| 33 |
"original_results": 8
|
| 34 |
}
|
| 35 |
],
|
publication-manifest.json
CHANGED
|
@@ -10,11 +10,11 @@
|
|
| 10 |
"bytes": 1129
|
| 11 |
},
|
| 12 |
"README.md": {
|
| 13 |
-
"sha256": "
|
| 14 |
"bytes": 1179
|
| 15 |
},
|
| 16 |
"REPORT.md": {
|
| 17 |
-
"sha256": "
|
| 18 |
"bytes": 1056
|
| 19 |
},
|
| 20 |
"THIRD_PARTY_NOTICES.md": {
|
|
@@ -30,20 +30,20 @@
|
|
| 30 |
"bytes": 36661
|
| 31 |
},
|
| 32 |
"comparison.json": {
|
| 33 |
-
"sha256": "
|
| 34 |
-
"bytes":
|
| 35 |
},
|
| 36 |
"data.json": {
|
| 37 |
-
"sha256": "
|
| 38 |
-
"bytes":
|
| 39 |
},
|
| 40 |
"episodes.csv": {
|
| 41 |
-
"sha256": "
|
| 42 |
-
"bytes":
|
| 43 |
},
|
| 44 |
"episodes.json": {
|
| 45 |
-
"sha256": "
|
| 46 |
-
"bytes":
|
| 47 |
},
|
| 48 |
"failure-reviews.json": {
|
| 49 |
"sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
|
|
@@ -62,7 +62,7 @@
|
|
| 62 |
"bytes": 1091
|
| 63 |
},
|
| 64 |
"protocol.json": {
|
| 65 |
-
"sha256": "
|
| 66 |
"bytes": 1083
|
| 67 |
},
|
| 68 |
"style.css": {
|
|
@@ -70,15 +70,15 @@
|
|
| 70 |
"bytes": 16560
|
| 71 |
},
|
| 72 |
"summary.json": {
|
| 73 |
-
"sha256": "
|
| 74 |
"bytes": 896
|
| 75 |
},
|
| 76 |
"task-index.json": {
|
| 77 |
-
"sha256": "
|
| 78 |
-
"bytes":
|
| 79 |
},
|
| 80 |
"manifest.json": {
|
| 81 |
-
"sha256": "
|
| 82 |
"bytes": 1671
|
| 83 |
}
|
| 84 |
}
|
|
|
|
| 10 |
"bytes": 1129
|
| 11 |
},
|
| 12 |
"README.md": {
|
| 13 |
+
"sha256": "e962b0773cc1cddd293ecd3182f6f36161e2396c7652751f19efe08fa992c1f9",
|
| 14 |
"bytes": 1179
|
| 15 |
},
|
| 16 |
"REPORT.md": {
|
| 17 |
+
"sha256": "1d1faa33c849daac0560cd4ede19c59d5fe1f7cd04e7a404f43714a5215006fe",
|
| 18 |
"bytes": 1056
|
| 19 |
},
|
| 20 |
"THIRD_PARTY_NOTICES.md": {
|
|
|
|
| 30 |
"bytes": 36661
|
| 31 |
},
|
| 32 |
"comparison.json": {
|
| 33 |
+
"sha256": "1d92156de449708d2f190642a7247c94b80dab37915174dba8f908a832621c98",
|
| 34 |
+
"bytes": 9461
|
| 35 |
},
|
| 36 |
"data.json": {
|
| 37 |
+
"sha256": "a8433e92fc2ca0ea0ab8debf5f697a0ee53e3bacb1968573e4b6b9a03b5f3731",
|
| 38 |
+
"bytes": 577958
|
| 39 |
},
|
| 40 |
"episodes.csv": {
|
| 41 |
+
"sha256": "887754b432854d3e2c05538688ef2e16944fdf08facb001e2591052ae5e35d4f",
|
| 42 |
+
"bytes": 3707
|
| 43 |
},
|
| 44 |
"episodes.json": {
|
| 45 |
+
"sha256": "f09de90724e941a862a8f716c57d049123a5f0a96252732bcb697feb603ad497",
|
| 46 |
+
"bytes": 408492
|
| 47 |
},
|
| 48 |
"failure-reviews.json": {
|
| 49 |
"sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
|
|
|
|
| 62 |
"bytes": 1091
|
| 63 |
},
|
| 64 |
"protocol.json": {
|
| 65 |
+
"sha256": "764850bb861719efbc63bf62ca4b2f2ce8e34797d6fb2d16aec94c1d3dc74894",
|
| 66 |
"bytes": 1083
|
| 67 |
},
|
| 68 |
"style.css": {
|
|
|
|
| 70 |
"bytes": 16560
|
| 71 |
},
|
| 72 |
"summary.json": {
|
| 73 |
+
"sha256": "c784cb052580571e67a596785372ac0f27cb924b19c41eb9d2e235b06f335f13",
|
| 74 |
"bytes": 896
|
| 75 |
},
|
| 76 |
"task-index.json": {
|
| 77 |
+
"sha256": "db238d1e28b1bf49e140a4bbd190962b20960bc3eae90ca0ac538f9ef3038e96",
|
| 78 |
+
"bytes": 140033
|
| 79 |
},
|
| 80 |
"manifest.json": {
|
| 81 |
+
"sha256": "b1f2ebc9cc75fc4375931ff9df5642fd800d0ec972f214e37af5284bcb4a8095",
|
| 82 |
"bytes": 1671
|
| 83 |
}
|
| 84 |
}
|
summary.json
CHANGED
|
@@ -1,37 +1,37 @@
|
|
| 1 |
{
|
| 2 |
"planned_tasks": 42,
|
| 3 |
-
"published_results":
|
| 4 |
-
"pending_tasks":
|
| 5 |
"families": [
|
| 6 |
{
|
| 7 |
"id": "task04",
|
| 8 |
"name": "RoboDojo",
|
| 9 |
"total": 42,
|
| 10 |
-
"completed":
|
| 11 |
-
"pending":
|
| 12 |
-
"successes":
|
| 13 |
-
"valid_results":
|
| 14 |
-
"success_rate": 0.
|
| 15 |
-
"input_tokens":
|
| 16 |
-
"cached_input_tokens":
|
| 17 |
-
"output_tokens":
|
| 18 |
-
"usage_complete":
|
| 19 |
"control_frequency_hz": 25,
|
| 20 |
"max_control_steps": 7500,
|
| 21 |
"preflight_results": 0,
|
| 22 |
-
"formal_results":
|
| 23 |
-
"modified_results":
|
| 24 |
"original_results": 8
|
| 25 |
}
|
| 26 |
],
|
| 27 |
"progress": {
|
| 28 |
-
"finished":
|
| 29 |
"interrupted": 2,
|
| 30 |
"native_failures": 8,
|
| 31 |
-
"native_successes":
|
| 32 |
"needs_review": 0,
|
| 33 |
"queued": 0,
|
| 34 |
-
"running":
|
| 35 |
},
|
| 36 |
"interrupted_attempts": 4,
|
| 37 |
"usage_incomplete_tasks": [],
|
|
|
|
| 1 |
{
|
| 2 |
"planned_tasks": 42,
|
| 3 |
+
"published_results": 39,
|
| 4 |
+
"pending_tasks": 3,
|
| 5 |
"families": [
|
| 6 |
{
|
| 7 |
"id": "task04",
|
| 8 |
"name": "RoboDojo",
|
| 9 |
"total": 42,
|
| 10 |
+
"completed": 39,
|
| 11 |
+
"pending": 3,
|
| 12 |
+
"successes": 31,
|
| 13 |
+
"valid_results": 39,
|
| 14 |
+
"success_rate": 0.7948717948717948,
|
| 15 |
+
"input_tokens": 229797708,
|
| 16 |
+
"cached_input_tokens": 225752320,
|
| 17 |
+
"output_tokens": 869279,
|
| 18 |
+
"usage_complete": 39,
|
| 19 |
"control_frequency_hz": 25,
|
| 20 |
"max_control_steps": 7500,
|
| 21 |
"preflight_results": 0,
|
| 22 |
+
"formal_results": 39,
|
| 23 |
+
"modified_results": 31,
|
| 24 |
"original_results": 8
|
| 25 |
}
|
| 26 |
],
|
| 27 |
"progress": {
|
| 28 |
+
"finished": 39,
|
| 29 |
"interrupted": 2,
|
| 30 |
"native_failures": 8,
|
| 31 |
+
"native_successes": 31,
|
| 32 |
"needs_review": 0,
|
| 33 |
"queued": 0,
|
| 34 |
+
"running": 1
|
| 35 |
},
|
| 36 |
"interrupted_attempts": 4,
|
| 37 |
"usage_incomplete_tasks": [],
|
task-index.json
CHANGED
|
@@ -592,8 +592,8 @@
|
|
| 592 |
"catalog_instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
|
| 593 |
"native_instruction": "Insert the three tubes into the rack one by one.",
|
| 594 |
"instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
|
| 595 |
-
"status": "
|
| 596 |
-
"episode_id":
|
| 597 |
"planned_protocol": {
|
| 598 |
"episodes": 1,
|
| 599 |
"seed": 0,
|
|
@@ -649,8 +649,8 @@
|
|
| 649 |
},
|
| 650 |
"display_slot": "07",
|
| 651 |
"display_key": "task04/07",
|
| 652 |
-
"run_status": "
|
| 653 |
-
"status_note": "
|
| 654 |
"attempt_history": [
|
| 655 |
{
|
| 656 |
"id": "07-robodojo-insert-tubes-codex-seed0-attempt01",
|
|
|
|
| 592 |
"catalog_instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
|
| 593 |
"native_instruction": "Insert the three tubes into the rack one by one.",
|
| 594 |
"instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
|
| 595 |
+
"status": "completed",
|
| 596 |
+
"episode_id": "task04-07-seed0-formal",
|
| 597 |
"planned_protocol": {
|
| 598 |
"episodes": 1,
|
| 599 |
"seed": 0,
|
|
|
|
| 649 |
},
|
| 650 |
"display_slot": "07",
|
| 651 |
"display_key": "task04/07",
|
| 652 |
+
"run_status": "finished",
|
| 653 |
+
"status_note": "Queued for evaluation.",
|
| 654 |
"attempt_history": [
|
| 655 |
{
|
| 656 |
"id": "07-robodojo-insert-tubes-codex-seed0-attempt01",
|