Spaces:
Running
Running
Publish codex SIMPLE seed-0 evaluation evidence
Browse files- README.md +2 -2
- REPORT.md +2 -2
- data.json +299 -33
- episodes.csv +1 -1
- episodes.json +266 -0
- manifest.json +8 -8
- protocol.json +10 -10
- publication-manifest.json +15 -15
- summary.json +17 -17
- task-index.json +5 -5
README.md
CHANGED
|
@@ -10,7 +10,7 @@ pinned: false
|
|
| 10 |
|
| 11 |
# Codex Benchmark
|
| 12 |
|
| 13 |
-
|
| 14 |
|
| 15 |
| Environment | Published / selected | Native successes / valid results |
|
| 16 |
| --- | ---: | ---: |
|
|
@@ -18,7 +18,7 @@ pinned: false
|
|
| 18 |
| LIBERO Long | 10 / 10 | 9 / 10 |
|
| 19 |
| RoboTwin | 10 / 10 | 7 / 10 |
|
| 20 |
| RoboDojo | 42 / 42 | 31 / 42 |
|
| 21 |
-
| SIMPLE G1 |
|
| 22 |
|
| 23 |
SIMPLE uses GPT-6 Astra with high reasoning effort and default service tier, one fresh seed-0 episode per task, 10000 native control intervals at 50 Hz, stepped MuJoCo physics with Isaac rendering, domain randomization level 0, arm gravity compensation, stop-on-success, recording and an 8-hour wall-clock timeout. Request retries remain within the same session; whole episodes are never retried automatically.
|
| 24 |
|
|
|
|
| 10 |
|
| 11 |
# Codex Benchmark
|
| 12 |
|
| 13 |
+
80 published results across 5 simulator families. This snapshot adds 8 of 12 authorized SIMPLE G1 results.
|
| 14 |
|
| 15 |
| Environment | Published / selected | Native successes / valid results |
|
| 16 |
| --- | ---: | ---: |
|
|
|
|
| 18 |
| LIBERO Long | 10 / 10 | 9 / 10 |
|
| 19 |
| RoboTwin | 10 / 10 | 7 / 10 |
|
| 20 |
| RoboDojo | 42 / 42 | 31 / 42 |
|
| 21 |
+
| SIMPLE G1 | 8 / 12 | 2 / 8 |
|
| 22 |
|
| 23 |
SIMPLE uses GPT-6 Astra with high reasoning effort and default service tier, one fresh seed-0 episode per task, 10000 native control intervals at 50 Hz, stepped MuJoCo physics with Isaac rendering, domain randomization level 0, arm gravity compensation, stop-on-success, recording and an 8-hour wall-clock timeout. Request retries remain within the same session; whole episodes are never retried automatically.
|
| 24 |
|
REPORT.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
# Codex Benchmark
|
| 2 |
|
| 3 |
-
|
| 4 |
|
| 5 |
| Environment | Published / selected | Native successes / valid results |
|
| 6 |
| --- | ---: | ---: |
|
|
@@ -8,7 +8,7 @@
|
|
| 8 |
| LIBERO Long | 10 / 10 | 9 / 10 |
|
| 9 |
| RoboTwin | 10 / 10 | 7 / 10 |
|
| 10 |
| RoboDojo | 42 / 42 | 31 / 42 |
|
| 11 |
-
| SIMPLE G1 |
|
| 12 |
|
| 13 |
SIMPLE uses GPT-6 Astra with high reasoning effort and default service tier, one fresh seed-0 episode per task, 10000 native control intervals at 50 Hz, stepped MuJoCo physics with Isaac rendering, domain randomization level 0, arm gravity compensation, stop-on-success, recording and an 8-hour wall-clock timeout. Request retries remain within the same session; whole episodes are never retried automatically.
|
| 14 |
|
|
|
|
| 1 |
# Codex Benchmark
|
| 2 |
|
| 3 |
+
80 published results across 5 simulator families. This snapshot adds 8 of 12 authorized SIMPLE G1 results.
|
| 4 |
|
| 5 |
| Environment | Published / selected | Native successes / valid results |
|
| 6 |
| --- | ---: | ---: |
|
|
|
|
| 8 |
| LIBERO Long | 10 / 10 | 9 / 10 |
|
| 9 |
| RoboTwin | 10 / 10 | 7 / 10 |
|
| 10 |
| RoboDojo | 42 / 42 | 31 / 42 |
|
| 11 |
+
| SIMPLE G1 | 8 / 12 | 2 / 8 |
|
| 12 |
|
| 13 |
SIMPLE uses GPT-6 Astra with high reasoning effort and default service tier, one fresh seed-0 episode per task, 10000 native control intervals at 50 Hz, stepped MuJoCo physics with Isaac rendering, domain randomization level 0, arm gravity compensation, stop-on-success, recording and an 8-hour wall-clock timeout. Request retries remain within the same session; whole episodes are never retried automatically.
|
| 14 |
|
data.json
CHANGED
|
@@ -6,7 +6,7 @@
|
|
| 6 |
"benchmark_complete": false,
|
| 7 |
"edition": "codex-simple-astra-high-seed0",
|
| 8 |
"created_at": "2026-10-09T00:09:42.989838+00:00",
|
| 9 |
-
"updated_at": "2026-10-10T01:
|
| 10 |
"model": "gpt-6-astra",
|
| 11 |
"effort": "high",
|
| 12 |
"seed": 0,
|
|
@@ -17,8 +17,8 @@
|
|
| 17 |
},
|
| 18 |
"summary": {
|
| 19 |
"planned_tasks": 84,
|
| 20 |
-
"published_results":
|
| 21 |
-
"pending_tasks":
|
| 22 |
"families": [
|
| 23 |
{
|
| 24 |
"id": "task01",
|
|
@@ -105,40 +105,40 @@
|
|
| 105 |
"id": "task06",
|
| 106 |
"name": "SIMPLE G1",
|
| 107 |
"total": 12,
|
| 108 |
-
"completed":
|
| 109 |
-
"pending":
|
| 110 |
"successes": 2,
|
| 111 |
-
"valid_results":
|
| 112 |
-
"success_rate": 0.
|
| 113 |
-
"usage_complete":
|
| 114 |
"control_frequency_hz": 50,
|
| 115 |
"max_control_steps": 10000,
|
| 116 |
"preflight_results": 0,
|
| 117 |
-
"formal_results":
|
| 118 |
"modified_results": 0,
|
| 119 |
-
"original_results":
|
| 120 |
-
"input_tokens":
|
| 121 |
-
"cached_input_tokens":
|
| 122 |
-
"output_tokens":
|
| 123 |
}
|
| 124 |
],
|
| 125 |
"interrupted_attempts": 4,
|
| 126 |
"usage_incomplete_tasks": [],
|
| 127 |
"execution_incomplete_tasks": [],
|
| 128 |
"deferred_tasks": [],
|
| 129 |
-
"new_evaluations":
|
| 130 |
"progress": {
|
| 131 |
-
"finished":
|
| 132 |
"running": 3,
|
| 133 |
-
"queued":
|
| 134 |
"needs_review": 0,
|
| 135 |
"interrupted": 0,
|
| 136 |
"native_successes": 51,
|
| 137 |
-
"native_failures":
|
| 138 |
},
|
| 139 |
"simple_campaign": {
|
| 140 |
"planned": 12,
|
| 141 |
-
"published":
|
| 142 |
"agent": "codex",
|
| 143 |
"model": "gpt-6-astra",
|
| 144 |
"effort": "high",
|
|
@@ -232,21 +232,21 @@
|
|
| 232 |
"id": "task06",
|
| 233 |
"name": "SIMPLE G1",
|
| 234 |
"total": 12,
|
| 235 |
-
"completed":
|
| 236 |
-
"pending":
|
| 237 |
"successes": 2,
|
| 238 |
-
"valid_results":
|
| 239 |
-
"success_rate": 0.
|
| 240 |
-
"usage_complete":
|
| 241 |
"control_frequency_hz": 50,
|
| 242 |
"max_control_steps": 10000,
|
| 243 |
"preflight_results": 0,
|
| 244 |
-
"formal_results":
|
| 245 |
"modified_results": 0,
|
| 246 |
-
"original_results":
|
| 247 |
-
"input_tokens":
|
| 248 |
-
"cached_input_tokens":
|
| 249 |
-
"output_tokens":
|
| 250 |
}
|
| 251 |
],
|
| 252 |
"tasks": [
|
|
@@ -6483,10 +6483,10 @@
|
|
| 6483 |
"catalog_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 6484 |
"native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 6485 |
"instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
|
| 6486 |
-
"status": "
|
| 6487 |
-
"episode_id":
|
| 6488 |
-
"run_status": "
|
| 6489 |
-
"status_note": "
|
| 6490 |
"planned_protocol": {
|
| 6491 |
"episodes": 1,
|
| 6492 |
"seed": 0,
|
|
@@ -6605,7 +6605,7 @@
|
|
| 6605 |
"instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
|
| 6606 |
"status": "pending",
|
| 6607 |
"episode_id": null,
|
| 6608 |
-
"run_status": "
|
| 6609 |
"status_note": "Authorized episode is queued, running, or awaiting evidence review.",
|
| 6610 |
"planned_protocol": {
|
| 6611 |
"episodes": 1,
|
|
@@ -26741,6 +26741,272 @@
|
|
| 26741 |
"selected_for_formal_metrics": true,
|
| 26742 |
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/06/seed-0/"
|
| 26743 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 26744 |
{
|
| 26745 |
"id": "task06-08-seed0-formal",
|
| 26746 |
"task_key": "task06/08",
|
|
|
|
| 6 |
"benchmark_complete": false,
|
| 7 |
"edition": "codex-simple-astra-high-seed0",
|
| 8 |
"created_at": "2026-10-09T00:09:42.989838+00:00",
|
| 9 |
+
"updated_at": "2026-10-10T01:24:39.237016+00:00",
|
| 10 |
"model": "gpt-6-astra",
|
| 11 |
"effort": "high",
|
| 12 |
"seed": 0,
|
|
|
|
| 17 |
},
|
| 18 |
"summary": {
|
| 19 |
"planned_tasks": 84,
|
| 20 |
+
"published_results": 80,
|
| 21 |
+
"pending_tasks": 4,
|
| 22 |
"families": [
|
| 23 |
{
|
| 24 |
"id": "task01",
|
|
|
|
| 105 |
"id": "task06",
|
| 106 |
"name": "SIMPLE G1",
|
| 107 |
"total": 12,
|
| 108 |
+
"completed": 8,
|
| 109 |
+
"pending": 4,
|
| 110 |
"successes": 2,
|
| 111 |
+
"valid_results": 8,
|
| 112 |
+
"success_rate": 0.25,
|
| 113 |
+
"usage_complete": 8,
|
| 114 |
"control_frequency_hz": 50,
|
| 115 |
"max_control_steps": 10000,
|
| 116 |
"preflight_results": 0,
|
| 117 |
+
"formal_results": 8,
|
| 118 |
"modified_results": 0,
|
| 119 |
+
"original_results": 8,
|
| 120 |
+
"input_tokens": 108175382,
|
| 121 |
+
"cached_input_tokens": 106998016,
|
| 122 |
+
"output_tokens": 283858
|
| 123 |
}
|
| 124 |
],
|
| 125 |
"interrupted_attempts": 4,
|
| 126 |
"usage_incomplete_tasks": [],
|
| 127 |
"execution_incomplete_tasks": [],
|
| 128 |
"deferred_tasks": [],
|
| 129 |
+
"new_evaluations": 8,
|
| 130 |
"progress": {
|
| 131 |
+
"finished": 80,
|
| 132 |
"running": 3,
|
| 133 |
+
"queued": 1,
|
| 134 |
"needs_review": 0,
|
| 135 |
"interrupted": 0,
|
| 136 |
"native_successes": 51,
|
| 137 |
+
"native_failures": 29
|
| 138 |
},
|
| 139 |
"simple_campaign": {
|
| 140 |
"planned": 12,
|
| 141 |
+
"published": 8,
|
| 142 |
"agent": "codex",
|
| 143 |
"model": "gpt-6-astra",
|
| 144 |
"effort": "high",
|
|
|
|
| 232 |
"id": "task06",
|
| 233 |
"name": "SIMPLE G1",
|
| 234 |
"total": 12,
|
| 235 |
+
"completed": 8,
|
| 236 |
+
"pending": 4,
|
| 237 |
"successes": 2,
|
| 238 |
+
"valid_results": 8,
|
| 239 |
+
"success_rate": 0.25,
|
| 240 |
+
"usage_complete": 8,
|
| 241 |
"control_frequency_hz": 50,
|
| 242 |
"max_control_steps": 10000,
|
| 243 |
"preflight_results": 0,
|
| 244 |
+
"formal_results": 8,
|
| 245 |
"modified_results": 0,
|
| 246 |
+
"original_results": 8,
|
| 247 |
+
"input_tokens": 108175382,
|
| 248 |
+
"cached_input_tokens": 106998016,
|
| 249 |
+
"output_tokens": 283858
|
| 250 |
}
|
| 251 |
],
|
| 252 |
"tasks": [
|
|
|
|
| 6483 |
"catalog_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 6484 |
"native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 6485 |
"instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
|
| 6486 |
+
"status": "completed",
|
| 6487 |
+
"episode_id": "task06-07-seed0-formal",
|
| 6488 |
+
"run_status": "finished",
|
| 6489 |
+
"status_note": "Native result, execution and usage complete.",
|
| 6490 |
"planned_protocol": {
|
| 6491 |
"episodes": 1,
|
| 6492 |
"seed": 0,
|
|
|
|
| 6605 |
"instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
|
| 6606 |
"status": "pending",
|
| 6607 |
"episode_id": null,
|
| 6608 |
+
"run_status": "running",
|
| 6609 |
"status_note": "Authorized episode is queued, running, or awaiting evidence review.",
|
| 6610 |
"planned_protocol": {
|
| 6611 |
"episodes": 1,
|
|
|
|
| 26741 |
"selected_for_formal_metrics": true,
|
| 26742 |
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/06/seed-0/"
|
| 26743 |
},
|
| 26744 |
+
{
|
| 26745 |
+
"id": "task06-07-seed0-formal",
|
| 26746 |
+
"task_key": "task06/07",
|
| 26747 |
+
"family": "task06",
|
| 26748 |
+
"slot": "07",
|
| 26749 |
+
"seed": 0,
|
| 26750 |
+
"episode": 1,
|
| 26751 |
+
"phase": "formal",
|
| 26752 |
+
"status": "completed",
|
| 26753 |
+
"success": false,
|
| 26754 |
+
"native_reward": 0.0,
|
| 26755 |
+
"valid": true,
|
| 26756 |
+
"execution": {
|
| 26757 |
+
"reason": null,
|
| 26758 |
+
"status": "finished"
|
| 26759 |
+
},
|
| 26760 |
+
"verdict": {
|
| 26761 |
+
"evidence_valid": true,
|
| 26762 |
+
"steps": 8645,
|
| 26763 |
+
"success": false,
|
| 26764 |
+
"termination": "stopped"
|
| 26765 |
+
},
|
| 26766 |
+
"steps": 8645,
|
| 26767 |
+
"simulation_time_s": null,
|
| 26768 |
+
"wall_time_s": 2551.587007,
|
| 26769 |
+
"model": "gpt-6-astra",
|
| 26770 |
+
"effort": "high",
|
| 26771 |
+
"harness": "codex",
|
| 26772 |
+
"codex_version": "0.160.0",
|
| 26773 |
+
"native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 26774 |
+
"instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 26775 |
+
"instruction_policy": "original_native",
|
| 26776 |
+
"usage": {
|
| 26777 |
+
"accounting": "reported-responses",
|
| 26778 |
+
"audit_complete": true,
|
| 26779 |
+
"cache_hit_rate": 0.9872509714529868,
|
| 26780 |
+
"cache_reported_input_tokens": 16738844,
|
| 26781 |
+
"cache_write_input_tokens": 0,
|
| 26782 |
+
"cache_write_reported_input_tokens": 16738844,
|
| 26783 |
+
"cached_input_tokens": 16525440,
|
| 26784 |
+
"completed_turns": 1,
|
| 26785 |
+
"cost_usd": null,
|
| 26786 |
+
"failed_turns": 0,
|
| 26787 |
+
"input_tokens": 16738844,
|
| 26788 |
+
"known_cache_write_input_tokens": 0,
|
| 26789 |
+
"known_cached_input_tokens": 16525440,
|
| 26790 |
+
"known_input_tokens": 16738844,
|
| 26791 |
+
"known_output_tokens": 38724,
|
| 26792 |
+
"known_reasoning_output_tokens": 20279,
|
| 26793 |
+
"output_tokens": 38724,
|
| 26794 |
+
"reasoning_output_tokens": 20279,
|
| 26795 |
+
"reasoning_reported_output_tokens": 38724,
|
| 26796 |
+
"reported_responses": {
|
| 26797 |
+
"cache_reported_input_tokens": 192,
|
| 26798 |
+
"cache_write_input_tokens": 192,
|
| 26799 |
+
"cache_write_reported_input_tokens": 192,
|
| 26800 |
+
"cached_input_tokens": 192,
|
| 26801 |
+
"input_tokens": 192,
|
| 26802 |
+
"output_tokens": 192,
|
| 26803 |
+
"reasoning_output_tokens": 192,
|
| 26804 |
+
"reasoning_reported_output_tokens": 192
|
| 26805 |
+
},
|
| 26806 |
+
"response_count": 192,
|
| 26807 |
+
"response_ids_complete": true,
|
| 26808 |
+
"schema": "rlebench/token-usage/1",
|
| 26809 |
+
"source": "Codex token_usage_record per response",
|
| 26810 |
+
"uncached_input_tokens": 213404,
|
| 26811 |
+
"unidentified_usage_records": 0
|
| 26812 |
+
},
|
| 26813 |
+
"call_activity": {
|
| 26814 |
+
"model_tool_calls": 191,
|
| 26815 |
+
"model_tool_calls_by_name": {
|
| 26816 |
+
"exec": 191
|
| 26817 |
+
},
|
| 26818 |
+
"nested_python_tool_invocations": null,
|
| 26819 |
+
"python_device_rpc_attempts": null,
|
| 26820 |
+
"python_device_rpc_attempts_by_action": null,
|
| 26821 |
+
"python_device_rpc_errors": null,
|
| 26822 |
+
"python_instrumented_model_tool_calls": null,
|
| 26823 |
+
"python_tool_invocations": null,
|
| 26824 |
+
"python_tool_invocations_by_origin": null,
|
| 26825 |
+
"schema": "rlebench/call-activity/1",
|
| 26826 |
+
"source": "Codex native sessions"
|
| 26827 |
+
},
|
| 26828 |
+
"media": {
|
| 26829 |
+
"passed": true,
|
| 26830 |
+
"width": 1280,
|
| 26831 |
+
"height": 360,
|
| 26832 |
+
"duration_s": 43.25,
|
| 26833 |
+
"speed": 4,
|
| 26834 |
+
"source_fps": 10,
|
| 26835 |
+
"output_fps": 20,
|
| 26836 |
+
"recording": {
|
| 26837 |
+
"accepted_samples": 1730,
|
| 26838 |
+
"captured_samples": 1730,
|
| 26839 |
+
"clock": "simulation",
|
| 26840 |
+
"dropped_samples": 0,
|
| 26841 |
+
"encoded_frames": 1730,
|
| 26842 |
+
"end_time_s": 172.90000000001464,
|
| 26843 |
+
"error": null,
|
| 26844 |
+
"experimental": true,
|
| 26845 |
+
"fps": 10,
|
| 26846 |
+
"received_samples": 1730,
|
| 26847 |
+
"schema": "roboenv/recording/1",
|
| 26848 |
+
"state": "closed",
|
| 26849 |
+
"status": "complete",
|
| 26850 |
+
"views": [
|
| 26851 |
+
{
|
| 26852 |
+
"fov_y": 45.0,
|
| 26853 |
+
"height": 360,
|
| 26854 |
+
"name": "camera_head_left",
|
| 26855 |
+
"pose": null,
|
| 26856 |
+
"source": "camera_head_left",
|
| 26857 |
+
"width": 640
|
| 26858 |
+
},
|
| 26859 |
+
{
|
| 26860 |
+
"fov_y": 45.0,
|
| 26861 |
+
"height": 360,
|
| 26862 |
+
"name": "camera_head_right",
|
| 26863 |
+
"pose": null,
|
| 26864 |
+
"source": "camera_head_right",
|
| 26865 |
+
"width": 640
|
| 26866 |
+
}
|
| 26867 |
+
]
|
| 26868 |
+
},
|
| 26869 |
+
"view_names": [
|
| 26870 |
+
"camera_head_left",
|
| 26871 |
+
"camera_head_right"
|
| 26872 |
+
],
|
| 26873 |
+
"sha256": "95adbcd6adbca86aa7e6a6cc90c6fe2930fa1d430ac07da60e5756b01ad3dd3b",
|
| 26874 |
+
"scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
|
| 26875 |
+
},
|
| 26876 |
+
"analysis": {
|
| 26877 |
+
"outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
|
| 26878 |
+
"duration": "\u4f7f\u7528 8,645 / 10,000 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
|
| 26879 |
+
"execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
|
| 26880 |
+
"cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
|
| 26881 |
+
},
|
| 26882 |
+
"provenance": {
|
| 26883 |
+
"sources": {
|
| 26884 |
+
"RLE-Bench-inhouse": {
|
| 26885 |
+
"build_inputs": [
|
| 26886 |
+
"pyproject.toml",
|
| 26887 |
+
"src",
|
| 26888 |
+
"tasks",
|
| 26889 |
+
"README.md",
|
| 26890 |
+
"Makefile",
|
| 26891 |
+
"tests",
|
| 26892 |
+
"docs"
|
| 26893 |
+
],
|
| 26894 |
+
"content_sha256": "5b1428b96104cb17b52215bce0ec33b880c5c591d01d4ecc2867cdc647940bda",
|
| 26895 |
+
"dirty": true,
|
| 26896 |
+
"revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
|
| 26897 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
|
| 26898 |
+
},
|
| 26899 |
+
"RoboEnv": {
|
| 26900 |
+
"build_inputs": [
|
| 26901 |
+
"pyproject.toml",
|
| 26902 |
+
"README.md",
|
| 26903 |
+
"src",
|
| 26904 |
+
"runtime/pyproject.toml",
|
| 26905 |
+
"runtime/README.md",
|
| 26906 |
+
"runtime/src",
|
| 26907 |
+
"runtime/environments.json",
|
| 26908 |
+
"runtime/locks",
|
| 26909 |
+
"catalog",
|
| 26910 |
+
"upstreams.lock.json",
|
| 26911 |
+
"third_party/patches",
|
| 26912 |
+
"docs/validation"
|
| 26913 |
+
],
|
| 26914 |
+
"content_sha256": "5e90464ff5b67edba2a8e653e68cba5f7065c19f92f34a29689703d61e04d305",
|
| 26915 |
+
"dirty": true,
|
| 26916 |
+
"revision": "b73356c09507eeffaf9ca9e90b668ef0b7f1fa92",
|
| 26917 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
|
| 26918 |
+
},
|
| 26919 |
+
"kinex": {
|
| 26920 |
+
"build_inputs": [
|
| 26921 |
+
"package.json",
|
| 26922 |
+
"package-lock.json",
|
| 26923 |
+
".nvmrc",
|
| 26924 |
+
"tsconfig.json",
|
| 26925 |
+
"VERSION",
|
| 26926 |
+
"src",
|
| 26927 |
+
"packages/core",
|
| 26928 |
+
"packages/setup",
|
| 26929 |
+
"script",
|
| 26930 |
+
"assets",
|
| 26931 |
+
"bin"
|
| 26932 |
+
],
|
| 26933 |
+
"content_sha256": "70f5107905efa45cace74e04b630660aaee4c80e1db2e3d540d6c81e0f4d0d38",
|
| 26934 |
+
"dirty": false,
|
| 26935 |
+
"revision": "caac19a8a36272972f762e0f74cfe381e8500048",
|
| 26936 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/kinex"
|
| 26937 |
+
}
|
| 26938 |
+
},
|
| 26939 |
+
"job": "task06-07-pick-and-place-and-hug-container-codex-seed0-attempt01",
|
| 26940 |
+
"attempt": 1,
|
| 26941 |
+
"harness": "stock Codex CLI",
|
| 26942 |
+
"control_interface": "public RoboEnv SDK and CLI",
|
| 26943 |
+
"kinex_agent_runtime_used": false,
|
| 26944 |
+
"gpu_index": 3,
|
| 26945 |
+
"gpu_model": "NVIDIA L40S",
|
| 26946 |
+
"campaign_dispatch_concurrency_limit": 5,
|
| 26947 |
+
"concurrency_note": "One GPU per active SIMPLE episode; Kinex and Codex share a five-GPU dispatch pool.",
|
| 26948 |
+
"codex_version": "0.160.0",
|
| 26949 |
+
"model": "gpt-6-astra",
|
| 26950 |
+
"effort": "high",
|
| 26951 |
+
"service_tier": "default",
|
| 26952 |
+
"fresh_session": true,
|
| 26953 |
+
"source_jobs": [],
|
| 26954 |
+
"resume_trajectory": false,
|
| 26955 |
+
"imported_skills": [],
|
| 26956 |
+
"automatic_harbor_retries": 0,
|
| 26957 |
+
"request_policy": {
|
| 26958 |
+
"max_request_retries": 50,
|
| 26959 |
+
"configuration": "explicit retry50 SSE",
|
| 26960 |
+
"usage_accounting": "reported-responses"
|
| 26961 |
+
},
|
| 26962 |
+
"classification": "formal",
|
| 26963 |
+
"measured_images": {
|
| 26964 |
+
"agent": "sha256:3fb135c4db8092900969023cfe52e4bddfc8960b6296318fda47c72e80e5f667",
|
| 26965 |
+
"task": "sha256:730c5fc8ca66ce7bb07635943de71bd443d99c4f6ce962e01d35e90a3aec5c28"
|
| 26966 |
+
},
|
| 26967 |
+
"session_original_sha256": "f57a16755d9b599e6a03f0620b4092819f73f48643c586385c5d54ffa8ff4115",
|
| 26968 |
+
"protocol_sha256": "6e24ceebccc59c6e91caf708994c458548bbb4c3539b206b9d85eebcb7ea6266"
|
| 26969 |
+
},
|
| 26970 |
+
"links": {
|
| 26971 |
+
"native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/session.jsonl",
|
| 26972 |
+
"trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/trajectory.json",
|
| 26973 |
+
"provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/provider-usage.jsonl",
|
| 26974 |
+
"transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/transcript.json",
|
| 26975 |
+
"verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/episode.json",
|
| 26976 |
+
"protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/protocol.json",
|
| 26977 |
+
"native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/instructions.json",
|
| 26978 |
+
"workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/workspace.tar.gz",
|
| 26979 |
+
"workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/workspace.json",
|
| 26980 |
+
"owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
|
| 26981 |
+
"recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/recording/manifest.json",
|
| 26982 |
+
"usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/usage.json",
|
| 26983 |
+
"analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/analysis.json",
|
| 26984 |
+
"provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/provenance.json",
|
| 26985 |
+
"video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/video.mp4",
|
| 26986 |
+
"poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/poster.jpg",
|
| 26987 |
+
"media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/media-validation.json",
|
| 26988 |
+
"final_observation": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/final-observation.json"
|
| 26989 |
+
},
|
| 26990 |
+
"resources": [
|
| 26991 |
+
{
|
| 26992 |
+
"name": "tools/robot.py",
|
| 26993 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/resources/tools/robot.py",
|
| 26994 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 26995 |
+
},
|
| 26996 |
+
{
|
| 26997 |
+
"name": "memos/g1_simple_manipulation.md",
|
| 26998 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/resources/memos/g1_simple_manipulation.md",
|
| 26999 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 27000 |
+
}
|
| 27001 |
+
],
|
| 27002 |
+
"session_counts": {
|
| 27003 |
+
"visible_events": 414,
|
| 27004 |
+
"observed_images": 108,
|
| 27005 |
+
"tool_errors": 1
|
| 27006 |
+
},
|
| 27007 |
+
"selected_for_formal_metrics": true,
|
| 27008 |
+
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/"
|
| 27009 |
+
},
|
| 27010 |
{
|
| 27011 |
"id": "task06-08-seed0-formal",
|
| 27012 |
"task_key": "task06/08",
|
episodes.csv
CHANGED
|
@@ -77,7 +77,7 @@ task06/03,simple/bend-pick,completed,0,True,False,9768,3956.495044,24248209,2404
|
|
| 77 |
task06/04,simple/bend-pick-and-place,completed,0,True,False,8560,3261.927668,31457829,31121920,57128
|
| 78 |
task06/05,simple/bend-handover,completed,0,True,False,9150,3541.240323,24778430,24583808,75261
|
| 79 |
task06/06,simple/handover,completed,0,True,False,4040,1176.656816,5562825,5475584,17747
|
| 80 |
-
task06/07,simple/pick-and-place-and-hug-container,
|
| 81 |
task06/08,simple/close-door,completed,0,True,True,1277,369.21498,1340798,1294208,6729
|
| 82 |
task06/09,simple/open-oven,pending,0,,,,,,,
|
| 83 |
task06/10,simple/open-faucet,pending,0,,,,,,,
|
|
|
|
| 77 |
task06/04,simple/bend-pick-and-place,completed,0,True,False,8560,3261.927668,31457829,31121920,57128
|
| 78 |
task06/05,simple/bend-handover,completed,0,True,False,9150,3541.240323,24778430,24583808,75261
|
| 79 |
task06/06,simple/handover,completed,0,True,False,4040,1176.656816,5562825,5475584,17747
|
| 80 |
+
task06/07,simple/pick-and-place-and-hug-container,completed,0,True,False,8645,2551.587007,16738844,16525440,38724
|
| 81 |
task06/08,simple/close-door,completed,0,True,True,1277,369.21498,1340798,1294208,6729
|
| 82 |
task06/09,simple/open-oven,pending,0,,,,,,,
|
| 83 |
task06/10,simple/open-faucet,pending,0,,,,,,,
|
episodes.json
CHANGED
|
@@ -20085,6 +20085,272 @@
|
|
| 20085 |
"selected_for_formal_metrics": true,
|
| 20086 |
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/06/seed-0/"
|
| 20087 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 20088 |
{
|
| 20089 |
"id": "task06-08-seed0-formal",
|
| 20090 |
"task_key": "task06/08",
|
|
|
|
| 20085 |
"selected_for_formal_metrics": true,
|
| 20086 |
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/06/seed-0/"
|
| 20087 |
},
|
| 20088 |
+
{
|
| 20089 |
+
"id": "task06-07-seed0-formal",
|
| 20090 |
+
"task_key": "task06/07",
|
| 20091 |
+
"family": "task06",
|
| 20092 |
+
"slot": "07",
|
| 20093 |
+
"seed": 0,
|
| 20094 |
+
"episode": 1,
|
| 20095 |
+
"phase": "formal",
|
| 20096 |
+
"status": "completed",
|
| 20097 |
+
"success": false,
|
| 20098 |
+
"native_reward": 0.0,
|
| 20099 |
+
"valid": true,
|
| 20100 |
+
"execution": {
|
| 20101 |
+
"reason": null,
|
| 20102 |
+
"status": "finished"
|
| 20103 |
+
},
|
| 20104 |
+
"verdict": {
|
| 20105 |
+
"evidence_valid": true,
|
| 20106 |
+
"steps": 8645,
|
| 20107 |
+
"success": false,
|
| 20108 |
+
"termination": "stopped"
|
| 20109 |
+
},
|
| 20110 |
+
"steps": 8645,
|
| 20111 |
+
"simulation_time_s": null,
|
| 20112 |
+
"wall_time_s": 2551.587007,
|
| 20113 |
+
"model": "gpt-6-astra",
|
| 20114 |
+
"effort": "high",
|
| 20115 |
+
"harness": "codex",
|
| 20116 |
+
"codex_version": "0.160.0",
|
| 20117 |
+
"native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 20118 |
+
"instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 20119 |
+
"instruction_policy": "original_native",
|
| 20120 |
+
"usage": {
|
| 20121 |
+
"accounting": "reported-responses",
|
| 20122 |
+
"audit_complete": true,
|
| 20123 |
+
"cache_hit_rate": 0.9872509714529868,
|
| 20124 |
+
"cache_reported_input_tokens": 16738844,
|
| 20125 |
+
"cache_write_input_tokens": 0,
|
| 20126 |
+
"cache_write_reported_input_tokens": 16738844,
|
| 20127 |
+
"cached_input_tokens": 16525440,
|
| 20128 |
+
"completed_turns": 1,
|
| 20129 |
+
"cost_usd": null,
|
| 20130 |
+
"failed_turns": 0,
|
| 20131 |
+
"input_tokens": 16738844,
|
| 20132 |
+
"known_cache_write_input_tokens": 0,
|
| 20133 |
+
"known_cached_input_tokens": 16525440,
|
| 20134 |
+
"known_input_tokens": 16738844,
|
| 20135 |
+
"known_output_tokens": 38724,
|
| 20136 |
+
"known_reasoning_output_tokens": 20279,
|
| 20137 |
+
"output_tokens": 38724,
|
| 20138 |
+
"reasoning_output_tokens": 20279,
|
| 20139 |
+
"reasoning_reported_output_tokens": 38724,
|
| 20140 |
+
"reported_responses": {
|
| 20141 |
+
"cache_reported_input_tokens": 192,
|
| 20142 |
+
"cache_write_input_tokens": 192,
|
| 20143 |
+
"cache_write_reported_input_tokens": 192,
|
| 20144 |
+
"cached_input_tokens": 192,
|
| 20145 |
+
"input_tokens": 192,
|
| 20146 |
+
"output_tokens": 192,
|
| 20147 |
+
"reasoning_output_tokens": 192,
|
| 20148 |
+
"reasoning_reported_output_tokens": 192
|
| 20149 |
+
},
|
| 20150 |
+
"response_count": 192,
|
| 20151 |
+
"response_ids_complete": true,
|
| 20152 |
+
"schema": "rlebench/token-usage/1",
|
| 20153 |
+
"source": "Codex token_usage_record per response",
|
| 20154 |
+
"uncached_input_tokens": 213404,
|
| 20155 |
+
"unidentified_usage_records": 0
|
| 20156 |
+
},
|
| 20157 |
+
"call_activity": {
|
| 20158 |
+
"model_tool_calls": 191,
|
| 20159 |
+
"model_tool_calls_by_name": {
|
| 20160 |
+
"exec": 191
|
| 20161 |
+
},
|
| 20162 |
+
"nested_python_tool_invocations": null,
|
| 20163 |
+
"python_device_rpc_attempts": null,
|
| 20164 |
+
"python_device_rpc_attempts_by_action": null,
|
| 20165 |
+
"python_device_rpc_errors": null,
|
| 20166 |
+
"python_instrumented_model_tool_calls": null,
|
| 20167 |
+
"python_tool_invocations": null,
|
| 20168 |
+
"python_tool_invocations_by_origin": null,
|
| 20169 |
+
"schema": "rlebench/call-activity/1",
|
| 20170 |
+
"source": "Codex native sessions"
|
| 20171 |
+
},
|
| 20172 |
+
"media": {
|
| 20173 |
+
"passed": true,
|
| 20174 |
+
"width": 1280,
|
| 20175 |
+
"height": 360,
|
| 20176 |
+
"duration_s": 43.25,
|
| 20177 |
+
"speed": 4,
|
| 20178 |
+
"source_fps": 10,
|
| 20179 |
+
"output_fps": 20,
|
| 20180 |
+
"recording": {
|
| 20181 |
+
"accepted_samples": 1730,
|
| 20182 |
+
"captured_samples": 1730,
|
| 20183 |
+
"clock": "simulation",
|
| 20184 |
+
"dropped_samples": 0,
|
| 20185 |
+
"encoded_frames": 1730,
|
| 20186 |
+
"end_time_s": 172.90000000001464,
|
| 20187 |
+
"error": null,
|
| 20188 |
+
"experimental": true,
|
| 20189 |
+
"fps": 10,
|
| 20190 |
+
"received_samples": 1730,
|
| 20191 |
+
"schema": "roboenv/recording/1",
|
| 20192 |
+
"state": "closed",
|
| 20193 |
+
"status": "complete",
|
| 20194 |
+
"views": [
|
| 20195 |
+
{
|
| 20196 |
+
"fov_y": 45.0,
|
| 20197 |
+
"height": 360,
|
| 20198 |
+
"name": "camera_head_left",
|
| 20199 |
+
"pose": null,
|
| 20200 |
+
"source": "camera_head_left",
|
| 20201 |
+
"width": 640
|
| 20202 |
+
},
|
| 20203 |
+
{
|
| 20204 |
+
"fov_y": 45.0,
|
| 20205 |
+
"height": 360,
|
| 20206 |
+
"name": "camera_head_right",
|
| 20207 |
+
"pose": null,
|
| 20208 |
+
"source": "camera_head_right",
|
| 20209 |
+
"width": 640
|
| 20210 |
+
}
|
| 20211 |
+
]
|
| 20212 |
+
},
|
| 20213 |
+
"view_names": [
|
| 20214 |
+
"camera_head_left",
|
| 20215 |
+
"camera_head_right"
|
| 20216 |
+
],
|
| 20217 |
+
"sha256": "95adbcd6adbca86aa7e6a6cc90c6fe2930fa1d430ac07da60e5756b01ad3dd3b",
|
| 20218 |
+
"scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
|
| 20219 |
+
},
|
| 20220 |
+
"analysis": {
|
| 20221 |
+
"outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
|
| 20222 |
+
"duration": "\u4f7f\u7528 8,645 / 10,000 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
|
| 20223 |
+
"execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
|
| 20224 |
+
"cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
|
| 20225 |
+
},
|
| 20226 |
+
"provenance": {
|
| 20227 |
+
"sources": {
|
| 20228 |
+
"RLE-Bench-inhouse": {
|
| 20229 |
+
"build_inputs": [
|
| 20230 |
+
"pyproject.toml",
|
| 20231 |
+
"src",
|
| 20232 |
+
"tasks",
|
| 20233 |
+
"README.md",
|
| 20234 |
+
"Makefile",
|
| 20235 |
+
"tests",
|
| 20236 |
+
"docs"
|
| 20237 |
+
],
|
| 20238 |
+
"content_sha256": "5b1428b96104cb17b52215bce0ec33b880c5c591d01d4ecc2867cdc647940bda",
|
| 20239 |
+
"dirty": true,
|
| 20240 |
+
"revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
|
| 20241 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
|
| 20242 |
+
},
|
| 20243 |
+
"RoboEnv": {
|
| 20244 |
+
"build_inputs": [
|
| 20245 |
+
"pyproject.toml",
|
| 20246 |
+
"README.md",
|
| 20247 |
+
"src",
|
| 20248 |
+
"runtime/pyproject.toml",
|
| 20249 |
+
"runtime/README.md",
|
| 20250 |
+
"runtime/src",
|
| 20251 |
+
"runtime/environments.json",
|
| 20252 |
+
"runtime/locks",
|
| 20253 |
+
"catalog",
|
| 20254 |
+
"upstreams.lock.json",
|
| 20255 |
+
"third_party/patches",
|
| 20256 |
+
"docs/validation"
|
| 20257 |
+
],
|
| 20258 |
+
"content_sha256": "5e90464ff5b67edba2a8e653e68cba5f7065c19f92f34a29689703d61e04d305",
|
| 20259 |
+
"dirty": true,
|
| 20260 |
+
"revision": "b73356c09507eeffaf9ca9e90b668ef0b7f1fa92",
|
| 20261 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
|
| 20262 |
+
},
|
| 20263 |
+
"kinex": {
|
| 20264 |
+
"build_inputs": [
|
| 20265 |
+
"package.json",
|
| 20266 |
+
"package-lock.json",
|
| 20267 |
+
".nvmrc",
|
| 20268 |
+
"tsconfig.json",
|
| 20269 |
+
"VERSION",
|
| 20270 |
+
"src",
|
| 20271 |
+
"packages/core",
|
| 20272 |
+
"packages/setup",
|
| 20273 |
+
"script",
|
| 20274 |
+
"assets",
|
| 20275 |
+
"bin"
|
| 20276 |
+
],
|
| 20277 |
+
"content_sha256": "70f5107905efa45cace74e04b630660aaee4c80e1db2e3d540d6c81e0f4d0d38",
|
| 20278 |
+
"dirty": false,
|
| 20279 |
+
"revision": "caac19a8a36272972f762e0f74cfe381e8500048",
|
| 20280 |
+
"source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/kinex"
|
| 20281 |
+
}
|
| 20282 |
+
},
|
| 20283 |
+
"job": "task06-07-pick-and-place-and-hug-container-codex-seed0-attempt01",
|
| 20284 |
+
"attempt": 1,
|
| 20285 |
+
"harness": "stock Codex CLI",
|
| 20286 |
+
"control_interface": "public RoboEnv SDK and CLI",
|
| 20287 |
+
"kinex_agent_runtime_used": false,
|
| 20288 |
+
"gpu_index": 3,
|
| 20289 |
+
"gpu_model": "NVIDIA L40S",
|
| 20290 |
+
"campaign_dispatch_concurrency_limit": 5,
|
| 20291 |
+
"concurrency_note": "One GPU per active SIMPLE episode; Kinex and Codex share a five-GPU dispatch pool.",
|
| 20292 |
+
"codex_version": "0.160.0",
|
| 20293 |
+
"model": "gpt-6-astra",
|
| 20294 |
+
"effort": "high",
|
| 20295 |
+
"service_tier": "default",
|
| 20296 |
+
"fresh_session": true,
|
| 20297 |
+
"source_jobs": [],
|
| 20298 |
+
"resume_trajectory": false,
|
| 20299 |
+
"imported_skills": [],
|
| 20300 |
+
"automatic_harbor_retries": 0,
|
| 20301 |
+
"request_policy": {
|
| 20302 |
+
"max_request_retries": 50,
|
| 20303 |
+
"configuration": "explicit retry50 SSE",
|
| 20304 |
+
"usage_accounting": "reported-responses"
|
| 20305 |
+
},
|
| 20306 |
+
"classification": "formal",
|
| 20307 |
+
"measured_images": {
|
| 20308 |
+
"agent": "sha256:3fb135c4db8092900969023cfe52e4bddfc8960b6296318fda47c72e80e5f667",
|
| 20309 |
+
"task": "sha256:730c5fc8ca66ce7bb07635943de71bd443d99c4f6ce962e01d35e90a3aec5c28"
|
| 20310 |
+
},
|
| 20311 |
+
"session_original_sha256": "f57a16755d9b599e6a03f0620b4092819f73f48643c586385c5d54ffa8ff4115",
|
| 20312 |
+
"protocol_sha256": "6e24ceebccc59c6e91caf708994c458548bbb4c3539b206b9d85eebcb7ea6266"
|
| 20313 |
+
},
|
| 20314 |
+
"links": {
|
| 20315 |
+
"native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/session.jsonl",
|
| 20316 |
+
"trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/trajectory.json",
|
| 20317 |
+
"provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/provider-usage.jsonl",
|
| 20318 |
+
"transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/transcript.json",
|
| 20319 |
+
"verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/episode.json",
|
| 20320 |
+
"protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/protocol.json",
|
| 20321 |
+
"native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/instructions.json",
|
| 20322 |
+
"workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/workspace.tar.gz",
|
| 20323 |
+
"workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/workspace.json",
|
| 20324 |
+
"owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
|
| 20325 |
+
"recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/recording/manifest.json",
|
| 20326 |
+
"usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/usage.json",
|
| 20327 |
+
"analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/analysis.json",
|
| 20328 |
+
"provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/provenance.json",
|
| 20329 |
+
"video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/video.mp4",
|
| 20330 |
+
"poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/poster.jpg",
|
| 20331 |
+
"media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/media-validation.json",
|
| 20332 |
+
"final_observation": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/final-observation.json"
|
| 20333 |
+
},
|
| 20334 |
+
"resources": [
|
| 20335 |
+
{
|
| 20336 |
+
"name": "tools/robot.py",
|
| 20337 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/resources/tools/robot.py",
|
| 20338 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 20339 |
+
},
|
| 20340 |
+
{
|
| 20341 |
+
"name": "memos/g1_simple_manipulation.md",
|
| 20342 |
+
"file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/resources/memos/g1_simple_manipulation.md",
|
| 20343 |
+
"kind": "Created during this episode; final workspace snapshot."
|
| 20344 |
+
}
|
| 20345 |
+
],
|
| 20346 |
+
"session_counts": {
|
| 20347 |
+
"visible_events": 414,
|
| 20348 |
+
"observed_images": 108,
|
| 20349 |
+
"tool_errors": 1
|
| 20350 |
+
},
|
| 20351 |
+
"selected_for_formal_metrics": true,
|
| 20352 |
+
"asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/"
|
| 20353 |
+
},
|
| 20354 |
{
|
| 20355 |
"id": "task06-08-seed0-formal",
|
| 20356 |
"task_key": "task06/08",
|
manifest.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
| 1 |
{
|
| 2 |
".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
|
| 3 |
"DATA_FORMAT.md": "e2fd4f6b5c8be9d9e008db7d0c1f5d52b419d68328a1c86625c9f67c49b03403",
|
| 4 |
-
"README.md": "
|
| 5 |
-
"REPORT.md": "
|
| 6 |
"THIRD_PARTY_NOTICES.md": "8aa8e3db100d7a570931c5d87f44f78705b42cf17e8147016b35e9aeddb06bba",
|
| 7 |
"app.js": "eb013cc9d24382410953a1a8c81e446a6592a81b32887749d17bfc863936556b",
|
| 8 |
"attempt-history.json": "8555f458c1dd362f5a07447342994fddf265ff23bbc40e9bd6f9984a892ddb4e",
|
|
@@ -19,9 +19,9 @@
|
|
| 19 |
"catalog/task06/11-push-office-chair/task.yaml": "ab19bf9b7cec8992e2dcc1fb30ad9f6ce0f8343063b44777b2b68afc39d7f426",
|
| 20 |
"catalog/task06/12-open-trash-can/task.yaml": "8d8b4b0ec848f91283e73416d257f5b0f816ac4b872fccd10b213e74d953085c",
|
| 21 |
"comparison.json": "487d479f3d03e72efaf3c19c1e63f5894bcfea2954cb1cc46327c379b5315ad3",
|
| 22 |
-
"data.json": "
|
| 23 |
-
"episodes.csv": "
|
| 24 |
-
"episodes.json": "
|
| 25 |
"failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
|
| 26 |
"favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
|
| 27 |
"index.html": "524b3f71cc36152a08125212ad1c9ae2ca30601907b378f69cce4d34390d0ee3",
|
|
@@ -30,8 +30,8 @@
|
|
| 30 |
"licenses/RoboTwin-LICENSE.txt": "c695d421718e54e6f3a60858f0c601125c3ecc60c199332c15947360561e2a9f",
|
| 31 |
"licenses/robocasa-LICENSE.txt": "5da18670b3f00c59847b1ded9c28dee59940d963b1e03b528b0108d9c5a09885",
|
| 32 |
"licenses/robosuite-LICENSE.txt": "177978cbece0a4c454c2aaec5b3f145b39270814874c43109da9e829c39d9cba",
|
| 33 |
-
"protocol.json": "
|
| 34 |
"style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
|
| 35 |
-
"summary.json": "
|
| 36 |
-
"task-index.json": "
|
| 37 |
}
|
|
|
|
| 1 |
{
|
| 2 |
".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
|
| 3 |
"DATA_FORMAT.md": "e2fd4f6b5c8be9d9e008db7d0c1f5d52b419d68328a1c86625c9f67c49b03403",
|
| 4 |
+
"README.md": "a69c662b67b32d762b71db8bfb458f33823a2ac65814cb13bf998f99c90cf53a",
|
| 5 |
+
"REPORT.md": "e8c55b5928c2ded32606f1552fd73634247d7921658cab0edbc320e0ffc84445",
|
| 6 |
"THIRD_PARTY_NOTICES.md": "8aa8e3db100d7a570931c5d87f44f78705b42cf17e8147016b35e9aeddb06bba",
|
| 7 |
"app.js": "eb013cc9d24382410953a1a8c81e446a6592a81b32887749d17bfc863936556b",
|
| 8 |
"attempt-history.json": "8555f458c1dd362f5a07447342994fddf265ff23bbc40e9bd6f9984a892ddb4e",
|
|
|
|
| 19 |
"catalog/task06/11-push-office-chair/task.yaml": "ab19bf9b7cec8992e2dcc1fb30ad9f6ce0f8343063b44777b2b68afc39d7f426",
|
| 20 |
"catalog/task06/12-open-trash-can/task.yaml": "8d8b4b0ec848f91283e73416d257f5b0f816ac4b872fccd10b213e74d953085c",
|
| 21 |
"comparison.json": "487d479f3d03e72efaf3c19c1e63f5894bcfea2954cb1cc46327c379b5315ad3",
|
| 22 |
+
"data.json": "ac940eeb5e78302389d387aff8cccde4e81ecdf4ccea4d6fbb7b54f43e1ccd51",
|
| 23 |
+
"episodes.csv": "7f4d5f46d65689ff303609a98ffa8a34ad875e7cc9b6018628fc368e87070ec9",
|
| 24 |
+
"episodes.json": "41c1c467c7c36745454b1f1da54e2cbd25823af225581cfa168472878f4ea0f9",
|
| 25 |
"failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
|
| 26 |
"favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
|
| 27 |
"index.html": "524b3f71cc36152a08125212ad1c9ae2ca30601907b378f69cce4d34390d0ee3",
|
|
|
|
| 30 |
"licenses/RoboTwin-LICENSE.txt": "c695d421718e54e6f3a60858f0c601125c3ecc60c199332c15947360561e2a9f",
|
| 31 |
"licenses/robocasa-LICENSE.txt": "5da18670b3f00c59847b1ded9c28dee59940d963b1e03b528b0108d9c5a09885",
|
| 32 |
"licenses/robosuite-LICENSE.txt": "177978cbece0a4c454c2aaec5b3f145b39270814874c43109da9e829c39d9cba",
|
| 33 |
+
"protocol.json": "96be82ffa28eedefaa70bff4f7d4daa9e56f70fed407c1beac1ce8b725a00861",
|
| 34 |
"style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
|
| 35 |
+
"summary.json": "dbb20c916a74b687e14aece840ffe67138889d90d3bd4b7d389593d497c9aede",
|
| 36 |
+
"task-index.json": "1f9542d87396e5e80c842c021f081077ab3f8cbace59c1bf61593b587d9020bc"
|
| 37 |
}
|
protocol.json
CHANGED
|
@@ -95,21 +95,21 @@
|
|
| 95 |
"id": "task06",
|
| 96 |
"name": "SIMPLE G1",
|
| 97 |
"total": 12,
|
| 98 |
-
"completed":
|
| 99 |
-
"pending":
|
| 100 |
"successes": 2,
|
| 101 |
-
"valid_results":
|
| 102 |
-
"success_rate": 0.
|
| 103 |
-
"usage_complete":
|
| 104 |
"control_frequency_hz": 50,
|
| 105 |
"max_control_steps": 10000,
|
| 106 |
"preflight_results": 0,
|
| 107 |
-
"formal_results":
|
| 108 |
"modified_results": 0,
|
| 109 |
-
"original_results":
|
| 110 |
-
"input_tokens":
|
| 111 |
-
"cached_input_tokens":
|
| 112 |
-
"output_tokens":
|
| 113 |
}
|
| 114 |
],
|
| 115 |
"usage_accounting": "reported-responses",
|
|
|
|
| 95 |
"id": "task06",
|
| 96 |
"name": "SIMPLE G1",
|
| 97 |
"total": 12,
|
| 98 |
+
"completed": 8,
|
| 99 |
+
"pending": 4,
|
| 100 |
"successes": 2,
|
| 101 |
+
"valid_results": 8,
|
| 102 |
+
"success_rate": 0.25,
|
| 103 |
+
"usage_complete": 8,
|
| 104 |
"control_frequency_hz": 50,
|
| 105 |
"max_control_steps": 10000,
|
| 106 |
"preflight_results": 0,
|
| 107 |
+
"formal_results": 8,
|
| 108 |
"modified_results": 0,
|
| 109 |
+
"original_results": 8,
|
| 110 |
+
"input_tokens": 108175382,
|
| 111 |
+
"cached_input_tokens": 106998016,
|
| 112 |
+
"output_tokens": 283858
|
| 113 |
}
|
| 114 |
],
|
| 115 |
"usage_accounting": "reported-responses",
|
publication-manifest.json
CHANGED
|
@@ -10,11 +10,11 @@
|
|
| 10 |
"bytes": 1129
|
| 11 |
},
|
| 12 |
"README.md": {
|
| 13 |
-
"sha256": "
|
| 14 |
"bytes": 1544
|
| 15 |
},
|
| 16 |
"REPORT.md": {
|
| 17 |
-
"sha256": "
|
| 18 |
"bytes": 1421
|
| 19 |
},
|
| 20 |
"THIRD_PARTY_NOTICES.md": {
|
|
@@ -82,16 +82,16 @@
|
|
| 82 |
"bytes": 134701
|
| 83 |
},
|
| 84 |
"data.json": {
|
| 85 |
-
"sha256": "
|
| 86 |
-
"bytes":
|
| 87 |
},
|
| 88 |
"episodes.csv": {
|
| 89 |
-
"sha256": "
|
| 90 |
-
"bytes":
|
| 91 |
},
|
| 92 |
"episodes.json": {
|
| 93 |
-
"sha256": "
|
| 94 |
-
"bytes":
|
| 95 |
},
|
| 96 |
"failure-reviews.json": {
|
| 97 |
"sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
|
|
@@ -126,23 +126,23 @@
|
|
| 126 |
"bytes": 1474
|
| 127 |
},
|
| 128 |
"protocol.json": {
|
| 129 |
-
"sha256": "
|
| 130 |
-
"bytes":
|
| 131 |
},
|
| 132 |
"style.css": {
|
| 133 |
"sha256": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
|
| 134 |
"bytes": 16560
|
| 135 |
},
|
| 136 |
"summary.json": {
|
| 137 |
-
"sha256": "
|
| 138 |
-
"bytes":
|
| 139 |
},
|
| 140 |
"task-index.json": {
|
| 141 |
-
"sha256": "
|
| 142 |
-
"bytes":
|
| 143 |
},
|
| 144 |
"manifest.json": {
|
| 145 |
-
"sha256": "
|
| 146 |
"bytes": 3481
|
| 147 |
}
|
| 148 |
}
|
|
|
|
| 10 |
"bytes": 1129
|
| 11 |
},
|
| 12 |
"README.md": {
|
| 13 |
+
"sha256": "a69c662b67b32d762b71db8bfb458f33823a2ac65814cb13bf998f99c90cf53a",
|
| 14 |
"bytes": 1544
|
| 15 |
},
|
| 16 |
"REPORT.md": {
|
| 17 |
+
"sha256": "e8c55b5928c2ded32606f1552fd73634247d7921658cab0edbc320e0ffc84445",
|
| 18 |
"bytes": 1421
|
| 19 |
},
|
| 20 |
"THIRD_PARTY_NOTICES.md": {
|
|
|
|
| 82 |
"bytes": 134701
|
| 83 |
},
|
| 84 |
"data.json": {
|
| 85 |
+
"sha256": "ac940eeb5e78302389d387aff8cccde4e81ecdf4ccea4d6fbb7b54f43e1ccd51",
|
| 86 |
+
"bytes": 1166026
|
| 87 |
},
|
| 88 |
"episodes.csv": {
|
| 89 |
+
"sha256": "7f4d5f46d65689ff303609a98ffa8a34ad875e7cc9b6018628fc368e87070ec9",
|
| 90 |
+
"bytes": 8111
|
| 91 |
},
|
| 92 |
"episodes.json": {
|
| 93 |
+
"sha256": "41c1c467c7c36745454b1f1da54e2cbd25823af225581cfa168472878f4ea0f9",
|
| 94 |
+
"bytes": 874883
|
| 95 |
},
|
| 96 |
"failure-reviews.json": {
|
| 97 |
"sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
|
|
|
|
| 126 |
"bytes": 1474
|
| 127 |
},
|
| 128 |
"protocol.json": {
|
| 129 |
+
"sha256": "96be82ffa28eedefaa70bff4f7d4daa9e56f70fed407c1beac1ce8b725a00861",
|
| 130 |
+
"bytes": 5438
|
| 131 |
},
|
| 132 |
"style.css": {
|
| 133 |
"sha256": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
|
| 134 |
"bytes": 16560
|
| 135 |
},
|
| 136 |
"summary.json": {
|
| 137 |
+
"sha256": "dbb20c916a74b687e14aece840ffe67138889d90d3bd4b7d389593d497c9aede",
|
| 138 |
+
"bytes": 3209
|
| 139 |
},
|
| 140 |
"task-index.json": {
|
| 141 |
+
"sha256": "1f9542d87396e5e80c842c021f081077ab3f8cbace59c1bf61593b587d9020bc",
|
| 142 |
+
"bytes": 229780
|
| 143 |
},
|
| 144 |
"manifest.json": {
|
| 145 |
+
"sha256": "f30322dfb31fe1049111c2bd2c2f026804ce81f9a7d025c7c90904e505fa5261",
|
| 146 |
"bytes": 3481
|
| 147 |
}
|
| 148 |
}
|
summary.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
| 1 |
{
|
| 2 |
"planned_tasks": 84,
|
| 3 |
-
"published_results":
|
| 4 |
-
"pending_tasks":
|
| 5 |
"families": [
|
| 6 |
{
|
| 7 |
"id": "task01",
|
|
@@ -88,40 +88,40 @@
|
|
| 88 |
"id": "task06",
|
| 89 |
"name": "SIMPLE G1",
|
| 90 |
"total": 12,
|
| 91 |
-
"completed":
|
| 92 |
-
"pending":
|
| 93 |
"successes": 2,
|
| 94 |
-
"valid_results":
|
| 95 |
-
"success_rate": 0.
|
| 96 |
-
"usage_complete":
|
| 97 |
"control_frequency_hz": 50,
|
| 98 |
"max_control_steps": 10000,
|
| 99 |
"preflight_results": 0,
|
| 100 |
-
"formal_results":
|
| 101 |
"modified_results": 0,
|
| 102 |
-
"original_results":
|
| 103 |
-
"input_tokens":
|
| 104 |
-
"cached_input_tokens":
|
| 105 |
-
"output_tokens":
|
| 106 |
}
|
| 107 |
],
|
| 108 |
"interrupted_attempts": 4,
|
| 109 |
"usage_incomplete_tasks": [],
|
| 110 |
"execution_incomplete_tasks": [],
|
| 111 |
"deferred_tasks": [],
|
| 112 |
-
"new_evaluations":
|
| 113 |
"progress": {
|
| 114 |
-
"finished":
|
| 115 |
"running": 3,
|
| 116 |
-
"queued":
|
| 117 |
"needs_review": 0,
|
| 118 |
"interrupted": 0,
|
| 119 |
"native_successes": 51,
|
| 120 |
-
"native_failures":
|
| 121 |
},
|
| 122 |
"simple_campaign": {
|
| 123 |
"planned": 12,
|
| 124 |
-
"published":
|
| 125 |
"agent": "codex",
|
| 126 |
"model": "gpt-6-astra",
|
| 127 |
"effort": "high",
|
|
|
|
| 1 |
{
|
| 2 |
"planned_tasks": 84,
|
| 3 |
+
"published_results": 80,
|
| 4 |
+
"pending_tasks": 4,
|
| 5 |
"families": [
|
| 6 |
{
|
| 7 |
"id": "task01",
|
|
|
|
| 88 |
"id": "task06",
|
| 89 |
"name": "SIMPLE G1",
|
| 90 |
"total": 12,
|
| 91 |
+
"completed": 8,
|
| 92 |
+
"pending": 4,
|
| 93 |
"successes": 2,
|
| 94 |
+
"valid_results": 8,
|
| 95 |
+
"success_rate": 0.25,
|
| 96 |
+
"usage_complete": 8,
|
| 97 |
"control_frequency_hz": 50,
|
| 98 |
"max_control_steps": 10000,
|
| 99 |
"preflight_results": 0,
|
| 100 |
+
"formal_results": 8,
|
| 101 |
"modified_results": 0,
|
| 102 |
+
"original_results": 8,
|
| 103 |
+
"input_tokens": 108175382,
|
| 104 |
+
"cached_input_tokens": 106998016,
|
| 105 |
+
"output_tokens": 283858
|
| 106 |
}
|
| 107 |
],
|
| 108 |
"interrupted_attempts": 4,
|
| 109 |
"usage_incomplete_tasks": [],
|
| 110 |
"execution_incomplete_tasks": [],
|
| 111 |
"deferred_tasks": [],
|
| 112 |
+
"new_evaluations": 8,
|
| 113 |
"progress": {
|
| 114 |
+
"finished": 80,
|
| 115 |
"running": 3,
|
| 116 |
+
"queued": 1,
|
| 117 |
"needs_review": 0,
|
| 118 |
"interrupted": 0,
|
| 119 |
"native_successes": 51,
|
| 120 |
+
"native_failures": 29
|
| 121 |
},
|
| 122 |
"simple_campaign": {
|
| 123 |
"planned": 12,
|
| 124 |
+
"published": 8,
|
| 125 |
"agent": "codex",
|
| 126 |
"model": "gpt-6-astra",
|
| 127 |
"effort": "high",
|
task-index.json
CHANGED
|
@@ -6232,10 +6232,10 @@
|
|
| 6232 |
"catalog_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 6233 |
"native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 6234 |
"instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
|
| 6235 |
-
"status": "
|
| 6236 |
-
"episode_id":
|
| 6237 |
-
"run_status": "
|
| 6238 |
-
"status_note": "
|
| 6239 |
"planned_protocol": {
|
| 6240 |
"episodes": 1,
|
| 6241 |
"seed": 0,
|
|
@@ -6354,7 +6354,7 @@
|
|
| 6354 |
"instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
|
| 6355 |
"status": "pending",
|
| 6356 |
"episode_id": null,
|
| 6357 |
-
"run_status": "
|
| 6358 |
"status_note": "Authorized episode is queued, running, or awaiting evidence review.",
|
| 6359 |
"planned_protocol": {
|
| 6360 |
"episodes": 1,
|
|
|
|
| 6232 |
"catalog_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 6233 |
"native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
|
| 6234 |
"instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
|
| 6235 |
+
"status": "completed",
|
| 6236 |
+
"episode_id": "task06-07-seed0-formal",
|
| 6237 |
+
"run_status": "finished",
|
| 6238 |
+
"status_note": "Native result, execution and usage complete.",
|
| 6239 |
"planned_protocol": {
|
| 6240 |
"episodes": 1,
|
| 6241 |
"seed": 0,
|
|
|
|
| 6354 |
"instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
|
| 6355 |
"status": "pending",
|
| 6356 |
"episode_id": null,
|
| 6357 |
+
"run_status": "running",
|
| 6358 |
"status_note": "Authorized episode is queued, running, or awaiting evidence review.",
|
| 6359 |
"planned_protocol": {
|
| 6360 |
"episodes": 1,
|