yzzhao commited on
Commit
b3186de
·
verified ·
1 Parent(s): f778dd1

Publish Codex Benchmark evaluation evidence

Browse files
Files changed (11) hide show
  1. README.md +1 -1
  2. REPORT.md +1 -1
  3. comparison.json +6 -6
  4. data.json +283 -32
  5. episodes.csv +1 -1
  6. episodes.json +251 -0
  7. manifest.json +9 -9
  8. protocol.json +11 -11
  9. publication-manifest.json +15 -15
  10. summary.json +16 -16
  11. task-index.json +4 -4
README.md CHANGED
@@ -10,7 +10,7 @@ pinned: false
10
 
11
  # Codex Benchmark
12
 
13
- 38/42 ordinary RoboDojo tasks published: 30 successes, 8 valid native failures, 4 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
14
 
15
  Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
16
 
 
10
 
11
  # Codex Benchmark
12
 
13
+ 39/42 ordinary RoboDojo tasks published: 31 successes, 8 valid native failures, 3 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
14
 
15
  Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
16
 
REPORT.md CHANGED
@@ -1,6 +1,6 @@
1
  # Codex Benchmark
2
 
3
- 38/42 ordinary RoboDojo tasks published: 30 successes, 8 valid native failures, 4 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
4
 
5
  Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
6
 
 
1
  # Codex Benchmark
2
 
3
+ 39/42 ordinary RoboDojo tasks published: 31 successes, 8 valid native failures, 3 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
4
 
5
  Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
6
 
comparison.json CHANGED
@@ -1,10 +1,10 @@
1
  {
2
  "baseline_revision": "1a7f1c9f178ba7aeb659e5b527a77978e9136072",
3
  "summary": {
4
- "completed": 38,
5
- "codex_successes": 30,
6
- "codex_success_rate_completed_subset": 0.7894736842105263,
7
- "kinex_successes_same_subset": 33,
8
  "kinex_successes_all42": 35,
9
  "all42_comparison_complete": false
10
  },
@@ -60,8 +60,8 @@
60
  {
61
  "task_key": "task04/07",
62
  "native_id": "robodojo/insert-tubes",
63
- "status": "pending",
64
- "codex_success": null,
65
  "kinex_success": true,
66
  "kinex_version": "0.10.3"
67
  },
 
1
  {
2
  "baseline_revision": "1a7f1c9f178ba7aeb659e5b527a77978e9136072",
3
  "summary": {
4
+ "completed": 39,
5
+ "codex_successes": 31,
6
+ "codex_success_rate_completed_subset": 0.7948717948717948,
7
+ "kinex_successes_same_subset": 34,
8
  "kinex_successes_all42": 35,
9
  "all42_comparison_complete": false
10
  },
 
60
  {
61
  "task_key": "task04/07",
62
  "native_id": "robodojo/insert-tubes",
63
+ "status": "completed",
64
+ "codex_success": true,
65
  "kinex_success": true,
66
  "kinex_version": "0.10.3"
67
  },
data.json CHANGED
@@ -6,7 +6,7 @@
6
  "benchmark_complete": false,
7
  "edition": "plain-codex-astra-high-seed0",
8
  "created_at": "2026-10-09T00:09:42.989838+00:00",
9
- "updated_at": "2026-10-09T04:32:58.105489+00:00",
10
  "model": "gpt-6-astra",
11
  "effort": "high",
12
  "seed": 0,
@@ -17,38 +17,38 @@
17
  },
18
  "summary": {
19
  "planned_tasks": 42,
20
- "published_results": 38,
21
- "pending_tasks": 4,
22
  "families": [
23
  {
24
  "id": "task04",
25
  "name": "RoboDojo",
26
  "total": 42,
27
- "completed": 38,
28
- "pending": 4,
29
- "successes": 30,
30
- "valid_results": 38,
31
- "success_rate": 0.7894736842105263,
32
- "input_tokens": 217872800,
33
- "cached_input_tokens": 213964032,
34
- "output_tokens": 824034,
35
- "usage_complete": 38,
36
  "control_frequency_hz": 25,
37
  "max_control_steps": 7500,
38
  "preflight_results": 0,
39
- "formal_results": 38,
40
- "modified_results": 30,
41
  "original_results": 8
42
  }
43
  ],
44
  "progress": {
45
- "finished": 38,
46
  "interrupted": 2,
47
  "native_failures": 8,
48
- "native_successes": 30,
49
  "needs_review": 0,
50
  "queued": 0,
51
- "running": 2
52
  },
53
  "interrupted_attempts": 4,
54
  "usage_incomplete_tasks": [],
@@ -59,20 +59,20 @@
59
  "id": "task04",
60
  "name": "RoboDojo",
61
  "total": 42,
62
- "completed": 38,
63
- "pending": 4,
64
- "successes": 30,
65
- "valid_results": 38,
66
- "success_rate": 0.7894736842105263,
67
- "input_tokens": 217872800,
68
- "cached_input_tokens": 213964032,
69
- "output_tokens": 824034,
70
- "usage_complete": 38,
71
  "control_frequency_hz": 25,
72
  "max_control_steps": 7500,
73
  "preflight_results": 0,
74
- "formal_results": 38,
75
- "modified_results": 30,
76
  "original_results": 8
77
  }
78
  ],
@@ -670,8 +670,8 @@
670
  "catalog_instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
671
  "native_instruction": "Insert the three tubes into the rack one by one.",
672
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
673
- "status": "pending",
674
- "episode_id": null,
675
  "planned_protocol": {
676
  "episodes": 1,
677
  "seed": 0,
@@ -727,8 +727,8 @@
727
  },
728
  "display_slot": "07",
729
  "display_key": "task04/07",
730
- "run_status": "running",
731
- "status_note": "Currently running.",
732
  "attempt_history": [
733
  {
734
  "id": "07-robodojo-insert-tubes-codex-seed0-attempt01",
@@ -5123,6 +5123,257 @@
5123
  "selected_for_formal_metrics": true,
5124
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/06/seed-0/"
5125
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5126
  {
5127
  "id": "task04-08-seed0-formal",
5128
  "task_key": "task04/08",
 
6
  "benchmark_complete": false,
7
  "edition": "plain-codex-astra-high-seed0",
8
  "created_at": "2026-10-09T00:09:42.989838+00:00",
9
+ "updated_at": "2026-10-09T05:17:09.534508+00:00",
10
  "model": "gpt-6-astra",
11
  "effort": "high",
12
  "seed": 0,
 
17
  },
18
  "summary": {
19
  "planned_tasks": 42,
20
+ "published_results": 39,
21
+ "pending_tasks": 3,
22
  "families": [
23
  {
24
  "id": "task04",
25
  "name": "RoboDojo",
26
  "total": 42,
27
+ "completed": 39,
28
+ "pending": 3,
29
+ "successes": 31,
30
+ "valid_results": 39,
31
+ "success_rate": 0.7948717948717948,
32
+ "input_tokens": 229797708,
33
+ "cached_input_tokens": 225752320,
34
+ "output_tokens": 869279,
35
+ "usage_complete": 39,
36
  "control_frequency_hz": 25,
37
  "max_control_steps": 7500,
38
  "preflight_results": 0,
39
+ "formal_results": 39,
40
+ "modified_results": 31,
41
  "original_results": 8
42
  }
43
  ],
44
  "progress": {
45
+ "finished": 39,
46
  "interrupted": 2,
47
  "native_failures": 8,
48
+ "native_successes": 31,
49
  "needs_review": 0,
50
  "queued": 0,
51
+ "running": 1
52
  },
53
  "interrupted_attempts": 4,
54
  "usage_incomplete_tasks": [],
 
59
  "id": "task04",
60
  "name": "RoboDojo",
61
  "total": 42,
62
+ "completed": 39,
63
+ "pending": 3,
64
+ "successes": 31,
65
+ "valid_results": 39,
66
+ "success_rate": 0.7948717948717948,
67
+ "input_tokens": 229797708,
68
+ "cached_input_tokens": 225752320,
69
+ "output_tokens": 869279,
70
+ "usage_complete": 39,
71
  "control_frequency_hz": 25,
72
  "max_control_steps": 7500,
73
  "preflight_results": 0,
74
+ "formal_results": 39,
75
+ "modified_results": 31,
76
  "original_results": 8
77
  }
78
  ],
 
670
  "catalog_instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
671
  "native_instruction": "Insert the three tubes into the rack one by one.",
672
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
673
+ "status": "completed",
674
+ "episode_id": "task04-07-seed0-formal",
675
  "planned_protocol": {
676
  "episodes": 1,
677
  "seed": 0,
 
727
  },
728
  "display_slot": "07",
729
  "display_key": "task04/07",
730
+ "run_status": "finished",
731
+ "status_note": "Queued for evaluation.",
732
  "attempt_history": [
733
  {
734
  "id": "07-robodojo-insert-tubes-codex-seed0-attempt01",
 
5123
  "selected_for_formal_metrics": true,
5124
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/06/seed-0/"
5125
  },
5126
+ {
5127
+ "id": "task04-07-seed0-formal",
5128
+ "task_key": "task04/07",
5129
+ "family": "task04",
5130
+ "slot": "07",
5131
+ "seed": 0,
5132
+ "episode": 1,
5133
+ "phase": "formal",
5134
+ "status": "completed",
5135
+ "success": true,
5136
+ "native_reward": 1.0,
5137
+ "valid": true,
5138
+ "execution": {
5139
+ "reason": null,
5140
+ "status": "finished"
5141
+ },
5142
+ "verdict": {
5143
+ "evidence_valid": true,
5144
+ "steps": 4162,
5145
+ "success": true,
5146
+ "termination": "success"
5147
+ },
5148
+ "steps": 4162,
5149
+ "simulation_time_s": null,
5150
+ "wall_time_s": 2428.860172,
5151
+ "model": "gpt-6-astra",
5152
+ "effort": "high",
5153
+ "harness": "codex",
5154
+ "codex_version": "0.160.0",
5155
+ "native_instruction": "Insert the three tubes into the rack one by one.",
5156
+ "instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
5157
+ "instruction_policy": "modified",
5158
+ "usage": {
5159
+ "accounting": "reported-responses",
5160
+ "audit_complete": true,
5161
+ "cache_hit_rate": 0.9885433078393561,
5162
+ "cache_reported_input_tokens": 11924908,
5163
+ "cache_write_input_tokens": 0,
5164
+ "cache_write_reported_input_tokens": 11924908,
5165
+ "cached_input_tokens": 11788288,
5166
+ "completed_turns": 1,
5167
+ "cost_usd": null,
5168
+ "failed_turns": 0,
5169
+ "input_tokens": 11924908,
5170
+ "known_cache_write_input_tokens": 0,
5171
+ "known_cached_input_tokens": 11788288,
5172
+ "known_input_tokens": 11924908,
5173
+ "known_output_tokens": 45245,
5174
+ "known_reasoning_output_tokens": 23793,
5175
+ "output_tokens": 45245,
5176
+ "reasoning_output_tokens": 23793,
5177
+ "reasoning_reported_output_tokens": 45245,
5178
+ "reported_responses": {
5179
+ "cache_reported_input_tokens": 168,
5180
+ "cache_write_input_tokens": 168,
5181
+ "cache_write_reported_input_tokens": 168,
5182
+ "cached_input_tokens": 168,
5183
+ "input_tokens": 168,
5184
+ "output_tokens": 168,
5185
+ "reasoning_output_tokens": 168,
5186
+ "reasoning_reported_output_tokens": 168
5187
+ },
5188
+ "response_count": 168,
5189
+ "response_ids_complete": true,
5190
+ "schema": "rlebench/token-usage/1",
5191
+ "source": "Codex token_usage_record per response",
5192
+ "uncached_input_tokens": 136620,
5193
+ "unidentified_usage_records": 0
5194
+ },
5195
+ "call_activity": {
5196
+ "model_tool_calls": 167,
5197
+ "model_tool_calls_by_name": {
5198
+ "exec": 167
5199
+ },
5200
+ "nested_python_tool_invocations": null,
5201
+ "python_device_rpc_attempts": null,
5202
+ "python_device_rpc_attempts_by_action": null,
5203
+ "python_device_rpc_errors": null,
5204
+ "python_instrumented_model_tool_calls": null,
5205
+ "python_tool_invocations": null,
5206
+ "python_tool_invocations_by_origin": null,
5207
+ "schema": "rlebench/call-activity/1",
5208
+ "source": "Codex native sessions"
5209
+ },
5210
+ "media": {
5211
+ "passed": true,
5212
+ "width": 2880,
5213
+ "height": 720,
5214
+ "duration_s": 41.6,
5215
+ "speed": 4,
5216
+ "source_fps": 10,
5217
+ "output_fps": 20,
5218
+ "recording": {
5219
+ "accepted_samples": 1666,
5220
+ "captured_samples": 1666,
5221
+ "clock": "simulation",
5222
+ "dropped_samples": 0,
5223
+ "encoded_frames": 1665,
5224
+ "end_time_s": 166.48000000000116,
5225
+ "error": null,
5226
+ "experimental": true,
5227
+ "fps": 10,
5228
+ "received_samples": 1666,
5229
+ "schema": "roboenv/recording/1",
5230
+ "state": "closed",
5231
+ "status": "complete",
5232
+ "views": [
5233
+ {
5234
+ "fov_y": 45.0,
5235
+ "height": 720,
5236
+ "name": "third_person",
5237
+ "pose": null,
5238
+ "source": "third_person",
5239
+ "width": 960
5240
+ },
5241
+ {
5242
+ "fov_y": 45.0,
5243
+ "height": 720,
5244
+ "name": "left_wrist",
5245
+ "pose": null,
5246
+ "source": "left_wrist",
5247
+ "width": 960
5248
+ },
5249
+ {
5250
+ "fov_y": 45.0,
5251
+ "height": 720,
5252
+ "name": "right_wrist",
5253
+ "pose": null,
5254
+ "source": "right_wrist",
5255
+ "width": 960
5256
+ }
5257
+ ]
5258
+ },
5259
+ "view_names": [
5260
+ "third_person",
5261
+ "left_wrist",
5262
+ "right_wrist"
5263
+ ],
5264
+ "sha256": "0eceba672ce4ecdeea98aa372cce1c4d7dd69cf118c2766b39ad4fee0204a3fb",
5265
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
5266
+ },
5267
+ "analysis": {
5268
+ "outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
5269
+ "duration": "\u4f7f\u7528 4,162 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
5270
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
5271
+ },
5272
+ "provenance": {
5273
+ "sources": {
5274
+ "RLE-Bench-inhouse": {
5275
+ "build_inputs": [
5276
+ "pyproject.toml",
5277
+ "src",
5278
+ "tasks",
5279
+ "README.md",
5280
+ "Makefile",
5281
+ "tests",
5282
+ "docs"
5283
+ ],
5284
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
5285
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
5286
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
5287
+ },
5288
+ "RoboEnv": {
5289
+ "build_inputs": [
5290
+ "pyproject.toml",
5291
+ "README.md",
5292
+ "src",
5293
+ "runtime/pyproject.toml",
5294
+ "runtime/README.md",
5295
+ "runtime/src",
5296
+ "runtime/environments.json",
5297
+ "runtime/locks",
5298
+ "catalog",
5299
+ "upstreams.lock.json",
5300
+ "third_party/patches",
5301
+ "docs/validation"
5302
+ ],
5303
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
5304
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
5305
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
5306
+ }
5307
+ },
5308
+ "job": "07-robodojo-insert-tubes-codex-seed0-attempt02",
5309
+ "attempt": 2,
5310
+ "harness": "stock Codex CLI",
5311
+ "codex_version": "0.160.0",
5312
+ "model": "gpt-6-astra",
5313
+ "effort": "high",
5314
+ "service_tier": "default",
5315
+ "fresh_session": true,
5316
+ "source_jobs": [],
5317
+ "resume_trajectory": false,
5318
+ "imported_skills": [],
5319
+ "automatic_harbor_retries": 0,
5320
+ "request_policy": {
5321
+ "max_request_retries": 50,
5322
+ "configuration": "explicit retry50 SSE",
5323
+ "usage_accounting": "reported-responses"
5324
+ },
5325
+ "classification": "formal",
5326
+ "measured_images": {
5327
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
5328
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
5329
+ },
5330
+ "session_original_sha256": "dced20541deac3be24265775d5b357aadecb6b53ad8b991fbf38cfcb1a690cfd",
5331
+ "protocol_sha256": "331cfc33f7aa656ce04ffa0d8e7927d625ea36dd825616041760649c239cee4b"
5332
+ },
5333
+ "links": {
5334
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/session.jsonl",
5335
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/trajectory.json",
5336
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/provider-usage.jsonl",
5337
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/transcript.json",
5338
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/episode.json",
5339
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/protocol.json",
5340
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/instructions.json",
5341
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/workspace.tar.gz",
5342
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/workspace.json",
5343
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
5344
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/evidence/episode/evidence/recording/manifest.json",
5345
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/usage.json",
5346
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/analysis.json",
5347
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/provenance.json",
5348
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/video.mp4",
5349
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/poster.jpg",
5350
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/media-validation.json"
5351
+ },
5352
+ "resources": [
5353
+ {
5354
+ "name": "tools/robot.py",
5355
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/tools/robot.py",
5356
+ "kind": "Created during this episode; final workspace snapshot."
5357
+ },
5358
+ {
5359
+ "name": "tools/tubes.py",
5360
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/tools/tubes.py",
5361
+ "kind": "Created during this episode; final workspace snapshot."
5362
+ },
5363
+ {
5364
+ "name": "memos/insert-tubes.md",
5365
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/memos/insert-tubes.md",
5366
+ "kind": "Created during this episode; final workspace snapshot."
5367
+ }
5368
+ ],
5369
+ "session_counts": {
5370
+ "visible_events": 362,
5371
+ "observed_images": 84,
5372
+ "tool_errors": 4
5373
+ },
5374
+ "selected_for_formal_metrics": true,
5375
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/"
5376
+ },
5377
  {
5378
  "id": "task04-08-seed0-formal",
5379
  "task_key": "task04/08",
episodes.csv CHANGED
@@ -5,7 +5,7 @@ task04/03,robodojo/store-laptop-and-headphones,completed,finished,False,5696,181
5
  task04/04,robodojo/cover-blocks,completed,finished,True,2184,2163308,2107392,9584
6
  task04/05,robodojo/play-xylophone,completed,finished,True,818,1397747,1321344,9368
7
  task04/06,robodojo/store-tools-in-toolbox,completed,finished,False,7421,15593052,15416704,57648
8
- task04/07,robodojo/insert-tubes,pending,running,,,,,
9
  task04/08,robodojo/deposit-coin,completed,finished,True,1432,2957393,2867840,15930
10
  task04/09,robodojo/fasten-screws,completed,finished,True,7340,5986459,5888000,23452
11
  task04/10,robodojo/play-stacking-toy,completed,finished,True,2460,4468354,4390912,20623
 
5
  task04/04,robodojo/cover-blocks,completed,finished,True,2184,2163308,2107392,9584
6
  task04/05,robodojo/play-xylophone,completed,finished,True,818,1397747,1321344,9368
7
  task04/06,robodojo/store-tools-in-toolbox,completed,finished,False,7421,15593052,15416704,57648
8
+ task04/07,robodojo/insert-tubes,completed,finished,True,4162,11924908,11788288,45245
9
  task04/08,robodojo/deposit-coin,completed,finished,True,1432,2957393,2867840,15930
10
  task04/09,robodojo/fasten-screws,completed,finished,True,7340,5986459,5888000,23452
11
  task04/10,robodojo/play-stacking-toy,completed,finished,True,2460,4468354,4390912,20623
episodes.json CHANGED
@@ -1239,6 +1239,257 @@
1239
  "selected_for_formal_metrics": true,
1240
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/06/seed-0/"
1241
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1242
  {
1243
  "id": "task04-08-seed0-formal",
1244
  "task_key": "task04/08",
 
1239
  "selected_for_formal_metrics": true,
1240
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/06/seed-0/"
1241
  },
1242
+ {
1243
+ "id": "task04-07-seed0-formal",
1244
+ "task_key": "task04/07",
1245
+ "family": "task04",
1246
+ "slot": "07",
1247
+ "seed": 0,
1248
+ "episode": 1,
1249
+ "phase": "formal",
1250
+ "status": "completed",
1251
+ "success": true,
1252
+ "native_reward": 1.0,
1253
+ "valid": true,
1254
+ "execution": {
1255
+ "reason": null,
1256
+ "status": "finished"
1257
+ },
1258
+ "verdict": {
1259
+ "evidence_valid": true,
1260
+ "steps": 4162,
1261
+ "success": true,
1262
+ "termination": "success"
1263
+ },
1264
+ "steps": 4162,
1265
+ "simulation_time_s": null,
1266
+ "wall_time_s": 2428.860172,
1267
+ "model": "gpt-6-astra",
1268
+ "effort": "high",
1269
+ "harness": "codex",
1270
+ "codex_version": "0.160.0",
1271
+ "native_instruction": "Insert the three tubes into the rack one by one.",
1272
+ "instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
1273
+ "instruction_policy": "modified",
1274
+ "usage": {
1275
+ "accounting": "reported-responses",
1276
+ "audit_complete": true,
1277
+ "cache_hit_rate": 0.9885433078393561,
1278
+ "cache_reported_input_tokens": 11924908,
1279
+ "cache_write_input_tokens": 0,
1280
+ "cache_write_reported_input_tokens": 11924908,
1281
+ "cached_input_tokens": 11788288,
1282
+ "completed_turns": 1,
1283
+ "cost_usd": null,
1284
+ "failed_turns": 0,
1285
+ "input_tokens": 11924908,
1286
+ "known_cache_write_input_tokens": 0,
1287
+ "known_cached_input_tokens": 11788288,
1288
+ "known_input_tokens": 11924908,
1289
+ "known_output_tokens": 45245,
1290
+ "known_reasoning_output_tokens": 23793,
1291
+ "output_tokens": 45245,
1292
+ "reasoning_output_tokens": 23793,
1293
+ "reasoning_reported_output_tokens": 45245,
1294
+ "reported_responses": {
1295
+ "cache_reported_input_tokens": 168,
1296
+ "cache_write_input_tokens": 168,
1297
+ "cache_write_reported_input_tokens": 168,
1298
+ "cached_input_tokens": 168,
1299
+ "input_tokens": 168,
1300
+ "output_tokens": 168,
1301
+ "reasoning_output_tokens": 168,
1302
+ "reasoning_reported_output_tokens": 168
1303
+ },
1304
+ "response_count": 168,
1305
+ "response_ids_complete": true,
1306
+ "schema": "rlebench/token-usage/1",
1307
+ "source": "Codex token_usage_record per response",
1308
+ "uncached_input_tokens": 136620,
1309
+ "unidentified_usage_records": 0
1310
+ },
1311
+ "call_activity": {
1312
+ "model_tool_calls": 167,
1313
+ "model_tool_calls_by_name": {
1314
+ "exec": 167
1315
+ },
1316
+ "nested_python_tool_invocations": null,
1317
+ "python_device_rpc_attempts": null,
1318
+ "python_device_rpc_attempts_by_action": null,
1319
+ "python_device_rpc_errors": null,
1320
+ "python_instrumented_model_tool_calls": null,
1321
+ "python_tool_invocations": null,
1322
+ "python_tool_invocations_by_origin": null,
1323
+ "schema": "rlebench/call-activity/1",
1324
+ "source": "Codex native sessions"
1325
+ },
1326
+ "media": {
1327
+ "passed": true,
1328
+ "width": 2880,
1329
+ "height": 720,
1330
+ "duration_s": 41.6,
1331
+ "speed": 4,
1332
+ "source_fps": 10,
1333
+ "output_fps": 20,
1334
+ "recording": {
1335
+ "accepted_samples": 1666,
1336
+ "captured_samples": 1666,
1337
+ "clock": "simulation",
1338
+ "dropped_samples": 0,
1339
+ "encoded_frames": 1665,
1340
+ "end_time_s": 166.48000000000116,
1341
+ "error": null,
1342
+ "experimental": true,
1343
+ "fps": 10,
1344
+ "received_samples": 1666,
1345
+ "schema": "roboenv/recording/1",
1346
+ "state": "closed",
1347
+ "status": "complete",
1348
+ "views": [
1349
+ {
1350
+ "fov_y": 45.0,
1351
+ "height": 720,
1352
+ "name": "third_person",
1353
+ "pose": null,
1354
+ "source": "third_person",
1355
+ "width": 960
1356
+ },
1357
+ {
1358
+ "fov_y": 45.0,
1359
+ "height": 720,
1360
+ "name": "left_wrist",
1361
+ "pose": null,
1362
+ "source": "left_wrist",
1363
+ "width": 960
1364
+ },
1365
+ {
1366
+ "fov_y": 45.0,
1367
+ "height": 720,
1368
+ "name": "right_wrist",
1369
+ "pose": null,
1370
+ "source": "right_wrist",
1371
+ "width": 960
1372
+ }
1373
+ ]
1374
+ },
1375
+ "view_names": [
1376
+ "third_person",
1377
+ "left_wrist",
1378
+ "right_wrist"
1379
+ ],
1380
+ "sha256": "0eceba672ce4ecdeea98aa372cce1c4d7dd69cf118c2766b39ad4fee0204a3fb",
1381
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
1382
+ },
1383
+ "analysis": {
1384
+ "outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
1385
+ "duration": "\u4f7f\u7528 4,162 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
1386
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
1387
+ },
1388
+ "provenance": {
1389
+ "sources": {
1390
+ "RLE-Bench-inhouse": {
1391
+ "build_inputs": [
1392
+ "pyproject.toml",
1393
+ "src",
1394
+ "tasks",
1395
+ "README.md",
1396
+ "Makefile",
1397
+ "tests",
1398
+ "docs"
1399
+ ],
1400
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
1401
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
1402
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
1403
+ },
1404
+ "RoboEnv": {
1405
+ "build_inputs": [
1406
+ "pyproject.toml",
1407
+ "README.md",
1408
+ "src",
1409
+ "runtime/pyproject.toml",
1410
+ "runtime/README.md",
1411
+ "runtime/src",
1412
+ "runtime/environments.json",
1413
+ "runtime/locks",
1414
+ "catalog",
1415
+ "upstreams.lock.json",
1416
+ "third_party/patches",
1417
+ "docs/validation"
1418
+ ],
1419
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
1420
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
1421
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
1422
+ }
1423
+ },
1424
+ "job": "07-robodojo-insert-tubes-codex-seed0-attempt02",
1425
+ "attempt": 2,
1426
+ "harness": "stock Codex CLI",
1427
+ "codex_version": "0.160.0",
1428
+ "model": "gpt-6-astra",
1429
+ "effort": "high",
1430
+ "service_tier": "default",
1431
+ "fresh_session": true,
1432
+ "source_jobs": [],
1433
+ "resume_trajectory": false,
1434
+ "imported_skills": [],
1435
+ "automatic_harbor_retries": 0,
1436
+ "request_policy": {
1437
+ "max_request_retries": 50,
1438
+ "configuration": "explicit retry50 SSE",
1439
+ "usage_accounting": "reported-responses"
1440
+ },
1441
+ "classification": "formal",
1442
+ "measured_images": {
1443
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
1444
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
1445
+ },
1446
+ "session_original_sha256": "dced20541deac3be24265775d5b357aadecb6b53ad8b991fbf38cfcb1a690cfd",
1447
+ "protocol_sha256": "331cfc33f7aa656ce04ffa0d8e7927d625ea36dd825616041760649c239cee4b"
1448
+ },
1449
+ "links": {
1450
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/session.jsonl",
1451
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/trajectory.json",
1452
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/agent/provider-usage.jsonl",
1453
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/transcript.json",
1454
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/episode.json",
1455
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/protocol.json",
1456
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/instructions.json",
1457
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/workspace.tar.gz",
1458
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/verifier/workspace.json",
1459
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
1460
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/native/evidence/episode/evidence/recording/manifest.json",
1461
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/usage.json",
1462
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/analysis.json",
1463
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/provenance.json",
1464
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/video.mp4",
1465
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/poster.jpg",
1466
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/media-validation.json"
1467
+ },
1468
+ "resources": [
1469
+ {
1470
+ "name": "tools/robot.py",
1471
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/tools/robot.py",
1472
+ "kind": "Created during this episode; final workspace snapshot."
1473
+ },
1474
+ {
1475
+ "name": "tools/tubes.py",
1476
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/tools/tubes.py",
1477
+ "kind": "Created during this episode; final workspace snapshot."
1478
+ },
1479
+ {
1480
+ "name": "memos/insert-tubes.md",
1481
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/resources/memos/insert-tubes.md",
1482
+ "kind": "Created during this episode; final workspace snapshot."
1483
+ }
1484
+ ],
1485
+ "session_counts": {
1486
+ "visible_events": 362,
1487
+ "observed_images": 84,
1488
+ "tool_errors": 4
1489
+ },
1490
+ "selected_for_formal_metrics": true,
1491
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/07/seed-0/"
1492
+ },
1493
  {
1494
  "id": "task04-08-seed0-formal",
1495
  "task_key": "task04/08",
manifest.json CHANGED
@@ -1,21 +1,21 @@
1
  {
2
  ".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
3
  "DATA_FORMAT.md": "90cde3a7afcec6312f6705d47a959bd2135f6ed599b1ef70944eeca209b96778",
4
- "README.md": "9f9b52831cb17eee2c8d3fc19c09fdc291cc29a0f88ebf0e98be6cdf87906299",
5
- "REPORT.md": "3a2a55ada3d01e0d1d92d29eb45e70391621fe9d7e47578fc1071b9ff1aca081",
6
  "THIRD_PARTY_NOTICES.md": "b90a0d2dc93b1686caed55e3c68f85a1c761cb95a4935c692a223ab7e19c4c28",
7
  "app.js": "3ef085aa517e167f8020eaef3ef26bbf4ca71f759e332c2358ddf3411c25c771",
8
  "attempt-history.json": "8555f458c1dd362f5a07447342994fddf265ff23bbc40e9bd6f9984a892ddb4e",
9
- "comparison.json": "6e915a569c3187e324cb03fb4912a2d972be2a24ad720095284e815fa44ff9bf",
10
- "data.json": "30b47c87895e06745a757ad7ccc083ae1aeb4f4ae2556854d11beaae64031fc8",
11
- "episodes.csv": "f180e3248f4d0b548426106e44f59a205e479dc64e8629e85b650e8c3b8a7cdf",
12
- "episodes.json": "238bbb5714c852c0f293566e303b265a464270da84e15a962b6b180a2cfdd7cb",
13
  "failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
14
  "favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
15
  "index.html": "da4a91fd226b8792395edbd3d2badda0e1fd69913d9c27480c12bcbdbbc73c37",
16
  "licenses/RoboDojo-LICENSE.txt": "7794bb06af8fe5485ca912454ad1665ccd7846c45f9c2b1258658e480948cdbe",
17
- "protocol.json": "6786af8fd7094f19c2c32cddc5afbeb833bd514f3b58a9c3ed1c3fd75ee71c38",
18
  "style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
19
- "summary.json": "2805cc2785b9503cb7e17eb2990d27cdbb182eda692cce5ef39f37be1f681374",
20
- "task-index.json": "be82a40de7884a230439b35e9f7e0aa68989b6ddf1be1a90cdb74d87b0f8fd92"
21
  }
 
1
  {
2
  ".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
3
  "DATA_FORMAT.md": "90cde3a7afcec6312f6705d47a959bd2135f6ed599b1ef70944eeca209b96778",
4
+ "README.md": "e962b0773cc1cddd293ecd3182f6f36161e2396c7652751f19efe08fa992c1f9",
5
+ "REPORT.md": "1d1faa33c849daac0560cd4ede19c59d5fe1f7cd04e7a404f43714a5215006fe",
6
  "THIRD_PARTY_NOTICES.md": "b90a0d2dc93b1686caed55e3c68f85a1c761cb95a4935c692a223ab7e19c4c28",
7
  "app.js": "3ef085aa517e167f8020eaef3ef26bbf4ca71f759e332c2358ddf3411c25c771",
8
  "attempt-history.json": "8555f458c1dd362f5a07447342994fddf265ff23bbc40e9bd6f9984a892ddb4e",
9
+ "comparison.json": "1d92156de449708d2f190642a7247c94b80dab37915174dba8f908a832621c98",
10
+ "data.json": "a8433e92fc2ca0ea0ab8debf5f697a0ee53e3bacb1968573e4b6b9a03b5f3731",
11
+ "episodes.csv": "887754b432854d3e2c05538688ef2e16944fdf08facb001e2591052ae5e35d4f",
12
+ "episodes.json": "f09de90724e941a862a8f716c57d049123a5f0a96252732bcb697feb603ad497",
13
  "failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
14
  "favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
15
  "index.html": "da4a91fd226b8792395edbd3d2badda0e1fd69913d9c27480c12bcbdbbc73c37",
16
  "licenses/RoboDojo-LICENSE.txt": "7794bb06af8fe5485ca912454ad1665ccd7846c45f9c2b1258658e480948cdbe",
17
+ "protocol.json": "764850bb861719efbc63bf62ca4b2f2ce8e34797d6fb2d16aec94c1d3dc74894",
18
  "style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
19
+ "summary.json": "c784cb052580571e67a596785372ac0f27cb924b19c41eb9d2e235b06f335f13",
20
+ "task-index.json": "db238d1e28b1bf49e140a4bbd190962b20960bc3eae90ca0ac538f9ef3038e96"
21
  }
protocol.json CHANGED
@@ -16,20 +16,20 @@
16
  "id": "task04",
17
  "name": "RoboDojo",
18
  "total": 42,
19
- "completed": 38,
20
- "pending": 4,
21
- "successes": 30,
22
- "valid_results": 38,
23
- "success_rate": 0.7894736842105263,
24
- "input_tokens": 217872800,
25
- "cached_input_tokens": 213964032,
26
- "output_tokens": 824034,
27
- "usage_complete": 38,
28
  "control_frequency_hz": 25,
29
  "max_control_steps": 7500,
30
  "preflight_results": 0,
31
- "formal_results": 38,
32
- "modified_results": 30,
33
  "original_results": 8
34
  }
35
  ],
 
16
  "id": "task04",
17
  "name": "RoboDojo",
18
  "total": 42,
19
+ "completed": 39,
20
+ "pending": 3,
21
+ "successes": 31,
22
+ "valid_results": 39,
23
+ "success_rate": 0.7948717948717948,
24
+ "input_tokens": 229797708,
25
+ "cached_input_tokens": 225752320,
26
+ "output_tokens": 869279,
27
+ "usage_complete": 39,
28
  "control_frequency_hz": 25,
29
  "max_control_steps": 7500,
30
  "preflight_results": 0,
31
+ "formal_results": 39,
32
+ "modified_results": 31,
33
  "original_results": 8
34
  }
35
  ],
publication-manifest.json CHANGED
@@ -10,11 +10,11 @@
10
  "bytes": 1129
11
  },
12
  "README.md": {
13
- "sha256": "9f9b52831cb17eee2c8d3fc19c09fdc291cc29a0f88ebf0e98be6cdf87906299",
14
  "bytes": 1179
15
  },
16
  "REPORT.md": {
17
- "sha256": "3a2a55ada3d01e0d1d92d29eb45e70391621fe9d7e47578fc1071b9ff1aca081",
18
  "bytes": 1056
19
  },
20
  "THIRD_PARTY_NOTICES.md": {
@@ -30,20 +30,20 @@
30
  "bytes": 36661
31
  },
32
  "comparison.json": {
33
- "sha256": "6e915a569c3187e324cb03fb4912a2d972be2a24ad720095284e815fa44ff9bf",
34
- "bytes": 9459
35
  },
36
  "data.json": {
37
- "sha256": "30b47c87895e06745a757ad7ccc083ae1aeb4f4ae2556854d11beaae64031fc8",
38
- "bytes": 566929
39
  },
40
  "episodes.csv": {
41
- "sha256": "f180e3248f4d0b548426106e44f59a205e479dc64e8629e85b650e8c3b8a7cdf",
42
- "bytes": 3675
43
  },
44
  "episodes.json": {
45
- "sha256": "238bbb5714c852c0f293566e303b265a464270da84e15a962b6b180a2cfdd7cb",
46
- "bytes": 397992
47
  },
48
  "failure-reviews.json": {
49
  "sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
@@ -62,7 +62,7 @@
62
  "bytes": 1091
63
  },
64
  "protocol.json": {
65
- "sha256": "6786af8fd7094f19c2c32cddc5afbeb833bd514f3b58a9c3ed1c3fd75ee71c38",
66
  "bytes": 1083
67
  },
68
  "style.css": {
@@ -70,15 +70,15 @@
70
  "bytes": 16560
71
  },
72
  "summary.json": {
73
- "sha256": "2805cc2785b9503cb7e17eb2990d27cdbb182eda692cce5ef39f37be1f681374",
74
  "bytes": 896
75
  },
76
  "task-index.json": {
77
- "sha256": "be82a40de7884a230439b35e9f7e0aa68989b6ddf1be1a90cdb74d87b0f8fd92",
78
- "bytes": 140006
79
  },
80
  "manifest.json": {
81
- "sha256": "beb76a718594026688d4c7991f73c9f0396de426ec244a3c08cd1e15ce13fead",
82
  "bytes": 1671
83
  }
84
  }
 
10
  "bytes": 1129
11
  },
12
  "README.md": {
13
+ "sha256": "e962b0773cc1cddd293ecd3182f6f36161e2396c7652751f19efe08fa992c1f9",
14
  "bytes": 1179
15
  },
16
  "REPORT.md": {
17
+ "sha256": "1d1faa33c849daac0560cd4ede19c59d5fe1f7cd04e7a404f43714a5215006fe",
18
  "bytes": 1056
19
  },
20
  "THIRD_PARTY_NOTICES.md": {
 
30
  "bytes": 36661
31
  },
32
  "comparison.json": {
33
+ "sha256": "1d92156de449708d2f190642a7247c94b80dab37915174dba8f908a832621c98",
34
+ "bytes": 9461
35
  },
36
  "data.json": {
37
+ "sha256": "a8433e92fc2ca0ea0ab8debf5f697a0ee53e3bacb1968573e4b6b9a03b5f3731",
38
+ "bytes": 577958
39
  },
40
  "episodes.csv": {
41
+ "sha256": "887754b432854d3e2c05538688ef2e16944fdf08facb001e2591052ae5e35d4f",
42
+ "bytes": 3707
43
  },
44
  "episodes.json": {
45
+ "sha256": "f09de90724e941a862a8f716c57d049123a5f0a96252732bcb697feb603ad497",
46
+ "bytes": 408492
47
  },
48
  "failure-reviews.json": {
49
  "sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
 
62
  "bytes": 1091
63
  },
64
  "protocol.json": {
65
+ "sha256": "764850bb861719efbc63bf62ca4b2f2ce8e34797d6fb2d16aec94c1d3dc74894",
66
  "bytes": 1083
67
  },
68
  "style.css": {
 
70
  "bytes": 16560
71
  },
72
  "summary.json": {
73
+ "sha256": "c784cb052580571e67a596785372ac0f27cb924b19c41eb9d2e235b06f335f13",
74
  "bytes": 896
75
  },
76
  "task-index.json": {
77
+ "sha256": "db238d1e28b1bf49e140a4bbd190962b20960bc3eae90ca0ac538f9ef3038e96",
78
+ "bytes": 140033
79
  },
80
  "manifest.json": {
81
+ "sha256": "b1f2ebc9cc75fc4375931ff9df5642fd800d0ec972f214e37af5284bcb4a8095",
82
  "bytes": 1671
83
  }
84
  }
summary.json CHANGED
@@ -1,37 +1,37 @@
1
  {
2
  "planned_tasks": 42,
3
- "published_results": 38,
4
- "pending_tasks": 4,
5
  "families": [
6
  {
7
  "id": "task04",
8
  "name": "RoboDojo",
9
  "total": 42,
10
- "completed": 38,
11
- "pending": 4,
12
- "successes": 30,
13
- "valid_results": 38,
14
- "success_rate": 0.7894736842105263,
15
- "input_tokens": 217872800,
16
- "cached_input_tokens": 213964032,
17
- "output_tokens": 824034,
18
- "usage_complete": 38,
19
  "control_frequency_hz": 25,
20
  "max_control_steps": 7500,
21
  "preflight_results": 0,
22
- "formal_results": 38,
23
- "modified_results": 30,
24
  "original_results": 8
25
  }
26
  ],
27
  "progress": {
28
- "finished": 38,
29
  "interrupted": 2,
30
  "native_failures": 8,
31
- "native_successes": 30,
32
  "needs_review": 0,
33
  "queued": 0,
34
- "running": 2
35
  },
36
  "interrupted_attempts": 4,
37
  "usage_incomplete_tasks": [],
 
1
  {
2
  "planned_tasks": 42,
3
+ "published_results": 39,
4
+ "pending_tasks": 3,
5
  "families": [
6
  {
7
  "id": "task04",
8
  "name": "RoboDojo",
9
  "total": 42,
10
+ "completed": 39,
11
+ "pending": 3,
12
+ "successes": 31,
13
+ "valid_results": 39,
14
+ "success_rate": 0.7948717948717948,
15
+ "input_tokens": 229797708,
16
+ "cached_input_tokens": 225752320,
17
+ "output_tokens": 869279,
18
+ "usage_complete": 39,
19
  "control_frequency_hz": 25,
20
  "max_control_steps": 7500,
21
  "preflight_results": 0,
22
+ "formal_results": 39,
23
+ "modified_results": 31,
24
  "original_results": 8
25
  }
26
  ],
27
  "progress": {
28
+ "finished": 39,
29
  "interrupted": 2,
30
  "native_failures": 8,
31
+ "native_successes": 31,
32
  "needs_review": 0,
33
  "queued": 0,
34
+ "running": 1
35
  },
36
  "interrupted_attempts": 4,
37
  "usage_incomplete_tasks": [],
task-index.json CHANGED
@@ -592,8 +592,8 @@
592
  "catalog_instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
593
  "native_instruction": "Insert the three tubes into the rack one by one.",
594
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
595
- "status": "pending",
596
- "episode_id": null,
597
  "planned_protocol": {
598
  "episodes": 1,
599
  "seed": 0,
@@ -649,8 +649,8 @@
649
  },
650
  "display_slot": "07",
651
  "display_key": "task04/07",
652
- "run_status": "running",
653
- "status_note": "Currently running.",
654
  "attempt_history": [
655
  {
656
  "id": "07-robodojo-insert-tubes-codex-seed0-attempt01",
 
592
  "catalog_instruction": "Insert the three tubes upright into the rack one by one, pointed ends down, until they are fully seated. Release them and return both arms to their starting poses.",
593
  "native_instruction": "Insert the three tubes into the rack one by one.",
594
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
595
+ "status": "completed",
596
+ "episode_id": "task04-07-seed0-formal",
597
  "planned_protocol": {
598
  "episodes": 1,
599
  "seed": 0,
 
649
  },
650
  "display_slot": "07",
651
  "display_key": "task04/07",
652
+ "run_status": "finished",
653
+ "status_note": "Queued for evaluation.",
654
  "attempt_history": [
655
  {
656
  "id": "07-robodojo-insert-tubes-codex-seed0-attempt01",