yzzhao commited on
Commit
3485569
·
verified ·
1 Parent(s): e0414df

Publish Codex Benchmark evaluation evidence

Browse files
Files changed (12) hide show
  1. README.md +1 -1
  2. REPORT.md +1 -1
  3. attempt-history.json +30 -30
  4. comparison.json +16 -16
  5. data.json +1563 -74
  6. episodes.csv +12 -12
  7. episodes.json +1561 -72
  8. manifest.json +10 -10
  9. protocol.json +12 -12
  10. publication-manifest.json +18 -18
  11. summary.json +18 -18
  12. task-index.json +63 -63
README.md CHANGED
@@ -10,7 +10,7 @@ pinned: false
10
 
11
  # Codex Benchmark
12
 
13
- 23/42 ordinary RoboDojo tasks published: 18 successes, 5 valid native failures, 19 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
14
 
15
  Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
16
 
 
10
 
11
  # Codex Benchmark
12
 
13
+ 29/42 ordinary RoboDojo tasks published: 21 successes, 8 valid native failures, 13 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
14
 
15
  Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
16
 
REPORT.md CHANGED
@@ -1,6 +1,6 @@
1
  # Codex Benchmark
2
 
3
- 23/42 ordinary RoboDojo tasks published: 18 successes, 5 valid native failures, 19 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
4
 
5
  Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
6
 
 
1
  # Codex Benchmark
2
 
3
+ 29/42 ordinary RoboDojo tasks published: 21 successes, 8 valid native failures, 13 pending. Pending and interrupted attempts are excluded from the success-rate denominator.
4
 
5
  Stock Codex CLI 0.160.0, GPT-6 Astra / high / default, seed 0, fresh independent workspace, 7,500 action frames, stepped simulation, 8-hour safety limit. Instructions match the final reviewed benchmark definitions.
6
 
attempt-history.json CHANGED
@@ -153,25 +153,25 @@
153
  "protocol_sha256": "558b51d6786cd5ba0a989f7b2b872a3e37424dda668128a679113df86f67e43d"
154
  },
155
  "links": {
156
- "native_session": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/session.jsonl",
157
- "trace": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/trajectory.json",
158
- "provider_usage": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/provider-usage.jsonl",
159
- "transcript": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/transcript.json",
160
- "verdict": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/episode.json",
161
- "protocol": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/protocol.json",
162
- "native_goal": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/instructions.json",
163
- "workspace": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/workspace.tar.gz",
164
- "workspace_changes": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/workspace.json",
165
- "owner_journal": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/evidence/episode/evidence/journal/actions.jsonl",
166
- "recording_manifest": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/evidence/episode/evidence/recording/manifest.json",
167
- "usage": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/usage.json",
168
- "analysis": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/analysis.json",
169
- "provenance": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/provenance.json"
170
  },
171
  "resources": [
172
  {
173
  "name": "tools/robot.py",
174
- "file": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/resources/tools/robot.py",
175
  "kind": "Created during this episode; final workspace snapshot."
176
  }
177
  ],
@@ -334,25 +334,25 @@
334
  "protocol_sha256": "331cfc33f7aa656ce04ffa0d8e7927d625ea36dd825616041760649c239cee4b"
335
  },
336
  "links": {
337
- "native_session": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/session.jsonl",
338
- "trace": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/trajectory.json",
339
- "provider_usage": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/provider-usage.jsonl",
340
- "transcript": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/transcript.json",
341
- "verdict": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/episode.json",
342
- "protocol": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/protocol.json",
343
- "native_goal": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/instructions.json",
344
- "workspace": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/workspace.tar.gz",
345
- "workspace_changes": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/workspace.json",
346
- "owner_journal": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/evidence/episode/evidence/journal/actions.jsonl",
347
- "recording_manifest": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/evidence/episode/evidence/recording/manifest.json",
348
- "usage": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/usage.json",
349
- "analysis": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/analysis.json",
350
- "provenance": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/provenance.json"
351
  },
352
  "resources": [
353
  {
354
  "name": "tools/tube_control.py",
355
- "file": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/resources/tools/tube_control.py",
356
  "kind": "Created during this episode; final workspace snapshot."
357
  }
358
  ],
 
153
  "protocol_sha256": "558b51d6786cd5ba0a989f7b2b872a3e37424dda668128a679113df86f67e43d"
154
  },
155
  "links": {
156
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/session.jsonl",
157
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/trajectory.json",
158
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/provider-usage.jsonl",
159
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/transcript.json",
160
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/episode.json",
161
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/protocol.json",
162
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/instructions.json",
163
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/workspace.tar.gz",
164
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/workspace.json",
165
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/evidence/episode/evidence/journal/actions.jsonl",
166
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/evidence/episode/evidence/recording/manifest.json",
167
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/usage.json",
168
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/analysis.json",
169
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/provenance.json"
170
  },
171
  "resources": [
172
  {
173
  "name": "tools/robot.py",
174
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/resources/tools/robot.py",
175
  "kind": "Created during this episode; final workspace snapshot."
176
  }
177
  ],
 
334
  "protocol_sha256": "331cfc33f7aa656ce04ffa0d8e7927d625ea36dd825616041760649c239cee4b"
335
  },
336
  "links": {
337
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/session.jsonl",
338
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/trajectory.json",
339
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/provider-usage.jsonl",
340
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/transcript.json",
341
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/episode.json",
342
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/protocol.json",
343
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/instructions.json",
344
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/workspace.tar.gz",
345
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/workspace.json",
346
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/evidence/episode/evidence/journal/actions.jsonl",
347
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/evidence/episode/evidence/recording/manifest.json",
348
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/usage.json",
349
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/analysis.json",
350
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/provenance.json"
351
  },
352
  "resources": [
353
  {
354
  "name": "tools/tube_control.py",
355
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/resources/tools/tube_control.py",
356
  "kind": "Created during this episode; final workspace snapshot."
357
  }
358
  ],
comparison.json CHANGED
@@ -1,10 +1,10 @@
1
  {
2
  "baseline_revision": "1a7f1c9f178ba7aeb659e5b527a77978e9136072",
3
  "summary": {
4
- "completed": 23,
5
- "codex_successes": 18,
6
- "codex_success_rate_completed_subset": 0.782608695652174,
7
- "kinex_successes_same_subset": 19,
8
  "kinex_successes_all42": 35,
9
  "all42_comparison_complete": false
10
  },
@@ -132,8 +132,8 @@
132
  {
133
  "task_key": "task04/17",
134
  "native_id": "robodojo/fill-pen-holder",
135
- "status": "pending",
136
- "codex_success": null,
137
  "kinex_success": false,
138
  "kinex_version": "0.10.3"
139
  },
@@ -204,8 +204,8 @@
204
  {
205
  "task_key": "task04/29",
206
  "native_id": "robodojo/pack-objects-into-box",
207
- "status": "pending",
208
- "codex_success": null,
209
  "kinex_success": true,
210
  "kinex_version": "0.10.2"
211
  },
@@ -244,32 +244,32 @@
244
  {
245
  "task_key": "task04/35",
246
  "native_id": "robodojo/pour-by-language",
247
- "status": "pending",
248
- "codex_success": null,
249
  "kinex_success": true,
250
  "kinex_version": "0.10.0"
251
  },
252
  {
253
  "task_key": "task04/36",
254
  "native_id": "robodojo/pour-liquid-into-cup",
255
- "status": "pending",
256
- "codex_success": null,
257
  "kinex_success": true,
258
  "kinex_version": "0.10.3"
259
  },
260
  {
261
  "task_key": "task04/38",
262
  "native_id": "robodojo/press-by-number",
263
- "status": "pending",
264
- "codex_success": null,
265
  "kinex_success": true,
266
  "kinex_version": "0.10.3"
267
  },
268
  {
269
  "task_key": "task04/39",
270
  "native_id": "robodojo/push-t",
271
- "status": "pending",
272
- "codex_success": null,
273
  "kinex_success": true,
274
  "kinex_version": "0.10.3"
275
  },
 
1
  {
2
  "baseline_revision": "1a7f1c9f178ba7aeb659e5b527a77978e9136072",
3
  "summary": {
4
+ "completed": 29,
5
+ "codex_successes": 21,
6
+ "codex_success_rate_completed_subset": 0.7241379310344828,
7
+ "kinex_successes_same_subset": 24,
8
  "kinex_successes_all42": 35,
9
  "all42_comparison_complete": false
10
  },
 
132
  {
133
  "task_key": "task04/17",
134
  "native_id": "robodojo/fill-pen-holder",
135
+ "status": "completed",
136
+ "codex_success": false,
137
  "kinex_success": false,
138
  "kinex_version": "0.10.3"
139
  },
 
204
  {
205
  "task_key": "task04/29",
206
  "native_id": "robodojo/pack-objects-into-box",
207
+ "status": "completed",
208
+ "codex_success": true,
209
  "kinex_success": true,
210
  "kinex_version": "0.10.2"
211
  },
 
244
  {
245
  "task_key": "task04/35",
246
  "native_id": "robodojo/pour-by-language",
247
+ "status": "completed",
248
+ "codex_success": false,
249
  "kinex_success": true,
250
  "kinex_version": "0.10.0"
251
  },
252
  {
253
  "task_key": "task04/36",
254
  "native_id": "robodojo/pour-liquid-into-cup",
255
+ "status": "completed",
256
+ "codex_success": true,
257
  "kinex_success": true,
258
  "kinex_version": "0.10.3"
259
  },
260
  {
261
  "task_key": "task04/38",
262
  "native_id": "robodojo/press-by-number",
263
+ "status": "completed",
264
+ "codex_success": true,
265
  "kinex_success": true,
266
  "kinex_version": "0.10.3"
267
  },
268
  {
269
  "task_key": "task04/39",
270
  "native_id": "robodojo/push-t",
271
+ "status": "completed",
272
+ "codex_success": false,
273
  "kinex_success": true,
274
  "kinex_version": "0.10.3"
275
  },
data.json CHANGED
@@ -6,7 +6,7 @@
6
  "benchmark_complete": false,
7
  "edition": "plain-codex-astra-high-seed0",
8
  "created_at": "2026-10-09T00:09:42.989838+00:00",
9
- "updated_at": "2026-10-09T03:48:09.209466+00:00",
10
  "model": "gpt-6-astra",
11
  "effort": "high",
12
  "seed": 0,
@@ -17,37 +17,37 @@
17
  },
18
  "summary": {
19
  "planned_tasks": 42,
20
- "published_results": 23,
21
- "pending_tasks": 19,
22
  "families": [
23
  {
24
  "id": "task04",
25
  "name": "RoboDojo",
26
  "total": 42,
27
- "completed": 23,
28
- "pending": 19,
29
- "successes": 18,
30
- "valid_results": 23,
31
- "success_rate": 0.782608695652174,
32
- "input_tokens": 113213163,
33
- "cached_input_tokens": 110735488,
34
- "output_tokens": 458950,
35
- "usage_complete": 23,
36
  "control_frequency_hz": 25,
37
  "max_control_steps": 7500,
38
  "preflight_results": 0,
39
- "formal_results": 23,
40
- "modified_results": 18,
41
- "original_results": 5
42
  }
43
  ],
44
  "progress": {
45
- "finished": 23,
46
  "interrupted": 2,
47
- "native_failures": 5,
48
- "native_successes": 18,
49
  "needs_review": 0,
50
- "queued": 10,
51
  "running": 7
52
  },
53
  "interrupted_attempts": 2,
@@ -59,21 +59,21 @@
59
  "id": "task04",
60
  "name": "RoboDojo",
61
  "total": 42,
62
- "completed": 23,
63
- "pending": 19,
64
- "successes": 18,
65
- "valid_results": 23,
66
- "success_rate": 0.782608695652174,
67
- "input_tokens": 113213163,
68
- "cached_input_tokens": 110735488,
69
- "output_tokens": 458950,
70
- "usage_complete": 23,
71
  "control_frequency_hz": 25,
72
  "max_control_steps": 7500,
73
  "preflight_results": 0,
74
- "formal_results": 23,
75
- "modified_results": 18,
76
- "original_results": 5
77
  }
78
  ],
79
  "tasks": [
@@ -1496,8 +1496,8 @@
1496
  "catalog_instruction": "Hold the pen holder with one hand and insert all the pens with the other, writing ends down and fully seated inside. Put the holder down upright, release it, and return both arms to their starting poses.",
1497
  "native_instruction": "Hold the pen holder with one hand, place all pens into it with the other hand, then put it back down.",
1498
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
1499
- "status": "pending",
1500
- "episode_id": null,
1501
  "planned_protocol": {
1502
  "episodes": 1,
1503
  "seed": 0,
@@ -1628,8 +1628,8 @@
1628
  },
1629
  "display_slot": "16",
1630
  "display_key": "task04/16",
1631
- "run_status": "running",
1632
- "status_note": "Currently running.",
1633
  "attempt_history": []
1634
  },
1635
  {
@@ -2346,8 +2346,8 @@
2346
  "catalog_instruction": "Place all four objects inside the bottom of the box, with their front sides facing left from the robot's viewpoint and their lengths aligned along the box. Keep the box upright with its long sides running left to right across the table. Release the objects and return both arms to their starting poses.",
2347
  "native_instruction": "Place all the objects into the box with their front sides facing left.",
2348
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2349
- "status": "pending",
2350
- "episode_id": null,
2351
  "planned_protocol": {
2352
  "episodes": 1,
2353
  "seed": 0,
@@ -2410,8 +2410,8 @@
2410
  },
2411
  "display_slot": "25",
2412
  "display_key": "task04/25",
2413
- "run_status": "running",
2414
- "status_note": "Currently running.",
2415
  "attempt_history": []
2416
  },
2417
  {
@@ -2777,8 +2777,8 @@
2777
  "catalog_instruction": "Complete the benchmark task: pour by language.",
2778
  "native_instruction": "Pour the liquid from the first violet bottle into the first black bowl, from the second red bottle into the second white bowl, and from the third turquoise bottle into the third brown bowl. Then reset the robot arm.",
2779
  "instruction_source": "runtime native task.instruction",
2780
- "status": "pending",
2781
- "episode_id": null,
2782
  "planned_protocol": {
2783
  "episodes": 1,
2784
  "seed": 0,
@@ -2790,8 +2790,8 @@
2790
  },
2791
  "display_slot": "30",
2792
  "display_key": "task04/30",
2793
- "run_status": "running",
2794
- "status_note": "Currently running.",
2795
  "attempt_history": []
2796
  },
2797
  {
@@ -2803,8 +2803,8 @@
2803
  "catalog_instruction": "Pour nearly all the liquid into the cup with almost no spillage. Keep the bottle tilted over the cup until the flow stops and the liquid settles, then return it upright, leaving only a small residue inside.",
2804
  "native_instruction": "Pour the liquid from the bottle into the cup.",
2805
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2806
- "status": "pending",
2807
- "episode_id": null,
2808
  "planned_protocol": {
2809
  "episodes": 1,
2810
  "seed": 0,
@@ -2870,8 +2870,8 @@
2870
  },
2871
  "display_slot": "31",
2872
  "display_key": "task04/31",
2873
- "run_status": "running",
2874
- "status_note": "Currently running.",
2875
  "attempt_history": []
2876
  },
2877
  {
@@ -2883,8 +2883,8 @@
2883
  "catalog_instruction": "Starting with the left red button, press and release each red button the number of times shown on its card, confirming that button's count with a press and release of the blue button before moving to the next. Return both arms to their starting poses when finished.",
2884
  "native_instruction": "Press the two red buttons the required number of times according to the number cards, then press the blue button to confirm.",
2885
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2886
- "status": "pending",
2887
- "episode_id": null,
2888
  "planned_protocol": {
2889
  "episodes": 1,
2890
  "seed": 0,
@@ -3035,8 +3035,8 @@
3035
  },
3036
  "display_slot": "32",
3037
  "display_key": "task04/32",
3038
- "run_status": "running",
3039
- "status_note": "Currently running.",
3040
  "attempt_history": []
3041
  },
3042
  {
@@ -3048,8 +3048,8 @@
3048
  "catalog_instruction": "Slide the T-shaped block along the table until it neatly matches the gray T-shaped pad in position and orientation. Keep the block on the table and return both arms to their starting poses when finished.",
3049
  "native_instruction": "Push the T-shaped block to align it precisely with the gray T-shaped pad.",
3050
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
3051
- "status": "pending",
3052
- "episode_id": null,
3053
  "planned_protocol": {
3054
  "episodes": 1,
3055
  "seed": 0,
@@ -3140,7 +3140,7 @@
3140
  },
3141
  "display_slot": "33",
3142
  "display_key": "task04/33",
3143
- "run_status": "queued",
3144
  "status_note": "Queued for evaluation.",
3145
  "attempt_history": []
3146
  },
@@ -3232,8 +3232,8 @@
3232
  },
3233
  "display_slot": "34",
3234
  "display_key": "task04/34",
3235
- "run_status": "queued",
3236
- "status_note": "Queued for evaluation.",
3237
  "attempt_history": []
3238
  },
3239
  {
@@ -3258,8 +3258,8 @@
3258
  },
3259
  "display_slot": "35",
3260
  "display_key": "task04/35",
3261
- "run_status": "queued",
3262
- "status_note": "Queued for evaluation.",
3263
  "attempt_history": []
3264
  },
3265
  {
@@ -3335,8 +3335,8 @@
3335
  },
3336
  "display_slot": "36",
3337
  "display_key": "task04/36",
3338
- "run_status": "queued",
3339
- "status_note": "Queued for evaluation.",
3340
  "attempt_history": []
3341
  },
3342
  {
@@ -3432,8 +3432,8 @@
3432
  },
3433
  "display_slot": "37",
3434
  "display_key": "task04/37",
3435
- "run_status": "queued",
3436
- "status_note": "Queued for evaluation.",
3437
  "attempt_history": []
3438
  },
3439
  {
@@ -3458,8 +3458,8 @@
3458
  },
3459
  "display_slot": "38",
3460
  "display_key": "task04/38",
3461
- "run_status": "queued",
3462
- "status_note": "Queued for evaluation.",
3463
  "attempt_history": []
3464
  },
3465
  {
@@ -3505,8 +3505,8 @@
3505
  },
3506
  "display_slot": "39",
3507
  "display_key": "task04/39",
3508
- "run_status": "queued",
3509
- "status_note": "Queued for evaluation.",
3510
  "attempt_history": []
3511
  },
3512
  {
@@ -7067,6 +7067,253 @@
7067
  "selected_for_formal_metrics": true,
7068
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/16/seed-0/"
7069
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7070
  {
7071
  "id": "task04-18-seed0-formal",
7072
  "task_key": "task04/18",
@@ -8792,16 +9039,16 @@
8792
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/28/seed-0/"
8793
  },
8794
  {
8795
- "id": "task04-31-seed0-formal",
8796
- "task_key": "task04/31",
8797
  "family": "task04",
8798
- "slot": "31",
8799
  "seed": 0,
8800
  "episode": 1,
8801
  "phase": "formal",
8802
  "status": "completed",
8803
- "success": false,
8804
- "native_reward": 0.0,
8805
  "valid": true,
8806
  "execution": {
8807
  "reason": null,
@@ -8809,9 +9056,255 @@
8809
  },
8810
  "verdict": {
8811
  "evidence_valid": true,
8812
- "steps": 863,
8813
- "success": false,
8814
- "termination": "stopped"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8815
  },
8816
  "steps": 863,
8817
  "simulation_time_s": null,
@@ -9534,6 +10027,1002 @@
9534
  },
9535
  "selected_for_formal_metrics": true,
9536
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/33/seed-0/"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9537
  }
9538
  ],
9539
  "failure_review": {
 
6
  "benchmark_complete": false,
7
  "edition": "plain-codex-astra-high-seed0",
8
  "created_at": "2026-10-09T00:09:42.989838+00:00",
9
+ "updated_at": "2026-10-09T04:02:32.635951+00:00",
10
  "model": "gpt-6-astra",
11
  "effort": "high",
12
  "seed": 0,
 
17
  },
18
  "summary": {
19
  "planned_tasks": 42,
20
+ "published_results": 29,
21
+ "pending_tasks": 13,
22
  "families": [
23
  {
24
  "id": "task04",
25
  "name": "RoboDojo",
26
  "total": 42,
27
+ "completed": 29,
28
+ "pending": 13,
29
+ "successes": 21,
30
+ "valid_results": 29,
31
+ "success_rate": 0.7241379310344828,
32
+ "input_tokens": 161829633,
33
+ "cached_input_tokens": 158729600,
34
+ "output_tokens": 619780,
35
+ "usage_complete": 29,
36
  "control_frequency_hz": 25,
37
  "max_control_steps": 7500,
38
  "preflight_results": 0,
39
+ "formal_results": 29,
40
+ "modified_results": 23,
41
+ "original_results": 6
42
  }
43
  ],
44
  "progress": {
45
+ "finished": 30,
46
  "interrupted": 2,
47
+ "native_failures": 8,
48
+ "native_successes": 22,
49
  "needs_review": 0,
50
+ "queued": 3,
51
  "running": 7
52
  },
53
  "interrupted_attempts": 2,
 
59
  "id": "task04",
60
  "name": "RoboDojo",
61
  "total": 42,
62
+ "completed": 29,
63
+ "pending": 13,
64
+ "successes": 21,
65
+ "valid_results": 29,
66
+ "success_rate": 0.7241379310344828,
67
+ "input_tokens": 161829633,
68
+ "cached_input_tokens": 158729600,
69
+ "output_tokens": 619780,
70
+ "usage_complete": 29,
71
  "control_frequency_hz": 25,
72
  "max_control_steps": 7500,
73
  "preflight_results": 0,
74
+ "formal_results": 29,
75
+ "modified_results": 23,
76
+ "original_results": 6
77
  }
78
  ],
79
  "tasks": [
 
1496
  "catalog_instruction": "Hold the pen holder with one hand and insert all the pens with the other, writing ends down and fully seated inside. Put the holder down upright, release it, and return both arms to their starting poses.",
1497
  "native_instruction": "Hold the pen holder with one hand, place all pens into it with the other hand, then put it back down.",
1498
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
1499
+ "status": "completed",
1500
+ "episode_id": "task04-17-seed0-formal",
1501
  "planned_protocol": {
1502
  "episodes": 1,
1503
  "seed": 0,
 
1628
  },
1629
  "display_slot": "16",
1630
  "display_key": "task04/16",
1631
+ "run_status": "finished",
1632
+ "status_note": "Queued for evaluation.",
1633
  "attempt_history": []
1634
  },
1635
  {
 
2346
  "catalog_instruction": "Place all four objects inside the bottom of the box, with their front sides facing left from the robot's viewpoint and their lengths aligned along the box. Keep the box upright with its long sides running left to right across the table. Release the objects and return both arms to their starting poses.",
2347
  "native_instruction": "Place all the objects into the box with their front sides facing left.",
2348
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2349
+ "status": "completed",
2350
+ "episode_id": "task04-29-seed0-formal",
2351
  "planned_protocol": {
2352
  "episodes": 1,
2353
  "seed": 0,
 
2410
  },
2411
  "display_slot": "25",
2412
  "display_key": "task04/25",
2413
+ "run_status": "finished",
2414
+ "status_note": "Queued for evaluation.",
2415
  "attempt_history": []
2416
  },
2417
  {
 
2777
  "catalog_instruction": "Complete the benchmark task: pour by language.",
2778
  "native_instruction": "Pour the liquid from the first violet bottle into the first black bowl, from the second red bottle into the second white bowl, and from the third turquoise bottle into the third brown bowl. Then reset the robot arm.",
2779
  "instruction_source": "runtime native task.instruction",
2780
+ "status": "completed",
2781
+ "episode_id": "task04-35-seed0-formal",
2782
  "planned_protocol": {
2783
  "episodes": 1,
2784
  "seed": 0,
 
2790
  },
2791
  "display_slot": "30",
2792
  "display_key": "task04/30",
2793
+ "run_status": "finished",
2794
+ "status_note": "Queued for evaluation.",
2795
  "attempt_history": []
2796
  },
2797
  {
 
2803
  "catalog_instruction": "Pour nearly all the liquid into the cup with almost no spillage. Keep the bottle tilted over the cup until the flow stops and the liquid settles, then return it upright, leaving only a small residue inside.",
2804
  "native_instruction": "Pour the liquid from the bottle into the cup.",
2805
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2806
+ "status": "completed",
2807
+ "episode_id": "task04-36-seed0-formal",
2808
  "planned_protocol": {
2809
  "episodes": 1,
2810
  "seed": 0,
 
2870
  },
2871
  "display_slot": "31",
2872
  "display_key": "task04/31",
2873
+ "run_status": "finished",
2874
+ "status_note": "Queued for evaluation.",
2875
  "attempt_history": []
2876
  },
2877
  {
 
2883
  "catalog_instruction": "Starting with the left red button, press and release each red button the number of times shown on its card, confirming that button's count with a press and release of the blue button before moving to the next. Return both arms to their starting poses when finished.",
2884
  "native_instruction": "Press the two red buttons the required number of times according to the number cards, then press the blue button to confirm.",
2885
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2886
+ "status": "completed",
2887
+ "episode_id": "task04-38-seed0-formal",
2888
  "planned_protocol": {
2889
  "episodes": 1,
2890
  "seed": 0,
 
3035
  },
3036
  "display_slot": "32",
3037
  "display_key": "task04/32",
3038
+ "run_status": "finished",
3039
+ "status_note": "Queued for evaluation.",
3040
  "attempt_history": []
3041
  },
3042
  {
 
3048
  "catalog_instruction": "Slide the T-shaped block along the table until it neatly matches the gray T-shaped pad in position and orientation. Keep the block on the table and return both arms to their starting poses when finished.",
3049
  "native_instruction": "Push the T-shaped block to align it precisely with the gray T-shaped pad.",
3050
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
3051
+ "status": "completed",
3052
+ "episode_id": "task04-39-seed0-formal",
3053
  "planned_protocol": {
3054
  "episodes": 1,
3055
  "seed": 0,
 
3140
  },
3141
  "display_slot": "33",
3142
  "display_key": "task04/33",
3143
+ "run_status": "finished",
3144
  "status_note": "Queued for evaluation.",
3145
  "attempt_history": []
3146
  },
 
3232
  },
3233
  "display_slot": "34",
3234
  "display_key": "task04/34",
3235
+ "run_status": "running",
3236
+ "status_note": "Currently running.",
3237
  "attempt_history": []
3238
  },
3239
  {
 
3258
  },
3259
  "display_slot": "35",
3260
  "display_key": "task04/35",
3261
+ "run_status": "finished",
3262
+ "status_note": "Evaluation finished; media preparation is in progress.",
3263
  "attempt_history": []
3264
  },
3265
  {
 
3335
  },
3336
  "display_slot": "36",
3337
  "display_key": "task04/36",
3338
+ "run_status": "running",
3339
+ "status_note": "Currently running.",
3340
  "attempt_history": []
3341
  },
3342
  {
 
3432
  },
3433
  "display_slot": "37",
3434
  "display_key": "task04/37",
3435
+ "run_status": "running",
3436
+ "status_note": "Currently running.",
3437
  "attempt_history": []
3438
  },
3439
  {
 
3458
  },
3459
  "display_slot": "38",
3460
  "display_key": "task04/38",
3461
+ "run_status": "running",
3462
+ "status_note": "Currently running.",
3463
  "attempt_history": []
3464
  },
3465
  {
 
3505
  },
3506
  "display_slot": "39",
3507
  "display_key": "task04/39",
3508
+ "run_status": "running",
3509
+ "status_note": "Currently running.",
3510
  "attempt_history": []
3511
  },
3512
  {
 
7067
  "selected_for_formal_metrics": true,
7068
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/16/seed-0/"
7069
  },
7070
+ {
7071
+ "id": "task04-17-seed0-formal",
7072
+ "task_key": "task04/17",
7073
+ "family": "task04",
7074
+ "slot": "17",
7075
+ "seed": 0,
7076
+ "episode": 1,
7077
+ "phase": "formal",
7078
+ "status": "completed",
7079
+ "success": false,
7080
+ "native_reward": 0.25,
7081
+ "valid": true,
7082
+ "execution": {
7083
+ "reason": null,
7084
+ "status": "finished"
7085
+ },
7086
+ "verdict": {
7087
+ "evidence_valid": true,
7088
+ "steps": 7440,
7089
+ "success": false,
7090
+ "termination": "stopped"
7091
+ },
7092
+ "steps": 7440,
7093
+ "simulation_time_s": null,
7094
+ "wall_time_s": 5845.33904,
7095
+ "model": "gpt-6-astra",
7096
+ "effort": "high",
7097
+ "harness": "codex",
7098
+ "codex_version": "0.160.0",
7099
+ "native_instruction": "Hold the pen holder with one hand, place all pens into it with the other hand, then put it back down.",
7100
+ "instruction": "Hold the pen holder with one hand and insert all the pens with the other, writing ends down and fully seated inside. Put the holder down upright, release it, and return both arms to their starting poses.",
7101
+ "instruction_policy": "modified",
7102
+ "usage": {
7103
+ "accounting": "reported-responses",
7104
+ "audit_complete": true,
7105
+ "cache_hit_rate": 0.992503006146394,
7106
+ "cache_reported_input_tokens": 29201838,
7107
+ "cache_write_input_tokens": 0,
7108
+ "cache_write_reported_input_tokens": 29201838,
7109
+ "cached_input_tokens": 28982912,
7110
+ "completed_turns": 1,
7111
+ "cost_usd": null,
7112
+ "failed_turns": 0,
7113
+ "input_tokens": 29201838,
7114
+ "known_cache_write_input_tokens": 0,
7115
+ "known_cached_input_tokens": 28982912,
7116
+ "known_input_tokens": 29201838,
7117
+ "known_output_tokens": 80788,
7118
+ "known_reasoning_output_tokens": 48714,
7119
+ "output_tokens": 80788,
7120
+ "reasoning_output_tokens": 48714,
7121
+ "reasoning_reported_output_tokens": 80788,
7122
+ "reported_responses": {
7123
+ "cache_reported_input_tokens": 289,
7124
+ "cache_write_input_tokens": 289,
7125
+ "cache_write_reported_input_tokens": 289,
7126
+ "cached_input_tokens": 289,
7127
+ "input_tokens": 289,
7128
+ "output_tokens": 289,
7129
+ "reasoning_output_tokens": 289,
7130
+ "reasoning_reported_output_tokens": 289
7131
+ },
7132
+ "response_count": 289,
7133
+ "response_ids_complete": true,
7134
+ "schema": "rlebench/token-usage/1",
7135
+ "source": "Codex token_usage_record per response",
7136
+ "uncached_input_tokens": 218926,
7137
+ "unidentified_usage_records": 0
7138
+ },
7139
+ "call_activity": {
7140
+ "model_tool_calls": 288,
7141
+ "model_tool_calls_by_name": {
7142
+ "exec": 288
7143
+ },
7144
+ "nested_python_tool_invocations": null,
7145
+ "python_device_rpc_attempts": null,
7146
+ "python_device_rpc_attempts_by_action": null,
7147
+ "python_device_rpc_errors": null,
7148
+ "python_instrumented_model_tool_calls": null,
7149
+ "python_tool_invocations": null,
7150
+ "python_tool_invocations_by_origin": null,
7151
+ "schema": "rlebench/call-activity/1",
7152
+ "source": "Codex native sessions"
7153
+ },
7154
+ "media": {
7155
+ "passed": true,
7156
+ "width": 2880,
7157
+ "height": 720,
7158
+ "duration_s": 74.4,
7159
+ "speed": 4,
7160
+ "source_fps": 10,
7161
+ "output_fps": 20,
7162
+ "recording": {
7163
+ "accepted_samples": 2977,
7164
+ "captured_samples": 2977,
7165
+ "clock": "simulation",
7166
+ "dropped_samples": 0,
7167
+ "encoded_frames": 2977,
7168
+ "end_time_s": 297.6000000000046,
7169
+ "error": null,
7170
+ "experimental": true,
7171
+ "fps": 10,
7172
+ "received_samples": 2977,
7173
+ "schema": "roboenv/recording/1",
7174
+ "state": "closed",
7175
+ "status": "complete",
7176
+ "views": [
7177
+ {
7178
+ "fov_y": 45.0,
7179
+ "height": 720,
7180
+ "name": "third_person",
7181
+ "pose": null,
7182
+ "source": "third_person",
7183
+ "width": 960
7184
+ },
7185
+ {
7186
+ "fov_y": 45.0,
7187
+ "height": 720,
7188
+ "name": "left_wrist",
7189
+ "pose": null,
7190
+ "source": "left_wrist",
7191
+ "width": 960
7192
+ },
7193
+ {
7194
+ "fov_y": 45.0,
7195
+ "height": 720,
7196
+ "name": "right_wrist",
7197
+ "pose": null,
7198
+ "source": "right_wrist",
7199
+ "width": 960
7200
+ }
7201
+ ]
7202
+ },
7203
+ "view_names": [
7204
+ "third_person",
7205
+ "left_wrist",
7206
+ "right_wrist"
7207
+ ],
7208
+ "sha256": "f3c9626399fbbdd5e6f15a39f9e9967e3ac90b6d1c85b0ba822ae16f36465301",
7209
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
7210
+ },
7211
+ "analysis": {
7212
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
7213
+ "duration": "\u4f7f\u7528 7,440 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
7214
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
7215
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
7216
+ },
7217
+ "provenance": {
7218
+ "sources": {
7219
+ "RLE-Bench-inhouse": {
7220
+ "build_inputs": [
7221
+ "pyproject.toml",
7222
+ "src",
7223
+ "tasks",
7224
+ "README.md",
7225
+ "Makefile",
7226
+ "tests",
7227
+ "docs"
7228
+ ],
7229
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
7230
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
7231
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
7232
+ },
7233
+ "RoboEnv": {
7234
+ "build_inputs": [
7235
+ "pyproject.toml",
7236
+ "README.md",
7237
+ "src",
7238
+ "runtime/pyproject.toml",
7239
+ "runtime/README.md",
7240
+ "runtime/src",
7241
+ "runtime/environments.json",
7242
+ "runtime/locks",
7243
+ "catalog",
7244
+ "upstreams.lock.json",
7245
+ "third_party/patches",
7246
+ "docs/validation"
7247
+ ],
7248
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
7249
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
7250
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
7251
+ }
7252
+ },
7253
+ "job": "17-robodojo-fill-pen-holder-codex-seed0-attempt01",
7254
+ "attempt": 1,
7255
+ "harness": "stock Codex CLI",
7256
+ "codex_version": "0.160.0",
7257
+ "model": "gpt-6-astra",
7258
+ "effort": "high",
7259
+ "service_tier": "default",
7260
+ "fresh_session": true,
7261
+ "source_jobs": [],
7262
+ "resume_trajectory": false,
7263
+ "imported_skills": [],
7264
+ "automatic_harbor_retries": 0,
7265
+ "request_policy": {
7266
+ "max_request_retries": 50,
7267
+ "configuration": "explicit retry50 SSE",
7268
+ "usage_accounting": "reported-responses"
7269
+ },
7270
+ "classification": "formal",
7271
+ "measured_images": {
7272
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
7273
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
7274
+ },
7275
+ "session_original_sha256": "4de9c3c264f52832303def2e45505f2814b7fc5a0fe89b7ca676fd6d8c693b71",
7276
+ "protocol_sha256": "8db05ad726248326912895583e4fe0eb492747d9f7702eb2b524abfb8da02a0a"
7277
+ },
7278
+ "links": {
7279
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/agent/session.jsonl",
7280
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/agent/trajectory.json",
7281
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/agent/provider-usage.jsonl",
7282
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/transcript.json",
7283
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/verifier/episode.json",
7284
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/verifier/protocol.json",
7285
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/instructions.json",
7286
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/verifier/workspace.tar.gz",
7287
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/verifier/workspace.json",
7288
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
7289
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/evidence/episode/evidence/recording/manifest.json",
7290
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/usage.json",
7291
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/analysis.json",
7292
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/provenance.json",
7293
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/video.mp4",
7294
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/poster.jpg",
7295
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/media-validation.json"
7296
+ },
7297
+ "resources": [
7298
+ {
7299
+ "name": "tools/robot.py",
7300
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/resources/tools/robot.py",
7301
+ "kind": "Created during this episode; final workspace snapshot."
7302
+ },
7303
+ {
7304
+ "name": "memos/robodojo-manipulation.md",
7305
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/resources/memos/robodojo-manipulation.md",
7306
+ "kind": "Created during this episode; final workspace snapshot."
7307
+ }
7308
+ ],
7309
+ "session_counts": {
7310
+ "visible_events": 608,
7311
+ "observed_images": 123,
7312
+ "tool_errors": 20
7313
+ },
7314
+ "selected_for_formal_metrics": true,
7315
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/"
7316
+ },
7317
  {
7318
  "id": "task04-18-seed0-formal",
7319
  "task_key": "task04/18",
 
9039
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/28/seed-0/"
9040
  },
9041
  {
9042
+ "id": "task04-29-seed0-formal",
9043
+ "task_key": "task04/29",
9044
  "family": "task04",
9045
+ "slot": "29",
9046
  "seed": 0,
9047
  "episode": 1,
9048
  "phase": "formal",
9049
  "status": "completed",
9050
+ "success": true,
9051
+ "native_reward": 1.0,
9052
  "valid": true,
9053
  "execution": {
9054
  "reason": null,
 
9056
  },
9057
  "verdict": {
9058
  "evidence_valid": true,
9059
+ "steps": 6433,
9060
+ "success": true,
9061
+ "termination": "success"
9062
+ },
9063
+ "steps": 6433,
9064
+ "simulation_time_s": null,
9065
+ "wall_time_s": 2657.323509,
9066
+ "model": "gpt-6-astra",
9067
+ "effort": "high",
9068
+ "harness": "codex",
9069
+ "codex_version": "0.160.0",
9070
+ "native_instruction": "Place all the objects into the box with their front sides facing left.",
9071
+ "instruction": "Place all four objects inside the bottom of the box, with their front sides facing left from the robot's viewpoint and their lengths aligned along the box. Keep the box upright with its long sides running left to right across the table. Release the objects and return both arms to their starting poses.",
9072
+ "instruction_policy": "modified",
9073
+ "usage": {
9074
+ "accounting": "reported-responses",
9075
+ "audit_complete": true,
9076
+ "cache_hit_rate": 0.985806960138145,
9077
+ "cache_reported_input_tokens": 12341824,
9078
+ "cache_write_input_tokens": 0,
9079
+ "cache_write_reported_input_tokens": 12341824,
9080
+ "cached_input_tokens": 12166656,
9081
+ "completed_turns": 1,
9082
+ "cost_usd": null,
9083
+ "failed_turns": 0,
9084
+ "input_tokens": 12341824,
9085
+ "known_cache_write_input_tokens": 0,
9086
+ "known_cached_input_tokens": 12166656,
9087
+ "known_input_tokens": 12341824,
9088
+ "known_output_tokens": 39288,
9089
+ "known_reasoning_output_tokens": 20764,
9090
+ "output_tokens": 39288,
9091
+ "reasoning_output_tokens": 20764,
9092
+ "reasoning_reported_output_tokens": 39288,
9093
+ "reported_responses": {
9094
+ "cache_reported_input_tokens": 172,
9095
+ "cache_write_input_tokens": 172,
9096
+ "cache_write_reported_input_tokens": 172,
9097
+ "cached_input_tokens": 172,
9098
+ "input_tokens": 172,
9099
+ "output_tokens": 172,
9100
+ "reasoning_output_tokens": 172,
9101
+ "reasoning_reported_output_tokens": 172
9102
+ },
9103
+ "response_count": 172,
9104
+ "response_ids_complete": true,
9105
+ "schema": "rlebench/token-usage/1",
9106
+ "source": "Codex token_usage_record per response",
9107
+ "uncached_input_tokens": 175168,
9108
+ "unidentified_usage_records": 0
9109
+ },
9110
+ "call_activity": {
9111
+ "model_tool_calls": 171,
9112
+ "model_tool_calls_by_name": {
9113
+ "exec": 171
9114
+ },
9115
+ "nested_python_tool_invocations": null,
9116
+ "python_device_rpc_attempts": null,
9117
+ "python_device_rpc_attempts_by_action": null,
9118
+ "python_device_rpc_errors": null,
9119
+ "python_instrumented_model_tool_calls": null,
9120
+ "python_tool_invocations": null,
9121
+ "python_tool_invocations_by_origin": null,
9122
+ "schema": "rlebench/call-activity/1",
9123
+ "source": "Codex native sessions"
9124
+ },
9125
+ "media": {
9126
+ "passed": true,
9127
+ "width": 2880,
9128
+ "height": 720,
9129
+ "duration_s": 64.35,
9130
+ "speed": 4,
9131
+ "source_fps": 10,
9132
+ "output_fps": 20,
9133
+ "recording": {
9134
+ "accepted_samples": 2574,
9135
+ "captured_samples": 2574,
9136
+ "clock": "simulation",
9137
+ "dropped_samples": 0,
9138
+ "encoded_frames": 2574,
9139
+ "end_time_s": 257.319999999984,
9140
+ "error": null,
9141
+ "experimental": true,
9142
+ "fps": 10,
9143
+ "received_samples": 2574,
9144
+ "schema": "roboenv/recording/1",
9145
+ "state": "closed",
9146
+ "status": "complete",
9147
+ "views": [
9148
+ {
9149
+ "fov_y": 45.0,
9150
+ "height": 720,
9151
+ "name": "third_person",
9152
+ "pose": null,
9153
+ "source": "third_person",
9154
+ "width": 960
9155
+ },
9156
+ {
9157
+ "fov_y": 45.0,
9158
+ "height": 720,
9159
+ "name": "left_wrist",
9160
+ "pose": null,
9161
+ "source": "left_wrist",
9162
+ "width": 960
9163
+ },
9164
+ {
9165
+ "fov_y": 45.0,
9166
+ "height": 720,
9167
+ "name": "right_wrist",
9168
+ "pose": null,
9169
+ "source": "right_wrist",
9170
+ "width": 960
9171
+ }
9172
+ ]
9173
+ },
9174
+ "view_names": [
9175
+ "third_person",
9176
+ "left_wrist",
9177
+ "right_wrist"
9178
+ ],
9179
+ "sha256": "8b4d310b6dc63abb0100121b03fe18a0b07a44a2a63f6df843064e356fb18898",
9180
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
9181
+ },
9182
+ "analysis": {
9183
+ "outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
9184
+ "duration": "\u4f7f\u7528 6,433 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
9185
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
9186
+ },
9187
+ "provenance": {
9188
+ "sources": {
9189
+ "RLE-Bench-inhouse": {
9190
+ "build_inputs": [
9191
+ "pyproject.toml",
9192
+ "src",
9193
+ "tasks",
9194
+ "README.md",
9195
+ "Makefile",
9196
+ "tests",
9197
+ "docs"
9198
+ ],
9199
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
9200
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
9201
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
9202
+ },
9203
+ "RoboEnv": {
9204
+ "build_inputs": [
9205
+ "pyproject.toml",
9206
+ "README.md",
9207
+ "src",
9208
+ "runtime/pyproject.toml",
9209
+ "runtime/README.md",
9210
+ "runtime/src",
9211
+ "runtime/environments.json",
9212
+ "runtime/locks",
9213
+ "catalog",
9214
+ "upstreams.lock.json",
9215
+ "third_party/patches",
9216
+ "docs/validation"
9217
+ ],
9218
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
9219
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
9220
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
9221
+ }
9222
+ },
9223
+ "job": "29-robodojo-pack-objects-into-box-codex-seed0-attempt01",
9224
+ "attempt": 1,
9225
+ "harness": "stock Codex CLI",
9226
+ "codex_version": "0.160.0",
9227
+ "model": "gpt-6-astra",
9228
+ "effort": "high",
9229
+ "service_tier": "default",
9230
+ "fresh_session": true,
9231
+ "source_jobs": [],
9232
+ "resume_trajectory": false,
9233
+ "imported_skills": [],
9234
+ "automatic_harbor_retries": 0,
9235
+ "request_policy": {
9236
+ "max_request_retries": 50,
9237
+ "configuration": "explicit retry50 SSE",
9238
+ "usage_accounting": "reported-responses"
9239
+ },
9240
+ "classification": "formal",
9241
+ "measured_images": {
9242
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
9243
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
9244
+ },
9245
+ "session_original_sha256": "a310e18ccc4536315f21afc01221789490558fff6e3c316dc38c7fafeeaca76b",
9246
+ "protocol_sha256": "d344c9afa60711fa33950eb1a8cfc805ff7c78c5ad9b476e4f3f524638f5980c"
9247
+ },
9248
+ "links": {
9249
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/agent/session.jsonl",
9250
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/agent/trajectory.json",
9251
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/agent/provider-usage.jsonl",
9252
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/transcript.json",
9253
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/verifier/episode.json",
9254
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/verifier/protocol.json",
9255
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/instructions.json",
9256
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/verifier/workspace.tar.gz",
9257
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/verifier/workspace.json",
9258
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
9259
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/evidence/episode/evidence/recording/manifest.json",
9260
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/usage.json",
9261
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/analysis.json",
9262
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/provenance.json",
9263
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/video.mp4",
9264
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/poster.jpg",
9265
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/media-validation.json"
9266
+ },
9267
+ "resources": [
9268
+ {
9269
+ "name": "tools/arm_control.py",
9270
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/resources/tools/arm_control.py",
9271
+ "kind": "Created during this episode; final workspace snapshot."
9272
+ },
9273
+ {
9274
+ "name": "memos/robodojo.md",
9275
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/resources/memos/robodojo.md",
9276
+ "kind": "Created during this episode; final workspace snapshot."
9277
+ }
9278
+ ],
9279
+ "session_counts": {
9280
+ "visible_events": 371,
9281
+ "observed_images": 56,
9282
+ "tool_errors": 9
9283
+ },
9284
+ "selected_for_formal_metrics": true,
9285
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/"
9286
+ },
9287
+ {
9288
+ "id": "task04-31-seed0-formal",
9289
+ "task_key": "task04/31",
9290
+ "family": "task04",
9291
+ "slot": "31",
9292
+ "seed": 0,
9293
+ "episode": 1,
9294
+ "phase": "formal",
9295
+ "status": "completed",
9296
+ "success": false,
9297
+ "native_reward": 0.0,
9298
+ "valid": true,
9299
+ "execution": {
9300
+ "reason": null,
9301
+ "status": "finished"
9302
+ },
9303
+ "verdict": {
9304
+ "evidence_valid": true,
9305
+ "steps": 863,
9306
+ "success": false,
9307
+ "termination": "stopped"
9308
  },
9309
  "steps": 863,
9310
  "simulation_time_s": null,
 
10027
  },
10028
  "selected_for_formal_metrics": true,
10029
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/33/seed-0/"
10030
+ },
10031
+ {
10032
+ "id": "task04-35-seed0-formal",
10033
+ "task_key": "task04/35",
10034
+ "family": "task04",
10035
+ "slot": "35",
10036
+ "seed": 0,
10037
+ "episode": 1,
10038
+ "phase": "formal",
10039
+ "status": "completed",
10040
+ "success": false,
10041
+ "native_reward": 0.0,
10042
+ "valid": true,
10043
+ "execution": {
10044
+ "reason": null,
10045
+ "status": "finished"
10046
+ },
10047
+ "verdict": {
10048
+ "evidence_valid": true,
10049
+ "steps": 1616,
10050
+ "success": false,
10051
+ "termination": "stopped"
10052
+ },
10053
+ "steps": 1616,
10054
+ "simulation_time_s": null,
10055
+ "wall_time_s": 834.470016,
10056
+ "model": "gpt-6-astra",
10057
+ "effort": "high",
10058
+ "harness": "codex",
10059
+ "codex_version": "0.160.0",
10060
+ "native_instruction": "Pour the liquid from the first violet bottle into the first black bowl, from the second red bottle into the second white bowl, and from the third turquoise bottle into the third brown bowl. Then reset the robot arm.",
10061
+ "instruction": "Pour the liquid from the first violet bottle into the first black bowl, from the second red bottle into the second white bowl, and from the third turquoise bottle into the third brown bowl. Then reset the robot arm.",
10062
+ "instruction_policy": "original_native",
10063
+ "usage": {
10064
+ "accounting": "reported-responses",
10065
+ "audit_complete": true,
10066
+ "cache_hit_rate": 0.9667383369019035,
10067
+ "cache_reported_input_tokens": 2677076,
10068
+ "cache_write_input_tokens": 0,
10069
+ "cache_write_reported_input_tokens": 2677076,
10070
+ "cached_input_tokens": 2588032,
10071
+ "completed_turns": 1,
10072
+ "cost_usd": null,
10073
+ "failed_turns": 0,
10074
+ "input_tokens": 2677076,
10075
+ "known_cache_write_input_tokens": 0,
10076
+ "known_cached_input_tokens": 2588032,
10077
+ "known_input_tokens": 2677076,
10078
+ "known_output_tokens": 11360,
10079
+ "known_reasoning_output_tokens": 3492,
10080
+ "output_tokens": 11360,
10081
+ "reasoning_output_tokens": 3492,
10082
+ "reasoning_reported_output_tokens": 11360,
10083
+ "reported_responses": {
10084
+ "cache_reported_input_tokens": 69,
10085
+ "cache_write_input_tokens": 69,
10086
+ "cache_write_reported_input_tokens": 69,
10087
+ "cached_input_tokens": 69,
10088
+ "input_tokens": 69,
10089
+ "output_tokens": 69,
10090
+ "reasoning_output_tokens": 69,
10091
+ "reasoning_reported_output_tokens": 69
10092
+ },
10093
+ "response_count": 69,
10094
+ "response_ids_complete": true,
10095
+ "schema": "rlebench/token-usage/1",
10096
+ "source": "Codex token_usage_record per response",
10097
+ "uncached_input_tokens": 89044,
10098
+ "unidentified_usage_records": 0
10099
+ },
10100
+ "call_activity": {
10101
+ "model_tool_calls": 68,
10102
+ "model_tool_calls_by_name": {
10103
+ "exec": 68
10104
+ },
10105
+ "nested_python_tool_invocations": null,
10106
+ "python_device_rpc_attempts": null,
10107
+ "python_device_rpc_attempts_by_action": null,
10108
+ "python_device_rpc_errors": null,
10109
+ "python_instrumented_model_tool_calls": null,
10110
+ "python_tool_invocations": null,
10111
+ "python_tool_invocations_by_origin": null,
10112
+ "schema": "rlebench/call-activity/1",
10113
+ "source": "Codex native sessions"
10114
+ },
10115
+ "media": {
10116
+ "passed": true,
10117
+ "width": 2880,
10118
+ "height": 720,
10119
+ "duration_s": 16.15,
10120
+ "speed": 4,
10121
+ "source_fps": 10,
10122
+ "output_fps": 20,
10123
+ "recording": {
10124
+ "accepted_samples": 648,
10125
+ "captured_samples": 648,
10126
+ "clock": "simulation",
10127
+ "dropped_samples": 0,
10128
+ "encoded_frames": 647,
10129
+ "end_time_s": 64.6399999999989,
10130
+ "error": null,
10131
+ "experimental": true,
10132
+ "fps": 10,
10133
+ "received_samples": 648,
10134
+ "schema": "roboenv/recording/1",
10135
+ "state": "closed",
10136
+ "status": "complete",
10137
+ "views": [
10138
+ {
10139
+ "fov_y": 45.0,
10140
+ "height": 720,
10141
+ "name": "third_person",
10142
+ "pose": null,
10143
+ "source": "third_person",
10144
+ "width": 960
10145
+ },
10146
+ {
10147
+ "fov_y": 45.0,
10148
+ "height": 720,
10149
+ "name": "left_wrist",
10150
+ "pose": null,
10151
+ "source": "left_wrist",
10152
+ "width": 960
10153
+ },
10154
+ {
10155
+ "fov_y": 45.0,
10156
+ "height": 720,
10157
+ "name": "right_wrist",
10158
+ "pose": null,
10159
+ "source": "right_wrist",
10160
+ "width": 960
10161
+ }
10162
+ ]
10163
+ },
10164
+ "view_names": [
10165
+ "third_person",
10166
+ "left_wrist",
10167
+ "right_wrist"
10168
+ ],
10169
+ "sha256": "207a29db68023f38d3e81e5c185ee358ef76f4dd8482e610e43b1fdc41d39da7",
10170
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
10171
+ },
10172
+ "analysis": {
10173
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
10174
+ "duration": "\u4f7f\u7528 1,616 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
10175
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
10176
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
10177
+ },
10178
+ "provenance": {
10179
+ "sources": {
10180
+ "RLE-Bench-inhouse": {
10181
+ "build_inputs": [
10182
+ "pyproject.toml",
10183
+ "src",
10184
+ "tasks",
10185
+ "README.md",
10186
+ "Makefile",
10187
+ "tests",
10188
+ "docs"
10189
+ ],
10190
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
10191
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
10192
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
10193
+ },
10194
+ "RoboEnv": {
10195
+ "build_inputs": [
10196
+ "pyproject.toml",
10197
+ "README.md",
10198
+ "src",
10199
+ "runtime/pyproject.toml",
10200
+ "runtime/README.md",
10201
+ "runtime/src",
10202
+ "runtime/environments.json",
10203
+ "runtime/locks",
10204
+ "catalog",
10205
+ "upstreams.lock.json",
10206
+ "third_party/patches",
10207
+ "docs/validation"
10208
+ ],
10209
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
10210
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
10211
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
10212
+ }
10213
+ },
10214
+ "job": "35-robodojo-pour-by-language-codex-seed0-attempt01",
10215
+ "attempt": 1,
10216
+ "harness": "stock Codex CLI",
10217
+ "codex_version": "0.160.0",
10218
+ "model": "gpt-6-astra",
10219
+ "effort": "high",
10220
+ "service_tier": "default",
10221
+ "fresh_session": true,
10222
+ "source_jobs": [],
10223
+ "resume_trajectory": false,
10224
+ "imported_skills": [],
10225
+ "automatic_harbor_retries": 0,
10226
+ "request_policy": {
10227
+ "max_request_retries": 50,
10228
+ "configuration": "explicit retry50 SSE",
10229
+ "usage_accounting": "reported-responses"
10230
+ },
10231
+ "classification": "formal",
10232
+ "measured_images": {
10233
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
10234
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
10235
+ },
10236
+ "session_original_sha256": "9823e4fb42ba34ba2a7c47bd1c5cd29624d9352da943026e6dd86a61f1da18b5",
10237
+ "protocol_sha256": "b10751bbc24694b2b82ddcb85171f0806a4d2c357f0001a79f4780ffb7e227aa"
10238
+ },
10239
+ "links": {
10240
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/agent/session.jsonl",
10241
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/agent/trajectory.json",
10242
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/agent/provider-usage.jsonl",
10243
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/transcript.json",
10244
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/verifier/episode.json",
10245
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/verifier/protocol.json",
10246
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/instructions.json",
10247
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/verifier/workspace.tar.gz",
10248
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/verifier/workspace.json",
10249
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
10250
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/evidence/episode/evidence/recording/manifest.json",
10251
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/usage.json",
10252
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/analysis.json",
10253
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/provenance.json",
10254
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/video.mp4",
10255
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/poster.jpg",
10256
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/media-validation.json"
10257
+ },
10258
+ "resources": [
10259
+ {
10260
+ "name": "tools/arx.py",
10261
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/resources/tools/arx.py",
10262
+ "kind": "Created during this episode; final workspace snapshot."
10263
+ },
10264
+ {
10265
+ "name": "memos/robodojo.md",
10266
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/resources/memos/robodojo.md",
10267
+ "kind": "Created during this episode; final workspace snapshot."
10268
+ }
10269
+ ],
10270
+ "session_counts": {
10271
+ "visible_events": 154,
10272
+ "observed_images": 17,
10273
+ "tool_errors": 2
10274
+ },
10275
+ "selected_for_formal_metrics": true,
10276
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/"
10277
+ },
10278
+ {
10279
+ "id": "task04-36-seed0-formal",
10280
+ "task_key": "task04/36",
10281
+ "family": "task04",
10282
+ "slot": "36",
10283
+ "seed": 0,
10284
+ "episode": 1,
10285
+ "phase": "formal",
10286
+ "status": "completed",
10287
+ "success": true,
10288
+ "native_reward": 1.0,
10289
+ "valid": true,
10290
+ "execution": {
10291
+ "reason": null,
10292
+ "status": "finished"
10293
+ },
10294
+ "verdict": {
10295
+ "evidence_valid": true,
10296
+ "steps": 1071,
10297
+ "success": true,
10298
+ "termination": "success"
10299
+ },
10300
+ "steps": 1071,
10301
+ "simulation_time_s": null,
10302
+ "wall_time_s": 777.348758,
10303
+ "model": "gpt-6-astra",
10304
+ "effort": "high",
10305
+ "harness": "codex",
10306
+ "codex_version": "0.160.0",
10307
+ "native_instruction": "Pour the liquid from the bottle into the cup.",
10308
+ "instruction": "Pour nearly all the liquid into the cup with almost no spillage. Keep the bottle tilted over the cup until the flow stops and the liquid settles, then return it upright, leaving only a small residue inside.",
10309
+ "instruction_policy": "modified",
10310
+ "usage": {
10311
+ "accounting": "reported-responses",
10312
+ "audit_complete": true,
10313
+ "cache_hit_rate": 0.9768983350616229,
10314
+ "cache_reported_input_tokens": 2577693,
10315
+ "cache_write_input_tokens": 0,
10316
+ "cache_write_reported_input_tokens": 2577693,
10317
+ "cached_input_tokens": 2518144,
10318
+ "completed_turns": 1,
10319
+ "cost_usd": null,
10320
+ "failed_turns": 0,
10321
+ "input_tokens": 2577693,
10322
+ "known_cache_write_input_tokens": 0,
10323
+ "known_cached_input_tokens": 2518144,
10324
+ "known_input_tokens": 2577693,
10325
+ "known_output_tokens": 15029,
10326
+ "known_reasoning_output_tokens": 6246,
10327
+ "output_tokens": 15029,
10328
+ "reasoning_output_tokens": 6246,
10329
+ "reasoning_reported_output_tokens": 15029,
10330
+ "reported_responses": {
10331
+ "cache_reported_input_tokens": 61,
10332
+ "cache_write_input_tokens": 61,
10333
+ "cache_write_reported_input_tokens": 61,
10334
+ "cached_input_tokens": 61,
10335
+ "input_tokens": 61,
10336
+ "output_tokens": 61,
10337
+ "reasoning_output_tokens": 61,
10338
+ "reasoning_reported_output_tokens": 61
10339
+ },
10340
+ "response_count": 61,
10341
+ "response_ids_complete": true,
10342
+ "schema": "rlebench/token-usage/1",
10343
+ "source": "Codex token_usage_record per response",
10344
+ "uncached_input_tokens": 59549,
10345
+ "unidentified_usage_records": 0
10346
+ },
10347
+ "call_activity": {
10348
+ "model_tool_calls": 60,
10349
+ "model_tool_calls_by_name": {
10350
+ "exec": 60
10351
+ },
10352
+ "nested_python_tool_invocations": null,
10353
+ "python_device_rpc_attempts": null,
10354
+ "python_device_rpc_attempts_by_action": null,
10355
+ "python_device_rpc_errors": null,
10356
+ "python_instrumented_model_tool_calls": null,
10357
+ "python_tool_invocations": null,
10358
+ "python_tool_invocations_by_origin": null,
10359
+ "schema": "rlebench/call-activity/1",
10360
+ "source": "Codex native sessions"
10361
+ },
10362
+ "media": {
10363
+ "passed": true,
10364
+ "width": 2880,
10365
+ "height": 720,
10366
+ "duration_s": 10.7,
10367
+ "speed": 4,
10368
+ "source_fps": 10,
10369
+ "output_fps": 20,
10370
+ "recording": {
10371
+ "accepted_samples": 430,
10372
+ "captured_samples": 430,
10373
+ "clock": "simulation",
10374
+ "dropped_samples": 0,
10375
+ "encoded_frames": 429,
10376
+ "end_time_s": 42.839999999999264,
10377
+ "error": null,
10378
+ "experimental": true,
10379
+ "fps": 10,
10380
+ "received_samples": 430,
10381
+ "schema": "roboenv/recording/1",
10382
+ "state": "closed",
10383
+ "status": "complete",
10384
+ "views": [
10385
+ {
10386
+ "fov_y": 45.0,
10387
+ "height": 720,
10388
+ "name": "third_person",
10389
+ "pose": null,
10390
+ "source": "third_person",
10391
+ "width": 960
10392
+ },
10393
+ {
10394
+ "fov_y": 45.0,
10395
+ "height": 720,
10396
+ "name": "left_wrist",
10397
+ "pose": null,
10398
+ "source": "left_wrist",
10399
+ "width": 960
10400
+ },
10401
+ {
10402
+ "fov_y": 45.0,
10403
+ "height": 720,
10404
+ "name": "right_wrist",
10405
+ "pose": null,
10406
+ "source": "right_wrist",
10407
+ "width": 960
10408
+ }
10409
+ ]
10410
+ },
10411
+ "view_names": [
10412
+ "third_person",
10413
+ "left_wrist",
10414
+ "right_wrist"
10415
+ ],
10416
+ "sha256": "0fe3ccbc26c9d14bc92ead5c4bcc79adbd0ceb97922b07e30730ea9a629065c1",
10417
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
10418
+ },
10419
+ "analysis": {
10420
+ "outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
10421
+ "duration": "\u4f7f\u7528 1,071 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
10422
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
10423
+ },
10424
+ "provenance": {
10425
+ "sources": {
10426
+ "RLE-Bench-inhouse": {
10427
+ "build_inputs": [
10428
+ "pyproject.toml",
10429
+ "src",
10430
+ "tasks",
10431
+ "README.md",
10432
+ "Makefile",
10433
+ "tests",
10434
+ "docs"
10435
+ ],
10436
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
10437
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
10438
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
10439
+ },
10440
+ "RoboEnv": {
10441
+ "build_inputs": [
10442
+ "pyproject.toml",
10443
+ "README.md",
10444
+ "src",
10445
+ "runtime/pyproject.toml",
10446
+ "runtime/README.md",
10447
+ "runtime/src",
10448
+ "runtime/environments.json",
10449
+ "runtime/locks",
10450
+ "catalog",
10451
+ "upstreams.lock.json",
10452
+ "third_party/patches",
10453
+ "docs/validation"
10454
+ ],
10455
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
10456
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
10457
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
10458
+ }
10459
+ },
10460
+ "job": "36-robodojo-pour-liquid-into-cup-codex-seed0-attempt01",
10461
+ "attempt": 1,
10462
+ "harness": "stock Codex CLI",
10463
+ "codex_version": "0.160.0",
10464
+ "model": "gpt-6-astra",
10465
+ "effort": "high",
10466
+ "service_tier": "default",
10467
+ "fresh_session": true,
10468
+ "source_jobs": [],
10469
+ "resume_trajectory": false,
10470
+ "imported_skills": [],
10471
+ "automatic_harbor_retries": 0,
10472
+ "request_policy": {
10473
+ "max_request_retries": 50,
10474
+ "configuration": "explicit retry50 SSE",
10475
+ "usage_accounting": "reported-responses"
10476
+ },
10477
+ "classification": "formal",
10478
+ "measured_images": {
10479
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
10480
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
10481
+ },
10482
+ "session_original_sha256": "7e5d9d89a62d36b2acc0686d0c45410afecd3f4a6a11d4909cf399edf099191d",
10483
+ "protocol_sha256": "b149476eb7b742f99fdc2849cf33dae7045952a02b90dd86fc3756a18c01a07b"
10484
+ },
10485
+ "links": {
10486
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/agent/session.jsonl",
10487
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/agent/trajectory.json",
10488
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/agent/provider-usage.jsonl",
10489
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/transcript.json",
10490
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/verifier/episode.json",
10491
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/verifier/protocol.json",
10492
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/instructions.json",
10493
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/verifier/workspace.tar.gz",
10494
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/verifier/workspace.json",
10495
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
10496
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/evidence/episode/evidence/recording/manifest.json",
10497
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/usage.json",
10498
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/analysis.json",
10499
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/provenance.json",
10500
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/video.mp4",
10501
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/poster.jpg",
10502
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/media-validation.json"
10503
+ },
10504
+ "resources": [
10505
+ {
10506
+ "name": "tools/pour_control.py",
10507
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/resources/tools/pour_control.py",
10508
+ "kind": "Created during this episode; final workspace snapshot."
10509
+ },
10510
+ {
10511
+ "name": "tools/robot_control.py",
10512
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/resources/tools/robot_control.py",
10513
+ "kind": "Created during this episode; final workspace snapshot."
10514
+ },
10515
+ {
10516
+ "name": "memos/pouring.md",
10517
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/resources/memos/pouring.md",
10518
+ "kind": "Created during this episode; final workspace snapshot."
10519
+ }
10520
+ ],
10521
+ "session_counts": {
10522
+ "visible_events": 135,
10523
+ "observed_images": 30,
10524
+ "tool_errors": 4
10525
+ },
10526
+ "selected_for_formal_metrics": true,
10527
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/"
10528
+ },
10529
+ {
10530
+ "id": "task04-38-seed0-formal",
10531
+ "task_key": "task04/38",
10532
+ "family": "task04",
10533
+ "slot": "38",
10534
+ "seed": 0,
10535
+ "episode": 1,
10536
+ "phase": "formal",
10537
+ "status": "completed",
10538
+ "success": true,
10539
+ "native_reward": 1.0,
10540
+ "valid": true,
10541
+ "execution": {
10542
+ "reason": null,
10543
+ "status": "finished"
10544
+ },
10545
+ "verdict": {
10546
+ "evidence_valid": true,
10547
+ "steps": 952,
10548
+ "success": true,
10549
+ "termination": "success"
10550
+ },
10551
+ "steps": 952,
10552
+ "simulation_time_s": null,
10553
+ "wall_time_s": 408.460414,
10554
+ "model": "gpt-6-astra",
10555
+ "effort": "high",
10556
+ "harness": "codex",
10557
+ "codex_version": "0.160.0",
10558
+ "native_instruction": "Press the two red buttons the required number of times according to the number cards, then press the blue button to confirm.",
10559
+ "instruction": "Starting with the left red button, press and release each red button the number of times shown on its card, confirming that button's count with a press and release of the blue button before moving to the next. Return both arms to their starting poses when finished.",
10560
+ "instruction_policy": "modified",
10561
+ "usage": {
10562
+ "accounting": "reported-responses",
10563
+ "audit_complete": true,
10564
+ "cache_hit_rate": 0.9579166678796666,
10565
+ "cache_reported_input_tokens": 1030503,
10566
+ "cache_write_input_tokens": 0,
10567
+ "cache_write_reported_input_tokens": 1030503,
10568
+ "cached_input_tokens": 987136,
10569
+ "completed_turns": 1,
10570
+ "cost_usd": null,
10571
+ "failed_turns": 0,
10572
+ "input_tokens": 1030503,
10573
+ "known_cache_write_input_tokens": 0,
10574
+ "known_cached_input_tokens": 987136,
10575
+ "known_input_tokens": 1030503,
10576
+ "known_output_tokens": 6991,
10577
+ "known_reasoning_output_tokens": 2046,
10578
+ "output_tokens": 6991,
10579
+ "reasoning_output_tokens": 2046,
10580
+ "reasoning_reported_output_tokens": 6991,
10581
+ "reported_responses": {
10582
+ "cache_reported_input_tokens": 30,
10583
+ "cache_write_input_tokens": 30,
10584
+ "cache_write_reported_input_tokens": 30,
10585
+ "cached_input_tokens": 30,
10586
+ "input_tokens": 30,
10587
+ "output_tokens": 30,
10588
+ "reasoning_output_tokens": 30,
10589
+ "reasoning_reported_output_tokens": 30
10590
+ },
10591
+ "response_count": 30,
10592
+ "response_ids_complete": true,
10593
+ "schema": "rlebench/token-usage/1",
10594
+ "source": "Codex token_usage_record per response",
10595
+ "uncached_input_tokens": 43367,
10596
+ "unidentified_usage_records": 0
10597
+ },
10598
+ "call_activity": {
10599
+ "model_tool_calls": 29,
10600
+ "model_tool_calls_by_name": {
10601
+ "exec": 29
10602
+ },
10603
+ "nested_python_tool_invocations": null,
10604
+ "python_device_rpc_attempts": null,
10605
+ "python_device_rpc_attempts_by_action": null,
10606
+ "python_device_rpc_errors": null,
10607
+ "python_instrumented_model_tool_calls": null,
10608
+ "python_tool_invocations": null,
10609
+ "python_tool_invocations_by_origin": null,
10610
+ "schema": "rlebench/call-activity/1",
10611
+ "source": "Codex native sessions"
10612
+ },
10613
+ "media": {
10614
+ "passed": true,
10615
+ "width": 2880,
10616
+ "height": 720,
10617
+ "duration_s": 9.5,
10618
+ "speed": 4,
10619
+ "source_fps": 10,
10620
+ "output_fps": 20,
10621
+ "recording": {
10622
+ "accepted_samples": 382,
10623
+ "captured_samples": 382,
10624
+ "clock": "simulation",
10625
+ "dropped_samples": 0,
10626
+ "encoded_frames": 381,
10627
+ "end_time_s": 38.079999999999366,
10628
+ "error": null,
10629
+ "experimental": true,
10630
+ "fps": 10,
10631
+ "received_samples": 382,
10632
+ "schema": "roboenv/recording/1",
10633
+ "state": "closed",
10634
+ "status": "complete",
10635
+ "views": [
10636
+ {
10637
+ "fov_y": 45.0,
10638
+ "height": 720,
10639
+ "name": "third_person",
10640
+ "pose": null,
10641
+ "source": "third_person",
10642
+ "width": 960
10643
+ },
10644
+ {
10645
+ "fov_y": 45.0,
10646
+ "height": 720,
10647
+ "name": "left_wrist",
10648
+ "pose": null,
10649
+ "source": "left_wrist",
10650
+ "width": 960
10651
+ },
10652
+ {
10653
+ "fov_y": 45.0,
10654
+ "height": 720,
10655
+ "name": "right_wrist",
10656
+ "pose": null,
10657
+ "source": "right_wrist",
10658
+ "width": 960
10659
+ }
10660
+ ]
10661
+ },
10662
+ "view_names": [
10663
+ "third_person",
10664
+ "left_wrist",
10665
+ "right_wrist"
10666
+ ],
10667
+ "sha256": "1191bf611481831d762b89e5a3da004b3ca6c531987daeece4ae34cd7bceb989",
10668
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
10669
+ },
10670
+ "analysis": {
10671
+ "outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
10672
+ "duration": "\u4f7f\u7528 952 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
10673
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
10674
+ },
10675
+ "provenance": {
10676
+ "sources": {
10677
+ "RLE-Bench-inhouse": {
10678
+ "build_inputs": [
10679
+ "pyproject.toml",
10680
+ "src",
10681
+ "tasks",
10682
+ "README.md",
10683
+ "Makefile",
10684
+ "tests",
10685
+ "docs"
10686
+ ],
10687
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
10688
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
10689
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
10690
+ },
10691
+ "RoboEnv": {
10692
+ "build_inputs": [
10693
+ "pyproject.toml",
10694
+ "README.md",
10695
+ "src",
10696
+ "runtime/pyproject.toml",
10697
+ "runtime/README.md",
10698
+ "runtime/src",
10699
+ "runtime/environments.json",
10700
+ "runtime/locks",
10701
+ "catalog",
10702
+ "upstreams.lock.json",
10703
+ "third_party/patches",
10704
+ "docs/validation"
10705
+ ],
10706
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
10707
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
10708
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
10709
+ }
10710
+ },
10711
+ "job": "38-robodojo-press-by-number-codex-seed0-attempt01",
10712
+ "attempt": 1,
10713
+ "harness": "stock Codex CLI",
10714
+ "codex_version": "0.160.0",
10715
+ "model": "gpt-6-astra",
10716
+ "effort": "high",
10717
+ "service_tier": "default",
10718
+ "fresh_session": true,
10719
+ "source_jobs": [],
10720
+ "resume_trajectory": false,
10721
+ "imported_skills": [],
10722
+ "automatic_harbor_retries": 0,
10723
+ "request_policy": {
10724
+ "max_request_retries": 50,
10725
+ "configuration": "explicit retry50 SSE",
10726
+ "usage_accounting": "reported-responses"
10727
+ },
10728
+ "classification": "formal",
10729
+ "measured_images": {
10730
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
10731
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
10732
+ },
10733
+ "session_original_sha256": "2988988996fbda95ad102c6dcdd76e9e645c552b62bb03210caf033a154c43ee",
10734
+ "protocol_sha256": "9fce1006e083eccaf342031dee9ee536219916e20037c3b853f01379591895a3"
10735
+ },
10736
+ "links": {
10737
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/agent/session.jsonl",
10738
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/agent/trajectory.json",
10739
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/agent/provider-usage.jsonl",
10740
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/transcript.json",
10741
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/verifier/episode.json",
10742
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/verifier/protocol.json",
10743
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/instructions.json",
10744
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/verifier/workspace.tar.gz",
10745
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/verifier/workspace.json",
10746
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
10747
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/evidence/episode/evidence/recording/manifest.json",
10748
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/usage.json",
10749
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/analysis.json",
10750
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/provenance.json",
10751
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/video.mp4",
10752
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/poster.jpg",
10753
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/media-validation.json"
10754
+ },
10755
+ "resources": [
10756
+ {
10757
+ "name": "tools/press_sequence.py",
10758
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/resources/tools/press_sequence.py",
10759
+ "kind": "Created during this episode; final workspace snapshot."
10760
+ },
10761
+ {
10762
+ "name": "tools/robot_control.py",
10763
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/resources/tools/robot_control.py",
10764
+ "kind": "Created during this episode; final workspace snapshot."
10765
+ },
10766
+ {
10767
+ "name": "memos/robodojo.md",
10768
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/resources/memos/robodojo.md",
10769
+ "kind": "Created during this episode; final workspace snapshot."
10770
+ }
10771
+ ],
10772
+ "session_counts": {
10773
+ "visible_events": 70,
10774
+ "observed_images": 19,
10775
+ "tool_errors": 3
10776
+ },
10777
+ "selected_for_formal_metrics": true,
10778
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/"
10779
+ },
10780
+ {
10781
+ "id": "task04-39-seed0-formal",
10782
+ "task_key": "task04/39",
10783
+ "family": "task04",
10784
+ "slot": "39",
10785
+ "seed": 0,
10786
+ "episode": 1,
10787
+ "phase": "formal",
10788
+ "status": "completed",
10789
+ "success": false,
10790
+ "native_reward": 0.0,
10791
+ "valid": true,
10792
+ "execution": {
10793
+ "reason": null,
10794
+ "status": "finished"
10795
+ },
10796
+ "verdict": {
10797
+ "evidence_valid": true,
10798
+ "steps": 535,
10799
+ "success": false,
10800
+ "termination": "stopped"
10801
+ },
10802
+ "steps": 535,
10803
+ "simulation_time_s": null,
10804
+ "wall_time_s": 346.926941,
10805
+ "model": "gpt-6-astra",
10806
+ "effort": "high",
10807
+ "harness": "codex",
10808
+ "codex_version": "0.160.0",
10809
+ "native_instruction": "Push the T-shaped block to align it precisely with the gray T-shaped pad.",
10810
+ "instruction": "Slide the T-shaped block along the table until it neatly matches the gray T-shaped pad in position and orientation. Keep the block on the table and return both arms to their starting poses when finished.",
10811
+ "instruction_policy": "modified",
10812
+ "usage": {
10813
+ "accounting": "reported-responses",
10814
+ "audit_complete": true,
10815
+ "cache_hit_rate": 0.9539017898864306,
10816
+ "cache_reported_input_tokens": 787536,
10817
+ "cache_write_input_tokens": 0,
10818
+ "cache_write_reported_input_tokens": 787536,
10819
+ "cached_input_tokens": 751232,
10820
+ "completed_turns": 1,
10821
+ "cost_usd": null,
10822
+ "failed_turns": 0,
10823
+ "input_tokens": 787536,
10824
+ "known_cache_write_input_tokens": 0,
10825
+ "known_cached_input_tokens": 751232,
10826
+ "known_input_tokens": 787536,
10827
+ "known_output_tokens": 7374,
10828
+ "known_reasoning_output_tokens": 2132,
10829
+ "output_tokens": 7374,
10830
+ "reasoning_output_tokens": 2132,
10831
+ "reasoning_reported_output_tokens": 7374,
10832
+ "reported_responses": {
10833
+ "cache_reported_input_tokens": 25,
10834
+ "cache_write_input_tokens": 25,
10835
+ "cache_write_reported_input_tokens": 25,
10836
+ "cached_input_tokens": 25,
10837
+ "input_tokens": 25,
10838
+ "output_tokens": 25,
10839
+ "reasoning_output_tokens": 25,
10840
+ "reasoning_reported_output_tokens": 25
10841
+ },
10842
+ "response_count": 25,
10843
+ "response_ids_complete": true,
10844
+ "schema": "rlebench/token-usage/1",
10845
+ "source": "Codex token_usage_record per response",
10846
+ "uncached_input_tokens": 36304,
10847
+ "unidentified_usage_records": 0
10848
+ },
10849
+ "call_activity": {
10850
+ "model_tool_calls": 24,
10851
+ "model_tool_calls_by_name": {
10852
+ "exec": 24
10853
+ },
10854
+ "nested_python_tool_invocations": null,
10855
+ "python_device_rpc_attempts": null,
10856
+ "python_device_rpc_attempts_by_action": null,
10857
+ "python_device_rpc_errors": null,
10858
+ "python_instrumented_model_tool_calls": null,
10859
+ "python_tool_invocations": null,
10860
+ "python_tool_invocations_by_origin": null,
10861
+ "schema": "rlebench/call-activity/1",
10862
+ "source": "Codex native sessions"
10863
+ },
10864
+ "media": {
10865
+ "passed": true,
10866
+ "width": 2880,
10867
+ "height": 720,
10868
+ "duration_s": 5.35,
10869
+ "speed": 4,
10870
+ "source_fps": 10,
10871
+ "output_fps": 20,
10872
+ "recording": {
10873
+ "accepted_samples": 215,
10874
+ "captured_samples": 215,
10875
+ "clock": "simulation",
10876
+ "dropped_samples": 0,
10877
+ "encoded_frames": 215,
10878
+ "end_time_s": 21.39999999999972,
10879
+ "error": null,
10880
+ "experimental": true,
10881
+ "fps": 10,
10882
+ "received_samples": 215,
10883
+ "schema": "roboenv/recording/1",
10884
+ "state": "closed",
10885
+ "status": "complete",
10886
+ "views": [
10887
+ {
10888
+ "fov_y": 45.0,
10889
+ "height": 720,
10890
+ "name": "third_person",
10891
+ "pose": null,
10892
+ "source": "third_person",
10893
+ "width": 960
10894
+ },
10895
+ {
10896
+ "fov_y": 45.0,
10897
+ "height": 720,
10898
+ "name": "left_wrist",
10899
+ "pose": null,
10900
+ "source": "left_wrist",
10901
+ "width": 960
10902
+ },
10903
+ {
10904
+ "fov_y": 45.0,
10905
+ "height": 720,
10906
+ "name": "right_wrist",
10907
+ "pose": null,
10908
+ "source": "right_wrist",
10909
+ "width": 960
10910
+ }
10911
+ ]
10912
+ },
10913
+ "view_names": [
10914
+ "third_person",
10915
+ "left_wrist",
10916
+ "right_wrist"
10917
+ ],
10918
+ "sha256": "4351eae383c6bb677519793c915be8d479148b0156ac87ecfb962fb3768ebf93",
10919
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
10920
+ },
10921
+ "analysis": {
10922
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
10923
+ "duration": "\u4f7f\u7528 535 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
10924
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
10925
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
10926
+ },
10927
+ "provenance": {
10928
+ "sources": {
10929
+ "RLE-Bench-inhouse": {
10930
+ "build_inputs": [
10931
+ "pyproject.toml",
10932
+ "src",
10933
+ "tasks",
10934
+ "README.md",
10935
+ "Makefile",
10936
+ "tests",
10937
+ "docs"
10938
+ ],
10939
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
10940
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
10941
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
10942
+ },
10943
+ "RoboEnv": {
10944
+ "build_inputs": [
10945
+ "pyproject.toml",
10946
+ "README.md",
10947
+ "src",
10948
+ "runtime/pyproject.toml",
10949
+ "runtime/README.md",
10950
+ "runtime/src",
10951
+ "runtime/environments.json",
10952
+ "runtime/locks",
10953
+ "catalog",
10954
+ "upstreams.lock.json",
10955
+ "third_party/patches",
10956
+ "docs/validation"
10957
+ ],
10958
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
10959
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
10960
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
10961
+ }
10962
+ },
10963
+ "job": "39-robodojo-push-t-codex-seed0-attempt01",
10964
+ "attempt": 1,
10965
+ "harness": "stock Codex CLI",
10966
+ "codex_version": "0.160.0",
10967
+ "model": "gpt-6-astra",
10968
+ "effort": "high",
10969
+ "service_tier": "default",
10970
+ "fresh_session": true,
10971
+ "source_jobs": [],
10972
+ "resume_trajectory": false,
10973
+ "imported_skills": [],
10974
+ "automatic_harbor_retries": 0,
10975
+ "request_policy": {
10976
+ "max_request_retries": 50,
10977
+ "configuration": "explicit retry50 SSE",
10978
+ "usage_accounting": "reported-responses"
10979
+ },
10980
+ "classification": "formal",
10981
+ "measured_images": {
10982
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
10983
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
10984
+ },
10985
+ "session_original_sha256": "494f4f61f1bd5d3072e7966059e167381f188a7a55853a3c66839da85925e26d",
10986
+ "protocol_sha256": "67d36ac75c05a3dace78b479283200ce78a0d5fa330bd3deb44a3033a0feeec3"
10987
+ },
10988
+ "links": {
10989
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/agent/session.jsonl",
10990
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/agent/trajectory.json",
10991
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/agent/provider-usage.jsonl",
10992
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/transcript.json",
10993
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/verifier/episode.json",
10994
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/verifier/protocol.json",
10995
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/instructions.json",
10996
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/verifier/workspace.tar.gz",
10997
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/verifier/workspace.json",
10998
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
10999
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/evidence/episode/evidence/recording/manifest.json",
11000
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/usage.json",
11001
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/analysis.json",
11002
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/provenance.json",
11003
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/video.mp4",
11004
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/poster.jpg",
11005
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/media-validation.json"
11006
+ },
11007
+ "resources": [
11008
+ {
11009
+ "name": "tools/arx_control.py",
11010
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/resources/tools/arx_control.py",
11011
+ "kind": "Created during this episode; final workspace snapshot."
11012
+ },
11013
+ {
11014
+ "name": "memos/robodojo_push_t.md",
11015
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/resources/memos/robodojo_push_t.md",
11016
+ "kind": "Created during this episode; final workspace snapshot."
11017
+ }
11018
+ ],
11019
+ "session_counts": {
11020
+ "visible_events": 59,
11021
+ "observed_images": 16,
11022
+ "tool_errors": 2
11023
+ },
11024
+ "selected_for_formal_metrics": true,
11025
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/"
11026
  }
11027
  ],
11028
  "failure_review": {
episodes.csv CHANGED
@@ -14,7 +14,7 @@ task04/12,robodojo/arrange-largest-number,completed,finished,True,2040,2403089,2
14
  task04/14,robodojo/build-tower,completed,finished,True,3359,4541614,4394368,24568
15
  task04/15,robodojo/classify-objects,completed,finished,True,2596,4079082,3972608,15725
16
  task04/16,robodojo/fill-egg-holder,completed,finished,True,3772,10433608,10253056,29150
17
- task04/17,robodojo/fill-pen-holder,pending,running,,,,,
18
  task04/18,robodojo/fold-clothes,completed,finished,True,2826,4706237,4633088,20162
19
  task04/20,robodojo/general-pickup,completed,finished,True,260,890742,818688,5727
20
  task04/21,robodojo/hang-mugs,completed,finished,True,3649,8564328,8449664,31426
@@ -23,21 +23,21 @@ task04/24,robodojo/insert-key,pending,running,,,,,
23
  task04/25,robodojo/make-kong,completed,finished,False,2476,4991204,4908544,24592
24
  task04/27,robodojo/match-and-pick-from-conveyor,completed,finished,True,484,1059308,1022976,5961
25
  task04/28,robodojo/organize-table,completed,finished,False,4143,6791820,6688768,30892
26
- task04/29,robodojo/pack-objects-into-box,pending,running,,,,,
27
  task04/31,robodojo/pick-from-conveyor-by-image,completed,finished,False,863,3797181,3703168,23800
28
  task04/32,robodojo/play-tic-tac-toe,completed,finished,True,2665,3137952,3077888,12655
29
  task04/33,robodojo/plug-in-charger,completed,finished,True,964,2129540,2076032,9894
30
  task04/34,robodojo/pour-balls-into-vase,pending,running,,,,,
31
- task04/35,robodojo/pour-by-language,pending,running,,,,,
32
- task04/36,robodojo/pour-liquid-into-cup,pending,running,,,,,
33
- task04/38,robodojo/press-by-number,pending,running,,,,,
34
- task04/39,robodojo/push-t,pending,queued,,,,,
35
- task04/41,robodojo/put-bottles-into-dustbin,pending,queued,,,,,
36
- task04/42,robodojo/solve-equation,pending,queued,,,,,
37
- task04/43,robodojo/sort-nesting-dolls-by-size,pending,queued,,,,,
38
- task04/45,robodojo/stack-blocks,pending,queued,,,,,
39
- task04/46,robodojo/stack-blocks-by-language,pending,queued,,,,,
40
- task04/48,robodojo/stack-bowls,pending,queued,,,,,
41
  task04/51,robodojo/swap-t,pending,queued,,,,,
42
  task04/52,robodojo/swap-blocks,pending,queued,,,,,
43
  task04/53,robodojo/sweep-blocks,pending,queued,,,,,
 
14
  task04/14,robodojo/build-tower,completed,finished,True,3359,4541614,4394368,24568
15
  task04/15,robodojo/classify-objects,completed,finished,True,2596,4079082,3972608,15725
16
  task04/16,robodojo/fill-egg-holder,completed,finished,True,3772,10433608,10253056,29150
17
+ task04/17,robodojo/fill-pen-holder,completed,finished,False,7440,29201838,28982912,80788
18
  task04/18,robodojo/fold-clothes,completed,finished,True,2826,4706237,4633088,20162
19
  task04/20,robodojo/general-pickup,completed,finished,True,260,890742,818688,5727
20
  task04/21,robodojo/hang-mugs,completed,finished,True,3649,8564328,8449664,31426
 
23
  task04/25,robodojo/make-kong,completed,finished,False,2476,4991204,4908544,24592
24
  task04/27,robodojo/match-and-pick-from-conveyor,completed,finished,True,484,1059308,1022976,5961
25
  task04/28,robodojo/organize-table,completed,finished,False,4143,6791820,6688768,30892
26
+ task04/29,robodojo/pack-objects-into-box,completed,finished,True,6433,12341824,12166656,39288
27
  task04/31,robodojo/pick-from-conveyor-by-image,completed,finished,False,863,3797181,3703168,23800
28
  task04/32,robodojo/play-tic-tac-toe,completed,finished,True,2665,3137952,3077888,12655
29
  task04/33,robodojo/plug-in-charger,completed,finished,True,964,2129540,2076032,9894
30
  task04/34,robodojo/pour-balls-into-vase,pending,running,,,,,
31
+ task04/35,robodojo/pour-by-language,completed,finished,False,1616,2677076,2588032,11360
32
+ task04/36,robodojo/pour-liquid-into-cup,completed,finished,True,1071,2577693,2518144,15029
33
+ task04/38,robodojo/press-by-number,completed,finished,True,952,1030503,987136,6991
34
+ task04/39,robodojo/push-t,completed,finished,False,535,787536,751232,7374
35
+ task04/41,robodojo/put-bottles-into-dustbin,pending,running,,,,,
36
+ task04/42,robodojo/solve-equation,pending,finished,,,,,
37
+ task04/43,robodojo/sort-nesting-dolls-by-size,pending,running,,,,,
38
+ task04/45,robodojo/stack-blocks,pending,running,,,,,
39
+ task04/46,robodojo/stack-blocks-by-language,pending,running,,,,,
40
+ task04/48,robodojo/stack-bowls,pending,running,,,,,
41
  task04/51,robodojo/swap-t,pending,queued,,,,,
42
  task04/52,robodojo/swap-blocks,pending,queued,,,,,
43
  task04/53,robodojo/sweep-blocks,pending,queued,,,,,
episodes.json CHANGED
@@ -3231,6 +3231,253 @@
3231
  "selected_for_formal_metrics": true,
3232
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/16/seed-0/"
3233
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3234
  {
3235
  "id": "task04-18-seed0-formal",
3236
  "task_key": "task04/18",
@@ -4956,16 +5203,16 @@
4956
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/28/seed-0/"
4957
  },
4958
  {
4959
- "id": "task04-31-seed0-formal",
4960
- "task_key": "task04/31",
4961
  "family": "task04",
4962
- "slot": "31",
4963
  "seed": 0,
4964
  "episode": 1,
4965
  "phase": "formal",
4966
  "status": "completed",
4967
- "success": false,
4968
- "native_reward": 0.0,
4969
  "valid": true,
4970
  "execution": {
4971
  "reason": null,
@@ -4973,61 +5220,61 @@
4973
  },
4974
  "verdict": {
4975
  "evidence_valid": true,
4976
- "steps": 863,
4977
- "success": false,
4978
- "termination": "stopped"
4979
  },
4980
- "steps": 863,
4981
  "simulation_time_s": null,
4982
- "wall_time_s": 1296.406492,
4983
  "model": "gpt-6-astra",
4984
  "effort": "high",
4985
  "harness": "codex",
4986
  "codex_version": "0.160.0",
4987
- "native_instruction": "Lift the basket more than 8 cm, identify the target object on the conveyor according to the image on the board, pick it up, and place it into the basket.",
4988
- "instruction": "Identify the target object on the conveyor from the image on the board. Lift and hold the basket more than 8 cm above its starting height, then pick up the matching object and place it down inside the basket. Keep the basket raised, with the object resting at its bottom and more than 8 cm above its original height.",
4989
  "instruction_policy": "modified",
4990
  "usage": {
4991
  "accounting": "reported-responses",
4992
  "audit_complete": true,
4993
- "cache_hit_rate": 0.9752413698477898,
4994
- "cache_reported_input_tokens": 3797181,
4995
  "cache_write_input_tokens": 0,
4996
- "cache_write_reported_input_tokens": 3797181,
4997
- "cached_input_tokens": 3703168,
4998
  "completed_turns": 1,
4999
  "cost_usd": null,
5000
  "failed_turns": 0,
5001
- "input_tokens": 3797181,
5002
  "known_cache_write_input_tokens": 0,
5003
- "known_cached_input_tokens": 3703168,
5004
- "known_input_tokens": 3797181,
5005
- "known_output_tokens": 23800,
5006
- "known_reasoning_output_tokens": 11319,
5007
- "output_tokens": 23800,
5008
- "reasoning_output_tokens": 11319,
5009
- "reasoning_reported_output_tokens": 23800,
5010
  "reported_responses": {
5011
- "cache_reported_input_tokens": 65,
5012
- "cache_write_input_tokens": 65,
5013
- "cache_write_reported_input_tokens": 65,
5014
- "cached_input_tokens": 65,
5015
- "input_tokens": 65,
5016
- "output_tokens": 65,
5017
- "reasoning_output_tokens": 65,
5018
- "reasoning_reported_output_tokens": 65
5019
- },
5020
- "response_count": 65,
5021
  "response_ids_complete": true,
5022
  "schema": "rlebench/token-usage/1",
5023
  "source": "Codex token_usage_record per response",
5024
- "uncached_input_tokens": 94013,
5025
  "unidentified_usage_records": 0
5026
  },
5027
  "call_activity": {
5028
- "model_tool_calls": 64,
5029
  "model_tool_calls_by_name": {
5030
- "exec": 64
5031
  },
5032
  "nested_python_tool_invocations": null,
5033
  "python_device_rpc_attempts": null,
@@ -5043,21 +5290,21 @@
5043
  "passed": true,
5044
  "width": 2880,
5045
  "height": 720,
5046
- "duration_s": 8.65,
5047
  "speed": 4,
5048
  "source_fps": 10,
5049
  "output_fps": 20,
5050
  "recording": {
5051
- "accepted_samples": 346,
5052
- "captured_samples": 346,
5053
  "clock": "simulation",
5054
  "dropped_samples": 0,
5055
- "encoded_frames": 346,
5056
- "end_time_s": 34.51999999999944,
5057
  "error": null,
5058
  "experimental": true,
5059
  "fps": 10,
5060
- "received_samples": 346,
5061
  "schema": "roboenv/recording/1",
5062
  "state": "closed",
5063
  "status": "complete",
@@ -5093,14 +5340,13 @@
5093
  "left_wrist",
5094
  "right_wrist"
5095
  ],
5096
- "sha256": "885151826ea1b79abd7772d71bb25c1fe86d2f09dcfddd47938e6381c5fbe6de",
5097
  "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
5098
  },
5099
  "analysis": {
5100
- "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
5101
- "duration": "\u4f7f\u7528 863 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
5102
- "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
5103
- "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
5104
  },
5105
  "provenance": {
5106
  "sources": {
@@ -5138,7 +5384,7 @@
5138
  "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
5139
  }
5140
  },
5141
- "job": "31-robodojo-pick-from-conveyor-by-image-codex-seed0-attempt01",
5142
  "attempt": 1,
5143
  "harness": "stock Codex CLI",
5144
  "codex_version": "0.160.0",
@@ -5160,42 +5406,289 @@
5160
  "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
5161
  "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
5162
  },
5163
- "session_original_sha256": "aa769c4c945fa18e304b7d4d5d8c9c603afb75f438dc03b941fd4b9eb08f7fe0",
5164
- "protocol_sha256": "9ce80cb246bf47f91aa2889530d7ef56aa5498f223bbd77c69f483653254c63d"
5165
  },
5166
  "links": {
5167
- "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/agent/session.jsonl",
5168
- "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/agent/trajectory.json",
5169
- "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/agent/provider-usage.jsonl",
5170
- "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/transcript.json",
5171
- "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/verifier/episode.json",
5172
- "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/verifier/protocol.json",
5173
- "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/instructions.json",
5174
- "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/verifier/workspace.tar.gz",
5175
- "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/verifier/workspace.json",
5176
- "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
5177
- "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/evidence/episode/evidence/recording/manifest.json",
5178
- "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/usage.json",
5179
- "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/analysis.json",
5180
- "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/provenance.json",
5181
- "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/video.mp4",
5182
- "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/poster.jpg",
5183
- "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/media-validation.json"
5184
  },
5185
  "resources": [
5186
  {
5187
- "name": "tools/robot.py",
5188
- "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/resources/tools/robot.py",
5189
  "kind": "Created during this episode; final workspace snapshot."
5190
  },
5191
  {
5192
  "name": "memos/robodojo.md",
5193
- "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/resources/memos/robodojo.md",
5194
  "kind": "Created during this episode; final workspace snapshot."
5195
  }
5196
  ],
5197
  "session_counts": {
5198
- "visible_events": 141,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5199
  "observed_images": 78,
5200
  "tool_errors": 7
5201
  },
@@ -5698,5 +6191,1001 @@
5698
  },
5699
  "selected_for_formal_metrics": true,
5700
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/33/seed-0/"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5701
  }
5702
  ]
 
3231
  "selected_for_formal_metrics": true,
3232
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/16/seed-0/"
3233
  },
3234
+ {
3235
+ "id": "task04-17-seed0-formal",
3236
+ "task_key": "task04/17",
3237
+ "family": "task04",
3238
+ "slot": "17",
3239
+ "seed": 0,
3240
+ "episode": 1,
3241
+ "phase": "formal",
3242
+ "status": "completed",
3243
+ "success": false,
3244
+ "native_reward": 0.25,
3245
+ "valid": true,
3246
+ "execution": {
3247
+ "reason": null,
3248
+ "status": "finished"
3249
+ },
3250
+ "verdict": {
3251
+ "evidence_valid": true,
3252
+ "steps": 7440,
3253
+ "success": false,
3254
+ "termination": "stopped"
3255
+ },
3256
+ "steps": 7440,
3257
+ "simulation_time_s": null,
3258
+ "wall_time_s": 5845.33904,
3259
+ "model": "gpt-6-astra",
3260
+ "effort": "high",
3261
+ "harness": "codex",
3262
+ "codex_version": "0.160.0",
3263
+ "native_instruction": "Hold the pen holder with one hand, place all pens into it with the other hand, then put it back down.",
3264
+ "instruction": "Hold the pen holder with one hand and insert all the pens with the other, writing ends down and fully seated inside. Put the holder down upright, release it, and return both arms to their starting poses.",
3265
+ "instruction_policy": "modified",
3266
+ "usage": {
3267
+ "accounting": "reported-responses",
3268
+ "audit_complete": true,
3269
+ "cache_hit_rate": 0.992503006146394,
3270
+ "cache_reported_input_tokens": 29201838,
3271
+ "cache_write_input_tokens": 0,
3272
+ "cache_write_reported_input_tokens": 29201838,
3273
+ "cached_input_tokens": 28982912,
3274
+ "completed_turns": 1,
3275
+ "cost_usd": null,
3276
+ "failed_turns": 0,
3277
+ "input_tokens": 29201838,
3278
+ "known_cache_write_input_tokens": 0,
3279
+ "known_cached_input_tokens": 28982912,
3280
+ "known_input_tokens": 29201838,
3281
+ "known_output_tokens": 80788,
3282
+ "known_reasoning_output_tokens": 48714,
3283
+ "output_tokens": 80788,
3284
+ "reasoning_output_tokens": 48714,
3285
+ "reasoning_reported_output_tokens": 80788,
3286
+ "reported_responses": {
3287
+ "cache_reported_input_tokens": 289,
3288
+ "cache_write_input_tokens": 289,
3289
+ "cache_write_reported_input_tokens": 289,
3290
+ "cached_input_tokens": 289,
3291
+ "input_tokens": 289,
3292
+ "output_tokens": 289,
3293
+ "reasoning_output_tokens": 289,
3294
+ "reasoning_reported_output_tokens": 289
3295
+ },
3296
+ "response_count": 289,
3297
+ "response_ids_complete": true,
3298
+ "schema": "rlebench/token-usage/1",
3299
+ "source": "Codex token_usage_record per response",
3300
+ "uncached_input_tokens": 218926,
3301
+ "unidentified_usage_records": 0
3302
+ },
3303
+ "call_activity": {
3304
+ "model_tool_calls": 288,
3305
+ "model_tool_calls_by_name": {
3306
+ "exec": 288
3307
+ },
3308
+ "nested_python_tool_invocations": null,
3309
+ "python_device_rpc_attempts": null,
3310
+ "python_device_rpc_attempts_by_action": null,
3311
+ "python_device_rpc_errors": null,
3312
+ "python_instrumented_model_tool_calls": null,
3313
+ "python_tool_invocations": null,
3314
+ "python_tool_invocations_by_origin": null,
3315
+ "schema": "rlebench/call-activity/1",
3316
+ "source": "Codex native sessions"
3317
+ },
3318
+ "media": {
3319
+ "passed": true,
3320
+ "width": 2880,
3321
+ "height": 720,
3322
+ "duration_s": 74.4,
3323
+ "speed": 4,
3324
+ "source_fps": 10,
3325
+ "output_fps": 20,
3326
+ "recording": {
3327
+ "accepted_samples": 2977,
3328
+ "captured_samples": 2977,
3329
+ "clock": "simulation",
3330
+ "dropped_samples": 0,
3331
+ "encoded_frames": 2977,
3332
+ "end_time_s": 297.6000000000046,
3333
+ "error": null,
3334
+ "experimental": true,
3335
+ "fps": 10,
3336
+ "received_samples": 2977,
3337
+ "schema": "roboenv/recording/1",
3338
+ "state": "closed",
3339
+ "status": "complete",
3340
+ "views": [
3341
+ {
3342
+ "fov_y": 45.0,
3343
+ "height": 720,
3344
+ "name": "third_person",
3345
+ "pose": null,
3346
+ "source": "third_person",
3347
+ "width": 960
3348
+ },
3349
+ {
3350
+ "fov_y": 45.0,
3351
+ "height": 720,
3352
+ "name": "left_wrist",
3353
+ "pose": null,
3354
+ "source": "left_wrist",
3355
+ "width": 960
3356
+ },
3357
+ {
3358
+ "fov_y": 45.0,
3359
+ "height": 720,
3360
+ "name": "right_wrist",
3361
+ "pose": null,
3362
+ "source": "right_wrist",
3363
+ "width": 960
3364
+ }
3365
+ ]
3366
+ },
3367
+ "view_names": [
3368
+ "third_person",
3369
+ "left_wrist",
3370
+ "right_wrist"
3371
+ ],
3372
+ "sha256": "f3c9626399fbbdd5e6f15a39f9e9967e3ac90b6d1c85b0ba822ae16f36465301",
3373
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
3374
+ },
3375
+ "analysis": {
3376
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
3377
+ "duration": "\u4f7f\u7528 7,440 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
3378
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
3379
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
3380
+ },
3381
+ "provenance": {
3382
+ "sources": {
3383
+ "RLE-Bench-inhouse": {
3384
+ "build_inputs": [
3385
+ "pyproject.toml",
3386
+ "src",
3387
+ "tasks",
3388
+ "README.md",
3389
+ "Makefile",
3390
+ "tests",
3391
+ "docs"
3392
+ ],
3393
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
3394
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
3395
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
3396
+ },
3397
+ "RoboEnv": {
3398
+ "build_inputs": [
3399
+ "pyproject.toml",
3400
+ "README.md",
3401
+ "src",
3402
+ "runtime/pyproject.toml",
3403
+ "runtime/README.md",
3404
+ "runtime/src",
3405
+ "runtime/environments.json",
3406
+ "runtime/locks",
3407
+ "catalog",
3408
+ "upstreams.lock.json",
3409
+ "third_party/patches",
3410
+ "docs/validation"
3411
+ ],
3412
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
3413
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
3414
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
3415
+ }
3416
+ },
3417
+ "job": "17-robodojo-fill-pen-holder-codex-seed0-attempt01",
3418
+ "attempt": 1,
3419
+ "harness": "stock Codex CLI",
3420
+ "codex_version": "0.160.0",
3421
+ "model": "gpt-6-astra",
3422
+ "effort": "high",
3423
+ "service_tier": "default",
3424
+ "fresh_session": true,
3425
+ "source_jobs": [],
3426
+ "resume_trajectory": false,
3427
+ "imported_skills": [],
3428
+ "automatic_harbor_retries": 0,
3429
+ "request_policy": {
3430
+ "max_request_retries": 50,
3431
+ "configuration": "explicit retry50 SSE",
3432
+ "usage_accounting": "reported-responses"
3433
+ },
3434
+ "classification": "formal",
3435
+ "measured_images": {
3436
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
3437
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
3438
+ },
3439
+ "session_original_sha256": "4de9c3c264f52832303def2e45505f2814b7fc5a0fe89b7ca676fd6d8c693b71",
3440
+ "protocol_sha256": "8db05ad726248326912895583e4fe0eb492747d9f7702eb2b524abfb8da02a0a"
3441
+ },
3442
+ "links": {
3443
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/agent/session.jsonl",
3444
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/agent/trajectory.json",
3445
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/agent/provider-usage.jsonl",
3446
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/transcript.json",
3447
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/verifier/episode.json",
3448
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/verifier/protocol.json",
3449
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/instructions.json",
3450
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/verifier/workspace.tar.gz",
3451
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/verifier/workspace.json",
3452
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
3453
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/native/evidence/episode/evidence/recording/manifest.json",
3454
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/usage.json",
3455
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/analysis.json",
3456
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/provenance.json",
3457
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/video.mp4",
3458
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/poster.jpg",
3459
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/media-validation.json"
3460
+ },
3461
+ "resources": [
3462
+ {
3463
+ "name": "tools/robot.py",
3464
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/resources/tools/robot.py",
3465
+ "kind": "Created during this episode; final workspace snapshot."
3466
+ },
3467
+ {
3468
+ "name": "memos/robodojo-manipulation.md",
3469
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/resources/memos/robodojo-manipulation.md",
3470
+ "kind": "Created during this episode; final workspace snapshot."
3471
+ }
3472
+ ],
3473
+ "session_counts": {
3474
+ "visible_events": 608,
3475
+ "observed_images": 123,
3476
+ "tool_errors": 20
3477
+ },
3478
+ "selected_for_formal_metrics": true,
3479
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/17/seed-0/"
3480
+ },
3481
  {
3482
  "id": "task04-18-seed0-formal",
3483
  "task_key": "task04/18",
 
5203
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/28/seed-0/"
5204
  },
5205
  {
5206
+ "id": "task04-29-seed0-formal",
5207
+ "task_key": "task04/29",
5208
  "family": "task04",
5209
+ "slot": "29",
5210
  "seed": 0,
5211
  "episode": 1,
5212
  "phase": "formal",
5213
  "status": "completed",
5214
+ "success": true,
5215
+ "native_reward": 1.0,
5216
  "valid": true,
5217
  "execution": {
5218
  "reason": null,
 
5220
  },
5221
  "verdict": {
5222
  "evidence_valid": true,
5223
+ "steps": 6433,
5224
+ "success": true,
5225
+ "termination": "success"
5226
  },
5227
+ "steps": 6433,
5228
  "simulation_time_s": null,
5229
+ "wall_time_s": 2657.323509,
5230
  "model": "gpt-6-astra",
5231
  "effort": "high",
5232
  "harness": "codex",
5233
  "codex_version": "0.160.0",
5234
+ "native_instruction": "Place all the objects into the box with their front sides facing left.",
5235
+ "instruction": "Place all four objects inside the bottom of the box, with their front sides facing left from the robot's viewpoint and their lengths aligned along the box. Keep the box upright with its long sides running left to right across the table. Release the objects and return both arms to their starting poses.",
5236
  "instruction_policy": "modified",
5237
  "usage": {
5238
  "accounting": "reported-responses",
5239
  "audit_complete": true,
5240
+ "cache_hit_rate": 0.985806960138145,
5241
+ "cache_reported_input_tokens": 12341824,
5242
  "cache_write_input_tokens": 0,
5243
+ "cache_write_reported_input_tokens": 12341824,
5244
+ "cached_input_tokens": 12166656,
5245
  "completed_turns": 1,
5246
  "cost_usd": null,
5247
  "failed_turns": 0,
5248
+ "input_tokens": 12341824,
5249
  "known_cache_write_input_tokens": 0,
5250
+ "known_cached_input_tokens": 12166656,
5251
+ "known_input_tokens": 12341824,
5252
+ "known_output_tokens": 39288,
5253
+ "known_reasoning_output_tokens": 20764,
5254
+ "output_tokens": 39288,
5255
+ "reasoning_output_tokens": 20764,
5256
+ "reasoning_reported_output_tokens": 39288,
5257
  "reported_responses": {
5258
+ "cache_reported_input_tokens": 172,
5259
+ "cache_write_input_tokens": 172,
5260
+ "cache_write_reported_input_tokens": 172,
5261
+ "cached_input_tokens": 172,
5262
+ "input_tokens": 172,
5263
+ "output_tokens": 172,
5264
+ "reasoning_output_tokens": 172,
5265
+ "reasoning_reported_output_tokens": 172
5266
+ },
5267
+ "response_count": 172,
5268
  "response_ids_complete": true,
5269
  "schema": "rlebench/token-usage/1",
5270
  "source": "Codex token_usage_record per response",
5271
+ "uncached_input_tokens": 175168,
5272
  "unidentified_usage_records": 0
5273
  },
5274
  "call_activity": {
5275
+ "model_tool_calls": 171,
5276
  "model_tool_calls_by_name": {
5277
+ "exec": 171
5278
  },
5279
  "nested_python_tool_invocations": null,
5280
  "python_device_rpc_attempts": null,
 
5290
  "passed": true,
5291
  "width": 2880,
5292
  "height": 720,
5293
+ "duration_s": 64.35,
5294
  "speed": 4,
5295
  "source_fps": 10,
5296
  "output_fps": 20,
5297
  "recording": {
5298
+ "accepted_samples": 2574,
5299
+ "captured_samples": 2574,
5300
  "clock": "simulation",
5301
  "dropped_samples": 0,
5302
+ "encoded_frames": 2574,
5303
+ "end_time_s": 257.319999999984,
5304
  "error": null,
5305
  "experimental": true,
5306
  "fps": 10,
5307
+ "received_samples": 2574,
5308
  "schema": "roboenv/recording/1",
5309
  "state": "closed",
5310
  "status": "complete",
 
5340
  "left_wrist",
5341
  "right_wrist"
5342
  ],
5343
+ "sha256": "8b4d310b6dc63abb0100121b03fe18a0b07a44a2a63f6df843064e356fb18898",
5344
  "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
5345
  },
5346
  "analysis": {
5347
+ "outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
5348
+ "duration": "\u4f7f\u7528 6,433 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
5349
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
 
5350
  },
5351
  "provenance": {
5352
  "sources": {
 
5384
  "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
5385
  }
5386
  },
5387
+ "job": "29-robodojo-pack-objects-into-box-codex-seed0-attempt01",
5388
  "attempt": 1,
5389
  "harness": "stock Codex CLI",
5390
  "codex_version": "0.160.0",
 
5406
  "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
5407
  "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
5408
  },
5409
+ "session_original_sha256": "a310e18ccc4536315f21afc01221789490558fff6e3c316dc38c7fafeeaca76b",
5410
+ "protocol_sha256": "d344c9afa60711fa33950eb1a8cfc805ff7c78c5ad9b476e4f3f524638f5980c"
5411
  },
5412
  "links": {
5413
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/agent/session.jsonl",
5414
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/agent/trajectory.json",
5415
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/agent/provider-usage.jsonl",
5416
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/transcript.json",
5417
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/verifier/episode.json",
5418
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/verifier/protocol.json",
5419
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/instructions.json",
5420
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/verifier/workspace.tar.gz",
5421
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/verifier/workspace.json",
5422
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
5423
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/native/evidence/episode/evidence/recording/manifest.json",
5424
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/usage.json",
5425
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/analysis.json",
5426
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/provenance.json",
5427
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/video.mp4",
5428
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/poster.jpg",
5429
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/media-validation.json"
5430
  },
5431
  "resources": [
5432
  {
5433
+ "name": "tools/arm_control.py",
5434
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/resources/tools/arm_control.py",
5435
  "kind": "Created during this episode; final workspace snapshot."
5436
  },
5437
  {
5438
  "name": "memos/robodojo.md",
5439
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/resources/memos/robodojo.md",
5440
  "kind": "Created during this episode; final workspace snapshot."
5441
  }
5442
  ],
5443
  "session_counts": {
5444
+ "visible_events": 371,
5445
+ "observed_images": 56,
5446
+ "tool_errors": 9
5447
+ },
5448
+ "selected_for_formal_metrics": true,
5449
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/29/seed-0/"
5450
+ },
5451
+ {
5452
+ "id": "task04-31-seed0-formal",
5453
+ "task_key": "task04/31",
5454
+ "family": "task04",
5455
+ "slot": "31",
5456
+ "seed": 0,
5457
+ "episode": 1,
5458
+ "phase": "formal",
5459
+ "status": "completed",
5460
+ "success": false,
5461
+ "native_reward": 0.0,
5462
+ "valid": true,
5463
+ "execution": {
5464
+ "reason": null,
5465
+ "status": "finished"
5466
+ },
5467
+ "verdict": {
5468
+ "evidence_valid": true,
5469
+ "steps": 863,
5470
+ "success": false,
5471
+ "termination": "stopped"
5472
+ },
5473
+ "steps": 863,
5474
+ "simulation_time_s": null,
5475
+ "wall_time_s": 1296.406492,
5476
+ "model": "gpt-6-astra",
5477
+ "effort": "high",
5478
+ "harness": "codex",
5479
+ "codex_version": "0.160.0",
5480
+ "native_instruction": "Lift the basket more than 8 cm, identify the target object on the conveyor according to the image on the board, pick it up, and place it into the basket.",
5481
+ "instruction": "Identify the target object on the conveyor from the image on the board. Lift and hold the basket more than 8 cm above its starting height, then pick up the matching object and place it down inside the basket. Keep the basket raised, with the object resting at its bottom and more than 8 cm above its original height.",
5482
+ "instruction_policy": "modified",
5483
+ "usage": {
5484
+ "accounting": "reported-responses",
5485
+ "audit_complete": true,
5486
+ "cache_hit_rate": 0.9752413698477898,
5487
+ "cache_reported_input_tokens": 3797181,
5488
+ "cache_write_input_tokens": 0,
5489
+ "cache_write_reported_input_tokens": 3797181,
5490
+ "cached_input_tokens": 3703168,
5491
+ "completed_turns": 1,
5492
+ "cost_usd": null,
5493
+ "failed_turns": 0,
5494
+ "input_tokens": 3797181,
5495
+ "known_cache_write_input_tokens": 0,
5496
+ "known_cached_input_tokens": 3703168,
5497
+ "known_input_tokens": 3797181,
5498
+ "known_output_tokens": 23800,
5499
+ "known_reasoning_output_tokens": 11319,
5500
+ "output_tokens": 23800,
5501
+ "reasoning_output_tokens": 11319,
5502
+ "reasoning_reported_output_tokens": 23800,
5503
+ "reported_responses": {
5504
+ "cache_reported_input_tokens": 65,
5505
+ "cache_write_input_tokens": 65,
5506
+ "cache_write_reported_input_tokens": 65,
5507
+ "cached_input_tokens": 65,
5508
+ "input_tokens": 65,
5509
+ "output_tokens": 65,
5510
+ "reasoning_output_tokens": 65,
5511
+ "reasoning_reported_output_tokens": 65
5512
+ },
5513
+ "response_count": 65,
5514
+ "response_ids_complete": true,
5515
+ "schema": "rlebench/token-usage/1",
5516
+ "source": "Codex token_usage_record per response",
5517
+ "uncached_input_tokens": 94013,
5518
+ "unidentified_usage_records": 0
5519
+ },
5520
+ "call_activity": {
5521
+ "model_tool_calls": 64,
5522
+ "model_tool_calls_by_name": {
5523
+ "exec": 64
5524
+ },
5525
+ "nested_python_tool_invocations": null,
5526
+ "python_device_rpc_attempts": null,
5527
+ "python_device_rpc_attempts_by_action": null,
5528
+ "python_device_rpc_errors": null,
5529
+ "python_instrumented_model_tool_calls": null,
5530
+ "python_tool_invocations": null,
5531
+ "python_tool_invocations_by_origin": null,
5532
+ "schema": "rlebench/call-activity/1",
5533
+ "source": "Codex native sessions"
5534
+ },
5535
+ "media": {
5536
+ "passed": true,
5537
+ "width": 2880,
5538
+ "height": 720,
5539
+ "duration_s": 8.65,
5540
+ "speed": 4,
5541
+ "source_fps": 10,
5542
+ "output_fps": 20,
5543
+ "recording": {
5544
+ "accepted_samples": 346,
5545
+ "captured_samples": 346,
5546
+ "clock": "simulation",
5547
+ "dropped_samples": 0,
5548
+ "encoded_frames": 346,
5549
+ "end_time_s": 34.51999999999944,
5550
+ "error": null,
5551
+ "experimental": true,
5552
+ "fps": 10,
5553
+ "received_samples": 346,
5554
+ "schema": "roboenv/recording/1",
5555
+ "state": "closed",
5556
+ "status": "complete",
5557
+ "views": [
5558
+ {
5559
+ "fov_y": 45.0,
5560
+ "height": 720,
5561
+ "name": "third_person",
5562
+ "pose": null,
5563
+ "source": "third_person",
5564
+ "width": 960
5565
+ },
5566
+ {
5567
+ "fov_y": 45.0,
5568
+ "height": 720,
5569
+ "name": "left_wrist",
5570
+ "pose": null,
5571
+ "source": "left_wrist",
5572
+ "width": 960
5573
+ },
5574
+ {
5575
+ "fov_y": 45.0,
5576
+ "height": 720,
5577
+ "name": "right_wrist",
5578
+ "pose": null,
5579
+ "source": "right_wrist",
5580
+ "width": 960
5581
+ }
5582
+ ]
5583
+ },
5584
+ "view_names": [
5585
+ "third_person",
5586
+ "left_wrist",
5587
+ "right_wrist"
5588
+ ],
5589
+ "sha256": "885151826ea1b79abd7772d71bb25c1fe86d2f09dcfddd47938e6381c5fbe6de",
5590
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
5591
+ },
5592
+ "analysis": {
5593
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
5594
+ "duration": "\u4f7f\u7528 863 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
5595
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
5596
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
5597
+ },
5598
+ "provenance": {
5599
+ "sources": {
5600
+ "RLE-Bench-inhouse": {
5601
+ "build_inputs": [
5602
+ "pyproject.toml",
5603
+ "src",
5604
+ "tasks",
5605
+ "README.md",
5606
+ "Makefile",
5607
+ "tests",
5608
+ "docs"
5609
+ ],
5610
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
5611
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
5612
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
5613
+ },
5614
+ "RoboEnv": {
5615
+ "build_inputs": [
5616
+ "pyproject.toml",
5617
+ "README.md",
5618
+ "src",
5619
+ "runtime/pyproject.toml",
5620
+ "runtime/README.md",
5621
+ "runtime/src",
5622
+ "runtime/environments.json",
5623
+ "runtime/locks",
5624
+ "catalog",
5625
+ "upstreams.lock.json",
5626
+ "third_party/patches",
5627
+ "docs/validation"
5628
+ ],
5629
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
5630
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
5631
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
5632
+ }
5633
+ },
5634
+ "job": "31-robodojo-pick-from-conveyor-by-image-codex-seed0-attempt01",
5635
+ "attempt": 1,
5636
+ "harness": "stock Codex CLI",
5637
+ "codex_version": "0.160.0",
5638
+ "model": "gpt-6-astra",
5639
+ "effort": "high",
5640
+ "service_tier": "default",
5641
+ "fresh_session": true,
5642
+ "source_jobs": [],
5643
+ "resume_trajectory": false,
5644
+ "imported_skills": [],
5645
+ "automatic_harbor_retries": 0,
5646
+ "request_policy": {
5647
+ "max_request_retries": 50,
5648
+ "configuration": "explicit retry50 SSE",
5649
+ "usage_accounting": "reported-responses"
5650
+ },
5651
+ "classification": "formal",
5652
+ "measured_images": {
5653
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
5654
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
5655
+ },
5656
+ "session_original_sha256": "aa769c4c945fa18e304b7d4d5d8c9c603afb75f438dc03b941fd4b9eb08f7fe0",
5657
+ "protocol_sha256": "9ce80cb246bf47f91aa2889530d7ef56aa5498f223bbd77c69f483653254c63d"
5658
+ },
5659
+ "links": {
5660
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/agent/session.jsonl",
5661
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/agent/trajectory.json",
5662
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/agent/provider-usage.jsonl",
5663
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/transcript.json",
5664
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/verifier/episode.json",
5665
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/verifier/protocol.json",
5666
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/instructions.json",
5667
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/verifier/workspace.tar.gz",
5668
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/verifier/workspace.json",
5669
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
5670
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/native/evidence/episode/evidence/recording/manifest.json",
5671
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/usage.json",
5672
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/analysis.json",
5673
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/provenance.json",
5674
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/video.mp4",
5675
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/poster.jpg",
5676
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/media-validation.json"
5677
+ },
5678
+ "resources": [
5679
+ {
5680
+ "name": "tools/robot.py",
5681
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/resources/tools/robot.py",
5682
+ "kind": "Created during this episode; final workspace snapshot."
5683
+ },
5684
+ {
5685
+ "name": "memos/robodojo.md",
5686
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/31/seed-0/resources/memos/robodojo.md",
5687
+ "kind": "Created during this episode; final workspace snapshot."
5688
+ }
5689
+ ],
5690
+ "session_counts": {
5691
+ "visible_events": 141,
5692
  "observed_images": 78,
5693
  "tool_errors": 7
5694
  },
 
6191
  },
6192
  "selected_for_formal_metrics": true,
6193
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/33/seed-0/"
6194
+ },
6195
+ {
6196
+ "id": "task04-35-seed0-formal",
6197
+ "task_key": "task04/35",
6198
+ "family": "task04",
6199
+ "slot": "35",
6200
+ "seed": 0,
6201
+ "episode": 1,
6202
+ "phase": "formal",
6203
+ "status": "completed",
6204
+ "success": false,
6205
+ "native_reward": 0.0,
6206
+ "valid": true,
6207
+ "execution": {
6208
+ "reason": null,
6209
+ "status": "finished"
6210
+ },
6211
+ "verdict": {
6212
+ "evidence_valid": true,
6213
+ "steps": 1616,
6214
+ "success": false,
6215
+ "termination": "stopped"
6216
+ },
6217
+ "steps": 1616,
6218
+ "simulation_time_s": null,
6219
+ "wall_time_s": 834.470016,
6220
+ "model": "gpt-6-astra",
6221
+ "effort": "high",
6222
+ "harness": "codex",
6223
+ "codex_version": "0.160.0",
6224
+ "native_instruction": "Pour the liquid from the first violet bottle into the first black bowl, from the second red bottle into the second white bowl, and from the third turquoise bottle into the third brown bowl. Then reset the robot arm.",
6225
+ "instruction": "Pour the liquid from the first violet bottle into the first black bowl, from the second red bottle into the second white bowl, and from the third turquoise bottle into the third brown bowl. Then reset the robot arm.",
6226
+ "instruction_policy": "original_native",
6227
+ "usage": {
6228
+ "accounting": "reported-responses",
6229
+ "audit_complete": true,
6230
+ "cache_hit_rate": 0.9667383369019035,
6231
+ "cache_reported_input_tokens": 2677076,
6232
+ "cache_write_input_tokens": 0,
6233
+ "cache_write_reported_input_tokens": 2677076,
6234
+ "cached_input_tokens": 2588032,
6235
+ "completed_turns": 1,
6236
+ "cost_usd": null,
6237
+ "failed_turns": 0,
6238
+ "input_tokens": 2677076,
6239
+ "known_cache_write_input_tokens": 0,
6240
+ "known_cached_input_tokens": 2588032,
6241
+ "known_input_tokens": 2677076,
6242
+ "known_output_tokens": 11360,
6243
+ "known_reasoning_output_tokens": 3492,
6244
+ "output_tokens": 11360,
6245
+ "reasoning_output_tokens": 3492,
6246
+ "reasoning_reported_output_tokens": 11360,
6247
+ "reported_responses": {
6248
+ "cache_reported_input_tokens": 69,
6249
+ "cache_write_input_tokens": 69,
6250
+ "cache_write_reported_input_tokens": 69,
6251
+ "cached_input_tokens": 69,
6252
+ "input_tokens": 69,
6253
+ "output_tokens": 69,
6254
+ "reasoning_output_tokens": 69,
6255
+ "reasoning_reported_output_tokens": 69
6256
+ },
6257
+ "response_count": 69,
6258
+ "response_ids_complete": true,
6259
+ "schema": "rlebench/token-usage/1",
6260
+ "source": "Codex token_usage_record per response",
6261
+ "uncached_input_tokens": 89044,
6262
+ "unidentified_usage_records": 0
6263
+ },
6264
+ "call_activity": {
6265
+ "model_tool_calls": 68,
6266
+ "model_tool_calls_by_name": {
6267
+ "exec": 68
6268
+ },
6269
+ "nested_python_tool_invocations": null,
6270
+ "python_device_rpc_attempts": null,
6271
+ "python_device_rpc_attempts_by_action": null,
6272
+ "python_device_rpc_errors": null,
6273
+ "python_instrumented_model_tool_calls": null,
6274
+ "python_tool_invocations": null,
6275
+ "python_tool_invocations_by_origin": null,
6276
+ "schema": "rlebench/call-activity/1",
6277
+ "source": "Codex native sessions"
6278
+ },
6279
+ "media": {
6280
+ "passed": true,
6281
+ "width": 2880,
6282
+ "height": 720,
6283
+ "duration_s": 16.15,
6284
+ "speed": 4,
6285
+ "source_fps": 10,
6286
+ "output_fps": 20,
6287
+ "recording": {
6288
+ "accepted_samples": 648,
6289
+ "captured_samples": 648,
6290
+ "clock": "simulation",
6291
+ "dropped_samples": 0,
6292
+ "encoded_frames": 647,
6293
+ "end_time_s": 64.6399999999989,
6294
+ "error": null,
6295
+ "experimental": true,
6296
+ "fps": 10,
6297
+ "received_samples": 648,
6298
+ "schema": "roboenv/recording/1",
6299
+ "state": "closed",
6300
+ "status": "complete",
6301
+ "views": [
6302
+ {
6303
+ "fov_y": 45.0,
6304
+ "height": 720,
6305
+ "name": "third_person",
6306
+ "pose": null,
6307
+ "source": "third_person",
6308
+ "width": 960
6309
+ },
6310
+ {
6311
+ "fov_y": 45.0,
6312
+ "height": 720,
6313
+ "name": "left_wrist",
6314
+ "pose": null,
6315
+ "source": "left_wrist",
6316
+ "width": 960
6317
+ },
6318
+ {
6319
+ "fov_y": 45.0,
6320
+ "height": 720,
6321
+ "name": "right_wrist",
6322
+ "pose": null,
6323
+ "source": "right_wrist",
6324
+ "width": 960
6325
+ }
6326
+ ]
6327
+ },
6328
+ "view_names": [
6329
+ "third_person",
6330
+ "left_wrist",
6331
+ "right_wrist"
6332
+ ],
6333
+ "sha256": "207a29db68023f38d3e81e5c185ee358ef76f4dd8482e610e43b1fdc41d39da7",
6334
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
6335
+ },
6336
+ "analysis": {
6337
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
6338
+ "duration": "\u4f7f\u7528 1,616 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
6339
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
6340
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
6341
+ },
6342
+ "provenance": {
6343
+ "sources": {
6344
+ "RLE-Bench-inhouse": {
6345
+ "build_inputs": [
6346
+ "pyproject.toml",
6347
+ "src",
6348
+ "tasks",
6349
+ "README.md",
6350
+ "Makefile",
6351
+ "tests",
6352
+ "docs"
6353
+ ],
6354
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
6355
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
6356
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
6357
+ },
6358
+ "RoboEnv": {
6359
+ "build_inputs": [
6360
+ "pyproject.toml",
6361
+ "README.md",
6362
+ "src",
6363
+ "runtime/pyproject.toml",
6364
+ "runtime/README.md",
6365
+ "runtime/src",
6366
+ "runtime/environments.json",
6367
+ "runtime/locks",
6368
+ "catalog",
6369
+ "upstreams.lock.json",
6370
+ "third_party/patches",
6371
+ "docs/validation"
6372
+ ],
6373
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
6374
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
6375
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
6376
+ }
6377
+ },
6378
+ "job": "35-robodojo-pour-by-language-codex-seed0-attempt01",
6379
+ "attempt": 1,
6380
+ "harness": "stock Codex CLI",
6381
+ "codex_version": "0.160.0",
6382
+ "model": "gpt-6-astra",
6383
+ "effort": "high",
6384
+ "service_tier": "default",
6385
+ "fresh_session": true,
6386
+ "source_jobs": [],
6387
+ "resume_trajectory": false,
6388
+ "imported_skills": [],
6389
+ "automatic_harbor_retries": 0,
6390
+ "request_policy": {
6391
+ "max_request_retries": 50,
6392
+ "configuration": "explicit retry50 SSE",
6393
+ "usage_accounting": "reported-responses"
6394
+ },
6395
+ "classification": "formal",
6396
+ "measured_images": {
6397
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
6398
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
6399
+ },
6400
+ "session_original_sha256": "9823e4fb42ba34ba2a7c47bd1c5cd29624d9352da943026e6dd86a61f1da18b5",
6401
+ "protocol_sha256": "b10751bbc24694b2b82ddcb85171f0806a4d2c357f0001a79f4780ffb7e227aa"
6402
+ },
6403
+ "links": {
6404
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/agent/session.jsonl",
6405
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/agent/trajectory.json",
6406
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/agent/provider-usage.jsonl",
6407
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/transcript.json",
6408
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/verifier/episode.json",
6409
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/verifier/protocol.json",
6410
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/instructions.json",
6411
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/verifier/workspace.tar.gz",
6412
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/verifier/workspace.json",
6413
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
6414
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/native/evidence/episode/evidence/recording/manifest.json",
6415
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/usage.json",
6416
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/analysis.json",
6417
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/provenance.json",
6418
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/video.mp4",
6419
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/poster.jpg",
6420
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/media-validation.json"
6421
+ },
6422
+ "resources": [
6423
+ {
6424
+ "name": "tools/arx.py",
6425
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/resources/tools/arx.py",
6426
+ "kind": "Created during this episode; final workspace snapshot."
6427
+ },
6428
+ {
6429
+ "name": "memos/robodojo.md",
6430
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/resources/memos/robodojo.md",
6431
+ "kind": "Created during this episode; final workspace snapshot."
6432
+ }
6433
+ ],
6434
+ "session_counts": {
6435
+ "visible_events": 154,
6436
+ "observed_images": 17,
6437
+ "tool_errors": 2
6438
+ },
6439
+ "selected_for_formal_metrics": true,
6440
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/35/seed-0/"
6441
+ },
6442
+ {
6443
+ "id": "task04-36-seed0-formal",
6444
+ "task_key": "task04/36",
6445
+ "family": "task04",
6446
+ "slot": "36",
6447
+ "seed": 0,
6448
+ "episode": 1,
6449
+ "phase": "formal",
6450
+ "status": "completed",
6451
+ "success": true,
6452
+ "native_reward": 1.0,
6453
+ "valid": true,
6454
+ "execution": {
6455
+ "reason": null,
6456
+ "status": "finished"
6457
+ },
6458
+ "verdict": {
6459
+ "evidence_valid": true,
6460
+ "steps": 1071,
6461
+ "success": true,
6462
+ "termination": "success"
6463
+ },
6464
+ "steps": 1071,
6465
+ "simulation_time_s": null,
6466
+ "wall_time_s": 777.348758,
6467
+ "model": "gpt-6-astra",
6468
+ "effort": "high",
6469
+ "harness": "codex",
6470
+ "codex_version": "0.160.0",
6471
+ "native_instruction": "Pour the liquid from the bottle into the cup.",
6472
+ "instruction": "Pour nearly all the liquid into the cup with almost no spillage. Keep the bottle tilted over the cup until the flow stops and the liquid settles, then return it upright, leaving only a small residue inside.",
6473
+ "instruction_policy": "modified",
6474
+ "usage": {
6475
+ "accounting": "reported-responses",
6476
+ "audit_complete": true,
6477
+ "cache_hit_rate": 0.9768983350616229,
6478
+ "cache_reported_input_tokens": 2577693,
6479
+ "cache_write_input_tokens": 0,
6480
+ "cache_write_reported_input_tokens": 2577693,
6481
+ "cached_input_tokens": 2518144,
6482
+ "completed_turns": 1,
6483
+ "cost_usd": null,
6484
+ "failed_turns": 0,
6485
+ "input_tokens": 2577693,
6486
+ "known_cache_write_input_tokens": 0,
6487
+ "known_cached_input_tokens": 2518144,
6488
+ "known_input_tokens": 2577693,
6489
+ "known_output_tokens": 15029,
6490
+ "known_reasoning_output_tokens": 6246,
6491
+ "output_tokens": 15029,
6492
+ "reasoning_output_tokens": 6246,
6493
+ "reasoning_reported_output_tokens": 15029,
6494
+ "reported_responses": {
6495
+ "cache_reported_input_tokens": 61,
6496
+ "cache_write_input_tokens": 61,
6497
+ "cache_write_reported_input_tokens": 61,
6498
+ "cached_input_tokens": 61,
6499
+ "input_tokens": 61,
6500
+ "output_tokens": 61,
6501
+ "reasoning_output_tokens": 61,
6502
+ "reasoning_reported_output_tokens": 61
6503
+ },
6504
+ "response_count": 61,
6505
+ "response_ids_complete": true,
6506
+ "schema": "rlebench/token-usage/1",
6507
+ "source": "Codex token_usage_record per response",
6508
+ "uncached_input_tokens": 59549,
6509
+ "unidentified_usage_records": 0
6510
+ },
6511
+ "call_activity": {
6512
+ "model_tool_calls": 60,
6513
+ "model_tool_calls_by_name": {
6514
+ "exec": 60
6515
+ },
6516
+ "nested_python_tool_invocations": null,
6517
+ "python_device_rpc_attempts": null,
6518
+ "python_device_rpc_attempts_by_action": null,
6519
+ "python_device_rpc_errors": null,
6520
+ "python_instrumented_model_tool_calls": null,
6521
+ "python_tool_invocations": null,
6522
+ "python_tool_invocations_by_origin": null,
6523
+ "schema": "rlebench/call-activity/1",
6524
+ "source": "Codex native sessions"
6525
+ },
6526
+ "media": {
6527
+ "passed": true,
6528
+ "width": 2880,
6529
+ "height": 720,
6530
+ "duration_s": 10.7,
6531
+ "speed": 4,
6532
+ "source_fps": 10,
6533
+ "output_fps": 20,
6534
+ "recording": {
6535
+ "accepted_samples": 430,
6536
+ "captured_samples": 430,
6537
+ "clock": "simulation",
6538
+ "dropped_samples": 0,
6539
+ "encoded_frames": 429,
6540
+ "end_time_s": 42.839999999999264,
6541
+ "error": null,
6542
+ "experimental": true,
6543
+ "fps": 10,
6544
+ "received_samples": 430,
6545
+ "schema": "roboenv/recording/1",
6546
+ "state": "closed",
6547
+ "status": "complete",
6548
+ "views": [
6549
+ {
6550
+ "fov_y": 45.0,
6551
+ "height": 720,
6552
+ "name": "third_person",
6553
+ "pose": null,
6554
+ "source": "third_person",
6555
+ "width": 960
6556
+ },
6557
+ {
6558
+ "fov_y": 45.0,
6559
+ "height": 720,
6560
+ "name": "left_wrist",
6561
+ "pose": null,
6562
+ "source": "left_wrist",
6563
+ "width": 960
6564
+ },
6565
+ {
6566
+ "fov_y": 45.0,
6567
+ "height": 720,
6568
+ "name": "right_wrist",
6569
+ "pose": null,
6570
+ "source": "right_wrist",
6571
+ "width": 960
6572
+ }
6573
+ ]
6574
+ },
6575
+ "view_names": [
6576
+ "third_person",
6577
+ "left_wrist",
6578
+ "right_wrist"
6579
+ ],
6580
+ "sha256": "0fe3ccbc26c9d14bc92ead5c4bcc79adbd0ceb97922b07e30730ea9a629065c1",
6581
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
6582
+ },
6583
+ "analysis": {
6584
+ "outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
6585
+ "duration": "\u4f7f\u7528 1,071 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
6586
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
6587
+ },
6588
+ "provenance": {
6589
+ "sources": {
6590
+ "RLE-Bench-inhouse": {
6591
+ "build_inputs": [
6592
+ "pyproject.toml",
6593
+ "src",
6594
+ "tasks",
6595
+ "README.md",
6596
+ "Makefile",
6597
+ "tests",
6598
+ "docs"
6599
+ ],
6600
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
6601
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
6602
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
6603
+ },
6604
+ "RoboEnv": {
6605
+ "build_inputs": [
6606
+ "pyproject.toml",
6607
+ "README.md",
6608
+ "src",
6609
+ "runtime/pyproject.toml",
6610
+ "runtime/README.md",
6611
+ "runtime/src",
6612
+ "runtime/environments.json",
6613
+ "runtime/locks",
6614
+ "catalog",
6615
+ "upstreams.lock.json",
6616
+ "third_party/patches",
6617
+ "docs/validation"
6618
+ ],
6619
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
6620
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
6621
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
6622
+ }
6623
+ },
6624
+ "job": "36-robodojo-pour-liquid-into-cup-codex-seed0-attempt01",
6625
+ "attempt": 1,
6626
+ "harness": "stock Codex CLI",
6627
+ "codex_version": "0.160.0",
6628
+ "model": "gpt-6-astra",
6629
+ "effort": "high",
6630
+ "service_tier": "default",
6631
+ "fresh_session": true,
6632
+ "source_jobs": [],
6633
+ "resume_trajectory": false,
6634
+ "imported_skills": [],
6635
+ "automatic_harbor_retries": 0,
6636
+ "request_policy": {
6637
+ "max_request_retries": 50,
6638
+ "configuration": "explicit retry50 SSE",
6639
+ "usage_accounting": "reported-responses"
6640
+ },
6641
+ "classification": "formal",
6642
+ "measured_images": {
6643
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
6644
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
6645
+ },
6646
+ "session_original_sha256": "7e5d9d89a62d36b2acc0686d0c45410afecd3f4a6a11d4909cf399edf099191d",
6647
+ "protocol_sha256": "b149476eb7b742f99fdc2849cf33dae7045952a02b90dd86fc3756a18c01a07b"
6648
+ },
6649
+ "links": {
6650
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/agent/session.jsonl",
6651
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/agent/trajectory.json",
6652
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/agent/provider-usage.jsonl",
6653
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/transcript.json",
6654
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/verifier/episode.json",
6655
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/verifier/protocol.json",
6656
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/instructions.json",
6657
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/verifier/workspace.tar.gz",
6658
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/verifier/workspace.json",
6659
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
6660
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/native/evidence/episode/evidence/recording/manifest.json",
6661
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/usage.json",
6662
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/analysis.json",
6663
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/provenance.json",
6664
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/video.mp4",
6665
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/poster.jpg",
6666
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/media-validation.json"
6667
+ },
6668
+ "resources": [
6669
+ {
6670
+ "name": "tools/pour_control.py",
6671
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/resources/tools/pour_control.py",
6672
+ "kind": "Created during this episode; final workspace snapshot."
6673
+ },
6674
+ {
6675
+ "name": "tools/robot_control.py",
6676
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/resources/tools/robot_control.py",
6677
+ "kind": "Created during this episode; final workspace snapshot."
6678
+ },
6679
+ {
6680
+ "name": "memos/pouring.md",
6681
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/resources/memos/pouring.md",
6682
+ "kind": "Created during this episode; final workspace snapshot."
6683
+ }
6684
+ ],
6685
+ "session_counts": {
6686
+ "visible_events": 135,
6687
+ "observed_images": 30,
6688
+ "tool_errors": 4
6689
+ },
6690
+ "selected_for_formal_metrics": true,
6691
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/36/seed-0/"
6692
+ },
6693
+ {
6694
+ "id": "task04-38-seed0-formal",
6695
+ "task_key": "task04/38",
6696
+ "family": "task04",
6697
+ "slot": "38",
6698
+ "seed": 0,
6699
+ "episode": 1,
6700
+ "phase": "formal",
6701
+ "status": "completed",
6702
+ "success": true,
6703
+ "native_reward": 1.0,
6704
+ "valid": true,
6705
+ "execution": {
6706
+ "reason": null,
6707
+ "status": "finished"
6708
+ },
6709
+ "verdict": {
6710
+ "evidence_valid": true,
6711
+ "steps": 952,
6712
+ "success": true,
6713
+ "termination": "success"
6714
+ },
6715
+ "steps": 952,
6716
+ "simulation_time_s": null,
6717
+ "wall_time_s": 408.460414,
6718
+ "model": "gpt-6-astra",
6719
+ "effort": "high",
6720
+ "harness": "codex",
6721
+ "codex_version": "0.160.0",
6722
+ "native_instruction": "Press the two red buttons the required number of times according to the number cards, then press the blue button to confirm.",
6723
+ "instruction": "Starting with the left red button, press and release each red button the number of times shown on its card, confirming that button's count with a press and release of the blue button before moving to the next. Return both arms to their starting poses when finished.",
6724
+ "instruction_policy": "modified",
6725
+ "usage": {
6726
+ "accounting": "reported-responses",
6727
+ "audit_complete": true,
6728
+ "cache_hit_rate": 0.9579166678796666,
6729
+ "cache_reported_input_tokens": 1030503,
6730
+ "cache_write_input_tokens": 0,
6731
+ "cache_write_reported_input_tokens": 1030503,
6732
+ "cached_input_tokens": 987136,
6733
+ "completed_turns": 1,
6734
+ "cost_usd": null,
6735
+ "failed_turns": 0,
6736
+ "input_tokens": 1030503,
6737
+ "known_cache_write_input_tokens": 0,
6738
+ "known_cached_input_tokens": 987136,
6739
+ "known_input_tokens": 1030503,
6740
+ "known_output_tokens": 6991,
6741
+ "known_reasoning_output_tokens": 2046,
6742
+ "output_tokens": 6991,
6743
+ "reasoning_output_tokens": 2046,
6744
+ "reasoning_reported_output_tokens": 6991,
6745
+ "reported_responses": {
6746
+ "cache_reported_input_tokens": 30,
6747
+ "cache_write_input_tokens": 30,
6748
+ "cache_write_reported_input_tokens": 30,
6749
+ "cached_input_tokens": 30,
6750
+ "input_tokens": 30,
6751
+ "output_tokens": 30,
6752
+ "reasoning_output_tokens": 30,
6753
+ "reasoning_reported_output_tokens": 30
6754
+ },
6755
+ "response_count": 30,
6756
+ "response_ids_complete": true,
6757
+ "schema": "rlebench/token-usage/1",
6758
+ "source": "Codex token_usage_record per response",
6759
+ "uncached_input_tokens": 43367,
6760
+ "unidentified_usage_records": 0
6761
+ },
6762
+ "call_activity": {
6763
+ "model_tool_calls": 29,
6764
+ "model_tool_calls_by_name": {
6765
+ "exec": 29
6766
+ },
6767
+ "nested_python_tool_invocations": null,
6768
+ "python_device_rpc_attempts": null,
6769
+ "python_device_rpc_attempts_by_action": null,
6770
+ "python_device_rpc_errors": null,
6771
+ "python_instrumented_model_tool_calls": null,
6772
+ "python_tool_invocations": null,
6773
+ "python_tool_invocations_by_origin": null,
6774
+ "schema": "rlebench/call-activity/1",
6775
+ "source": "Codex native sessions"
6776
+ },
6777
+ "media": {
6778
+ "passed": true,
6779
+ "width": 2880,
6780
+ "height": 720,
6781
+ "duration_s": 9.5,
6782
+ "speed": 4,
6783
+ "source_fps": 10,
6784
+ "output_fps": 20,
6785
+ "recording": {
6786
+ "accepted_samples": 382,
6787
+ "captured_samples": 382,
6788
+ "clock": "simulation",
6789
+ "dropped_samples": 0,
6790
+ "encoded_frames": 381,
6791
+ "end_time_s": 38.079999999999366,
6792
+ "error": null,
6793
+ "experimental": true,
6794
+ "fps": 10,
6795
+ "received_samples": 382,
6796
+ "schema": "roboenv/recording/1",
6797
+ "state": "closed",
6798
+ "status": "complete",
6799
+ "views": [
6800
+ {
6801
+ "fov_y": 45.0,
6802
+ "height": 720,
6803
+ "name": "third_person",
6804
+ "pose": null,
6805
+ "source": "third_person",
6806
+ "width": 960
6807
+ },
6808
+ {
6809
+ "fov_y": 45.0,
6810
+ "height": 720,
6811
+ "name": "left_wrist",
6812
+ "pose": null,
6813
+ "source": "left_wrist",
6814
+ "width": 960
6815
+ },
6816
+ {
6817
+ "fov_y": 45.0,
6818
+ "height": 720,
6819
+ "name": "right_wrist",
6820
+ "pose": null,
6821
+ "source": "right_wrist",
6822
+ "width": 960
6823
+ }
6824
+ ]
6825
+ },
6826
+ "view_names": [
6827
+ "third_person",
6828
+ "left_wrist",
6829
+ "right_wrist"
6830
+ ],
6831
+ "sha256": "1191bf611481831d762b89e5a3da004b3ca6c531987daeece4ae34cd7bceb989",
6832
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
6833
+ },
6834
+ "analysis": {
6835
+ "outcome": "\u539f\u751f\u5224\u5b9a\u6210\u529f\u3002",
6836
+ "duration": "\u4f7f\u7528 952 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1asuccess\u3002",
6837
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002"
6838
+ },
6839
+ "provenance": {
6840
+ "sources": {
6841
+ "RLE-Bench-inhouse": {
6842
+ "build_inputs": [
6843
+ "pyproject.toml",
6844
+ "src",
6845
+ "tasks",
6846
+ "README.md",
6847
+ "Makefile",
6848
+ "tests",
6849
+ "docs"
6850
+ ],
6851
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
6852
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
6853
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
6854
+ },
6855
+ "RoboEnv": {
6856
+ "build_inputs": [
6857
+ "pyproject.toml",
6858
+ "README.md",
6859
+ "src",
6860
+ "runtime/pyproject.toml",
6861
+ "runtime/README.md",
6862
+ "runtime/src",
6863
+ "runtime/environments.json",
6864
+ "runtime/locks",
6865
+ "catalog",
6866
+ "upstreams.lock.json",
6867
+ "third_party/patches",
6868
+ "docs/validation"
6869
+ ],
6870
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
6871
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
6872
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
6873
+ }
6874
+ },
6875
+ "job": "38-robodojo-press-by-number-codex-seed0-attempt01",
6876
+ "attempt": 1,
6877
+ "harness": "stock Codex CLI",
6878
+ "codex_version": "0.160.0",
6879
+ "model": "gpt-6-astra",
6880
+ "effort": "high",
6881
+ "service_tier": "default",
6882
+ "fresh_session": true,
6883
+ "source_jobs": [],
6884
+ "resume_trajectory": false,
6885
+ "imported_skills": [],
6886
+ "automatic_harbor_retries": 0,
6887
+ "request_policy": {
6888
+ "max_request_retries": 50,
6889
+ "configuration": "explicit retry50 SSE",
6890
+ "usage_accounting": "reported-responses"
6891
+ },
6892
+ "classification": "formal",
6893
+ "measured_images": {
6894
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
6895
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
6896
+ },
6897
+ "session_original_sha256": "2988988996fbda95ad102c6dcdd76e9e645c552b62bb03210caf033a154c43ee",
6898
+ "protocol_sha256": "9fce1006e083eccaf342031dee9ee536219916e20037c3b853f01379591895a3"
6899
+ },
6900
+ "links": {
6901
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/agent/session.jsonl",
6902
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/agent/trajectory.json",
6903
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/agent/provider-usage.jsonl",
6904
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/transcript.json",
6905
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/verifier/episode.json",
6906
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/verifier/protocol.json",
6907
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/instructions.json",
6908
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/verifier/workspace.tar.gz",
6909
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/verifier/workspace.json",
6910
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
6911
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/native/evidence/episode/evidence/recording/manifest.json",
6912
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/usage.json",
6913
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/analysis.json",
6914
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/provenance.json",
6915
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/video.mp4",
6916
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/poster.jpg",
6917
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/media-validation.json"
6918
+ },
6919
+ "resources": [
6920
+ {
6921
+ "name": "tools/press_sequence.py",
6922
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/resources/tools/press_sequence.py",
6923
+ "kind": "Created during this episode; final workspace snapshot."
6924
+ },
6925
+ {
6926
+ "name": "tools/robot_control.py",
6927
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/resources/tools/robot_control.py",
6928
+ "kind": "Created during this episode; final workspace snapshot."
6929
+ },
6930
+ {
6931
+ "name": "memos/robodojo.md",
6932
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/resources/memos/robodojo.md",
6933
+ "kind": "Created during this episode; final workspace snapshot."
6934
+ }
6935
+ ],
6936
+ "session_counts": {
6937
+ "visible_events": 70,
6938
+ "observed_images": 19,
6939
+ "tool_errors": 3
6940
+ },
6941
+ "selected_for_formal_metrics": true,
6942
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/38/seed-0/"
6943
+ },
6944
+ {
6945
+ "id": "task04-39-seed0-formal",
6946
+ "task_key": "task04/39",
6947
+ "family": "task04",
6948
+ "slot": "39",
6949
+ "seed": 0,
6950
+ "episode": 1,
6951
+ "phase": "formal",
6952
+ "status": "completed",
6953
+ "success": false,
6954
+ "native_reward": 0.0,
6955
+ "valid": true,
6956
+ "execution": {
6957
+ "reason": null,
6958
+ "status": "finished"
6959
+ },
6960
+ "verdict": {
6961
+ "evidence_valid": true,
6962
+ "steps": 535,
6963
+ "success": false,
6964
+ "termination": "stopped"
6965
+ },
6966
+ "steps": 535,
6967
+ "simulation_time_s": null,
6968
+ "wall_time_s": 346.926941,
6969
+ "model": "gpt-6-astra",
6970
+ "effort": "high",
6971
+ "harness": "codex",
6972
+ "codex_version": "0.160.0",
6973
+ "native_instruction": "Push the T-shaped block to align it precisely with the gray T-shaped pad.",
6974
+ "instruction": "Slide the T-shaped block along the table until it neatly matches the gray T-shaped pad in position and orientation. Keep the block on the table and return both arms to their starting poses when finished.",
6975
+ "instruction_policy": "modified",
6976
+ "usage": {
6977
+ "accounting": "reported-responses",
6978
+ "audit_complete": true,
6979
+ "cache_hit_rate": 0.9539017898864306,
6980
+ "cache_reported_input_tokens": 787536,
6981
+ "cache_write_input_tokens": 0,
6982
+ "cache_write_reported_input_tokens": 787536,
6983
+ "cached_input_tokens": 751232,
6984
+ "completed_turns": 1,
6985
+ "cost_usd": null,
6986
+ "failed_turns": 0,
6987
+ "input_tokens": 787536,
6988
+ "known_cache_write_input_tokens": 0,
6989
+ "known_cached_input_tokens": 751232,
6990
+ "known_input_tokens": 787536,
6991
+ "known_output_tokens": 7374,
6992
+ "known_reasoning_output_tokens": 2132,
6993
+ "output_tokens": 7374,
6994
+ "reasoning_output_tokens": 2132,
6995
+ "reasoning_reported_output_tokens": 7374,
6996
+ "reported_responses": {
6997
+ "cache_reported_input_tokens": 25,
6998
+ "cache_write_input_tokens": 25,
6999
+ "cache_write_reported_input_tokens": 25,
7000
+ "cached_input_tokens": 25,
7001
+ "input_tokens": 25,
7002
+ "output_tokens": 25,
7003
+ "reasoning_output_tokens": 25,
7004
+ "reasoning_reported_output_tokens": 25
7005
+ },
7006
+ "response_count": 25,
7007
+ "response_ids_complete": true,
7008
+ "schema": "rlebench/token-usage/1",
7009
+ "source": "Codex token_usage_record per response",
7010
+ "uncached_input_tokens": 36304,
7011
+ "unidentified_usage_records": 0
7012
+ },
7013
+ "call_activity": {
7014
+ "model_tool_calls": 24,
7015
+ "model_tool_calls_by_name": {
7016
+ "exec": 24
7017
+ },
7018
+ "nested_python_tool_invocations": null,
7019
+ "python_device_rpc_attempts": null,
7020
+ "python_device_rpc_attempts_by_action": null,
7021
+ "python_device_rpc_errors": null,
7022
+ "python_instrumented_model_tool_calls": null,
7023
+ "python_tool_invocations": null,
7024
+ "python_tool_invocations_by_origin": null,
7025
+ "schema": "rlebench/call-activity/1",
7026
+ "source": "Codex native sessions"
7027
+ },
7028
+ "media": {
7029
+ "passed": true,
7030
+ "width": 2880,
7031
+ "height": 720,
7032
+ "duration_s": 5.35,
7033
+ "speed": 4,
7034
+ "source_fps": 10,
7035
+ "output_fps": 20,
7036
+ "recording": {
7037
+ "accepted_samples": 215,
7038
+ "captured_samples": 215,
7039
+ "clock": "simulation",
7040
+ "dropped_samples": 0,
7041
+ "encoded_frames": 215,
7042
+ "end_time_s": 21.39999999999972,
7043
+ "error": null,
7044
+ "experimental": true,
7045
+ "fps": 10,
7046
+ "received_samples": 215,
7047
+ "schema": "roboenv/recording/1",
7048
+ "state": "closed",
7049
+ "status": "complete",
7050
+ "views": [
7051
+ {
7052
+ "fov_y": 45.0,
7053
+ "height": 720,
7054
+ "name": "third_person",
7055
+ "pose": null,
7056
+ "source": "third_person",
7057
+ "width": 960
7058
+ },
7059
+ {
7060
+ "fov_y": 45.0,
7061
+ "height": 720,
7062
+ "name": "left_wrist",
7063
+ "pose": null,
7064
+ "source": "left_wrist",
7065
+ "width": 960
7066
+ },
7067
+ {
7068
+ "fov_y": 45.0,
7069
+ "height": 720,
7070
+ "name": "right_wrist",
7071
+ "pose": null,
7072
+ "source": "right_wrist",
7073
+ "width": 960
7074
+ }
7075
+ ]
7076
+ },
7077
+ "view_names": [
7078
+ "third_person",
7079
+ "left_wrist",
7080
+ "right_wrist"
7081
+ ],
7082
+ "sha256": "4351eae383c6bb677519793c915be8d479148b0156ac87ecfb962fb3768ebf93",
7083
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
7084
+ },
7085
+ "analysis": {
7086
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
7087
+ "duration": "\u4f7f\u7528 535 / 7,500 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
7088
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
7089
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
7090
+ },
7091
+ "provenance": {
7092
+ "sources": {
7093
+ "RLE-Bench-inhouse": {
7094
+ "build_inputs": [
7095
+ "pyproject.toml",
7096
+ "src",
7097
+ "tasks",
7098
+ "README.md",
7099
+ "Makefile",
7100
+ "tests",
7101
+ "docs"
7102
+ ],
7103
+ "content_sha256": "dce899056cc4d68057e15fc743e8853721d2c02a1ef051561d31607f7636bd10",
7104
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
7105
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
7106
+ },
7107
+ "RoboEnv": {
7108
+ "build_inputs": [
7109
+ "pyproject.toml",
7110
+ "README.md",
7111
+ "src",
7112
+ "runtime/pyproject.toml",
7113
+ "runtime/README.md",
7114
+ "runtime/src",
7115
+ "runtime/environments.json",
7116
+ "runtime/locks",
7117
+ "catalog",
7118
+ "upstreams.lock.json",
7119
+ "third_party/patches",
7120
+ "docs/validation"
7121
+ ],
7122
+ "content_sha256": "742ec96151dd0d5549d4febb7e091167a35d7886af224fa26d4c464197ea6050",
7123
+ "revision": "5f487ee5121a775aa6ea90919c8f5e3d31318906",
7124
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
7125
+ }
7126
+ },
7127
+ "job": "39-robodojo-push-t-codex-seed0-attempt01",
7128
+ "attempt": 1,
7129
+ "harness": "stock Codex CLI",
7130
+ "codex_version": "0.160.0",
7131
+ "model": "gpt-6-astra",
7132
+ "effort": "high",
7133
+ "service_tier": "default",
7134
+ "fresh_session": true,
7135
+ "source_jobs": [],
7136
+ "resume_trajectory": false,
7137
+ "imported_skills": [],
7138
+ "automatic_harbor_retries": 0,
7139
+ "request_policy": {
7140
+ "max_request_retries": 50,
7141
+ "configuration": "explicit retry50 SSE",
7142
+ "usage_accounting": "reported-responses"
7143
+ },
7144
+ "classification": "formal",
7145
+ "measured_images": {
7146
+ "agent": "sha256:25d96b00418501e325d0554afd10e5f882fa4f99ba3893ed8987eda62c648978",
7147
+ "task": "sha256:055fd5f30401593e71395b11f73477b9bb004b010f08e2cd51add60e7bf58f1a"
7148
+ },
7149
+ "session_original_sha256": "494f4f61f1bd5d3072e7966059e167381f188a7a55853a3c66839da85925e26d",
7150
+ "protocol_sha256": "67d36ac75c05a3dace78b479283200ce78a0d5fa330bd3deb44a3033a0feeec3"
7151
+ },
7152
+ "links": {
7153
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/agent/session.jsonl",
7154
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/agent/trajectory.json",
7155
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/agent/provider-usage.jsonl",
7156
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/transcript.json",
7157
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/verifier/episode.json",
7158
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/verifier/protocol.json",
7159
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/instructions.json",
7160
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/verifier/workspace.tar.gz",
7161
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/verifier/workspace.json",
7162
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
7163
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/native/evidence/episode/evidence/recording/manifest.json",
7164
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/usage.json",
7165
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/analysis.json",
7166
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/provenance.json",
7167
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/video.mp4",
7168
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/poster.jpg",
7169
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/media-validation.json"
7170
+ },
7171
+ "resources": [
7172
+ {
7173
+ "name": "tools/arx_control.py",
7174
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/resources/tools/arx_control.py",
7175
+ "kind": "Created during this episode; final workspace snapshot."
7176
+ },
7177
+ {
7178
+ "name": "memos/robodojo_push_t.md",
7179
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/resources/memos/robodojo_push_t.md",
7180
+ "kind": "Created during this episode; final workspace snapshot."
7181
+ }
7182
+ ],
7183
+ "session_counts": {
7184
+ "visible_events": 59,
7185
+ "observed_images": 16,
7186
+ "tool_errors": 2
7187
+ },
7188
+ "selected_for_formal_metrics": true,
7189
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task04/39/seed-0/"
7190
  }
7191
  ]
manifest.json CHANGED
@@ -1,21 +1,21 @@
1
  {
2
  ".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
3
  "DATA_FORMAT.md": "90cde3a7afcec6312f6705d47a959bd2135f6ed599b1ef70944eeca209b96778",
4
- "README.md": "40d380ceee51464c8ce19ca06c1ede4c6ed01904a9602ad6d735dfbd07a43cb1",
5
- "REPORT.md": "de51db09674c2e3e6eac1c373ed92e3645b5abd803e56cc09b93f20909e84135",
6
  "THIRD_PARTY_NOTICES.md": "b90a0d2dc93b1686caed55e3c68f85a1c761cb95a4935c692a223ab7e19c4c28",
7
  "app.js": "3ef085aa517e167f8020eaef3ef26bbf4ca71f759e332c2358ddf3411c25c771",
8
- "attempt-history.json": "2e7d12b2592690933f5f76bc571eba44716d5abe5a3dcd5345a3d0b1037fc8ce",
9
- "comparison.json": "873299391995b841ece2295cf26d5d53dc8d1a362f2facb304399898af99240c",
10
- "data.json": "1ba0afe87e9de1c4da2745b6b0bdd892bf3ea6596a75240ee13b0f696b6f8e9d",
11
- "episodes.csv": "e7f61f185b5b49fe851eb4d87da32f44cacf5a56fcca59ac7f460b05326a5417",
12
- "episodes.json": "877616b6f6ac859237a183762f84cbb6eefdea528bea22da77f6fa8cb9c8c1c6",
13
  "failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
14
  "favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
15
  "index.html": "da4a91fd226b8792395edbd3d2badda0e1fd69913d9c27480c12bcbdbbc73c37",
16
  "licenses/RoboDojo-LICENSE.txt": "7794bb06af8fe5485ca912454ad1665ccd7846c45f9c2b1258658e480948cdbe",
17
- "protocol.json": "079cf9ff4a1b75723dbac9dd787646a35683df685a353b6c2b97c239f8844aaf",
18
  "style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
19
- "summary.json": "0ebac68a42899257475f673ffcb2f80a6efbbd833bc0f9865be887faf3830d94",
20
- "task-index.json": "df194740f0b0919f5aa384c21365710a0d549b98fa5bb3d7dc7d4da849489a5f"
21
  }
 
1
  {
2
  ".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
3
  "DATA_FORMAT.md": "90cde3a7afcec6312f6705d47a959bd2135f6ed599b1ef70944eeca209b96778",
4
+ "README.md": "6958bdd297d5705fac4e4b401dd97aedc3eafdcbf1048f96bfee5fc376ac1b15",
5
+ "REPORT.md": "4e6d72c902855ea269cfab9fe71083234b3f76479b3e38cb712fcca220984c8f",
6
  "THIRD_PARTY_NOTICES.md": "b90a0d2dc93b1686caed55e3c68f85a1c761cb95a4935c692a223ab7e19c4c28",
7
  "app.js": "3ef085aa517e167f8020eaef3ef26bbf4ca71f759e332c2358ddf3411c25c771",
8
+ "attempt-history.json": "fd568316c8f2cdc89fb61b035cac5b4491ba2e6159eeb3a4195d5cef1d64ae4b",
9
+ "comparison.json": "2125bd27ceadfc247a6187a91e5d94eaad7c81434c355f52ea7d4c9610c447ca",
10
+ "data.json": "f4cccf3171db8a2aa47c1c1d832cbd90096a22b4c953055a4797978e26bf47ad",
11
+ "episodes.csv": "6ceccf563f6fa824a050c1d21127c4be4066a5c6618240330a90b33e2a2fcf55",
12
+ "episodes.json": "467efa0b909e3712c44e8ff3c3b84176a476320a7583f2358b4efc5330eb7378",
13
  "failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
14
  "favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
15
  "index.html": "da4a91fd226b8792395edbd3d2badda0e1fd69913d9c27480c12bcbdbbc73c37",
16
  "licenses/RoboDojo-LICENSE.txt": "7794bb06af8fe5485ca912454ad1665ccd7846c45f9c2b1258658e480948cdbe",
17
+ "protocol.json": "3f843f0f197f83d8d7a0b1a37b1dde46fb4be60ff43e7e654c7d74bd73defdac",
18
  "style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
19
+ "summary.json": "dbb6cb38fc6889cac4e443fddf929d348984e37dc7944f59ca44a38aa6535bc2",
20
+ "task-index.json": "db5a5e2992f392fd2f76b8f73a9d99abffe9156bf2722a9926687abd90af3592"
21
  }
protocol.json CHANGED
@@ -16,21 +16,21 @@
16
  "id": "task04",
17
  "name": "RoboDojo",
18
  "total": 42,
19
- "completed": 23,
20
- "pending": 19,
21
- "successes": 18,
22
- "valid_results": 23,
23
- "success_rate": 0.782608695652174,
24
- "input_tokens": 113213163,
25
- "cached_input_tokens": 110735488,
26
- "output_tokens": 458950,
27
- "usage_complete": 23,
28
  "control_frequency_hz": 25,
29
  "max_control_steps": 7500,
30
  "preflight_results": 0,
31
- "formal_results": 23,
32
- "modified_results": 18,
33
- "original_results": 5
34
  }
35
  ],
36
  "usage_accounting": "reported-responses",
 
16
  "id": "task04",
17
  "name": "RoboDojo",
18
  "total": 42,
19
+ "completed": 29,
20
+ "pending": 13,
21
+ "successes": 21,
22
+ "valid_results": 29,
23
+ "success_rate": 0.7241379310344828,
24
+ "input_tokens": 161829633,
25
+ "cached_input_tokens": 158729600,
26
+ "output_tokens": 619780,
27
+ "usage_complete": 29,
28
  "control_frequency_hz": 25,
29
  "max_control_steps": 7500,
30
  "preflight_results": 0,
31
+ "formal_results": 29,
32
+ "modified_results": 23,
33
+ "original_results": 6
34
  }
35
  ],
36
  "usage_accounting": "reported-responses",
publication-manifest.json CHANGED
@@ -10,11 +10,11 @@
10
  "bytes": 1129
11
  },
12
  "README.md": {
13
- "sha256": "40d380ceee51464c8ce19ca06c1ede4c6ed01904a9602ad6d735dfbd07a43cb1",
14
  "bytes": 1180
15
  },
16
  "REPORT.md": {
17
- "sha256": "de51db09674c2e3e6eac1c373ed92e3645b5abd803e56cc09b93f20909e84135",
18
  "bytes": 1057
19
  },
20
  "THIRD_PARTY_NOTICES.md": {
@@ -26,24 +26,24 @@
26
  "bytes": 24230
27
  },
28
  "attempt-history.json": {
29
- "sha256": "2e7d12b2592690933f5f76bc571eba44716d5abe5a3dcd5345a3d0b1037fc8ce",
30
- "bytes": 16089
31
  },
32
  "comparison.json": {
33
- "sha256": "873299391995b841ece2295cf26d5d53dc8d1a362f2facb304399898af99240c",
34
- "bytes": 9425
35
  },
36
  "data.json": {
37
- "sha256": "1ba0afe87e9de1c4da2745b6b0bdd892bf3ea6596a75240ee13b0f696b6f8e9d",
38
- "bytes": 396121
39
  },
40
  "episodes.csv": {
41
- "sha256": "e7f61f185b5b49fe851eb4d87da32f44cacf5a56fcca59ac7f460b05326a5417",
42
- "bytes": 3221
43
  },
44
  "episodes.json": {
45
- "sha256": "877616b6f6ac859237a183762f84cbb6eefdea528bea22da77f6fa8cb9c8c1c6",
46
- "bytes": 240930
47
  },
48
  "failure-reviews.json": {
49
  "sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
@@ -62,23 +62,23 @@
62
  "bytes": 1091
63
  },
64
  "protocol.json": {
65
- "sha256": "079cf9ff4a1b75723dbac9dd787646a35683df685a353b6c2b97c239f8844aaf",
66
- "bytes": 1083
67
  },
68
  "style.css": {
69
  "sha256": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
70
  "bytes": 16560
71
  },
72
  "summary.json": {
73
- "sha256": "0ebac68a42899257475f673ffcb2f80a6efbbd833bc0f9865be887faf3830d94",
74
  "bytes": 898
75
  },
76
  "task-index.json": {
77
- "sha256": "df194740f0b0919f5aa384c21365710a0d549b98fa5bb3d7dc7d4da849489a5f",
78
- "bytes": 131812
79
  },
80
  "manifest.json": {
81
- "sha256": "bf2591bb794f94502a98fc5a1cd686f32cb1481eca97a651c9f66b542eef2c6c",
82
  "bytes": 1671
83
  }
84
  }
 
10
  "bytes": 1129
11
  },
12
  "README.md": {
13
+ "sha256": "6958bdd297d5705fac4e4b401dd97aedc3eafdcbf1048f96bfee5fc376ac1b15",
14
  "bytes": 1180
15
  },
16
  "REPORT.md": {
17
+ "sha256": "4e6d72c902855ea269cfab9fe71083234b3f76479b3e38cb712fcca220984c8f",
18
  "bytes": 1057
19
  },
20
  "THIRD_PARTY_NOTICES.md": {
 
26
  "bytes": 24230
27
  },
28
  "attempt-history.json": {
29
+ "sha256": "fd568316c8f2cdc89fb61b035cac5b4491ba2e6159eeb3a4195d5cef1d64ae4b",
30
+ "bytes": 18219
31
  },
32
  "comparison.json": {
33
+ "sha256": "2125bd27ceadfc247a6187a91e5d94eaad7c81434c355f52ea7d4c9610c447ca",
34
+ "bytes": 9441
35
  },
36
  "data.json": {
37
+ "sha256": "f4cccf3171db8a2aa47c1c1d832cbd90096a22b4c953055a4797978e26bf47ad",
38
+ "bytes": 462685
39
  },
40
  "episodes.csv": {
41
+ "sha256": "6ceccf563f6fa824a050c1d21127c4be4066a5c6618240330a90b33e2a2fcf55",
42
+ "bytes": 3409
43
  },
44
  "episodes.json": {
45
+ "sha256": "467efa0b909e3712c44e8ff3c3b84176a476320a7583f2358b4efc5330eb7378",
46
+ "bytes": 304337
47
  },
48
  "failure-reviews.json": {
49
  "sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
 
62
  "bytes": 1091
63
  },
64
  "protocol.json": {
65
+ "sha256": "3f843f0f197f83d8d7a0b1a37b1dde46fb4be60ff43e7e654c7d74bd73defdac",
66
+ "bytes": 1084
67
  },
68
  "style.css": {
69
  "sha256": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
70
  "bytes": 16560
71
  },
72
  "summary.json": {
73
+ "sha256": "dbb6cb38fc6889cac4e443fddf929d348984e37dc7944f59ca44a38aa6535bc2",
74
  "bytes": 898
75
  },
76
  "task-index.json": {
77
+ "sha256": "db5a5e2992f392fd2f76b8f73a9d99abffe9156bf2722a9926687abd90af3592",
78
+ "bytes": 133978
79
  },
80
  "manifest.json": {
81
+ "sha256": "1d02f419af5e29225ced689bef2ecbfccc44103ad6b4fbb9fde09dea276d0643",
82
  "bytes": 1671
83
  }
84
  }
summary.json CHANGED
@@ -1,36 +1,36 @@
1
  {
2
  "planned_tasks": 42,
3
- "published_results": 23,
4
- "pending_tasks": 19,
5
  "families": [
6
  {
7
  "id": "task04",
8
  "name": "RoboDojo",
9
  "total": 42,
10
- "completed": 23,
11
- "pending": 19,
12
- "successes": 18,
13
- "valid_results": 23,
14
- "success_rate": 0.782608695652174,
15
- "input_tokens": 113213163,
16
- "cached_input_tokens": 110735488,
17
- "output_tokens": 458950,
18
- "usage_complete": 23,
19
  "control_frequency_hz": 25,
20
  "max_control_steps": 7500,
21
  "preflight_results": 0,
22
- "formal_results": 23,
23
- "modified_results": 18,
24
- "original_results": 5
25
  }
26
  ],
27
  "progress": {
28
- "finished": 23,
29
  "interrupted": 2,
30
- "native_failures": 5,
31
- "native_successes": 18,
32
  "needs_review": 0,
33
- "queued": 10,
34
  "running": 7
35
  },
36
  "interrupted_attempts": 2,
 
1
  {
2
  "planned_tasks": 42,
3
+ "published_results": 29,
4
+ "pending_tasks": 13,
5
  "families": [
6
  {
7
  "id": "task04",
8
  "name": "RoboDojo",
9
  "total": 42,
10
+ "completed": 29,
11
+ "pending": 13,
12
+ "successes": 21,
13
+ "valid_results": 29,
14
+ "success_rate": 0.7241379310344828,
15
+ "input_tokens": 161829633,
16
+ "cached_input_tokens": 158729600,
17
+ "output_tokens": 619780,
18
+ "usage_complete": 29,
19
  "control_frequency_hz": 25,
20
  "max_control_steps": 7500,
21
  "preflight_results": 0,
22
+ "formal_results": 29,
23
+ "modified_results": 23,
24
+ "original_results": 6
25
  }
26
  ],
27
  "progress": {
28
+ "finished": 30,
29
  "interrupted": 2,
30
+ "native_failures": 8,
31
+ "native_successes": 22,
32
  "needs_review": 0,
33
+ "queued": 3,
34
  "running": 7
35
  },
36
  "interrupted_attempts": 2,
task-index.json CHANGED
@@ -107,20 +107,20 @@
107
  "status": "interrupted"
108
  },
109
  "links": {
110
- "native_session": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/session.jsonl",
111
- "trace": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/trajectory.json",
112
- "provider_usage": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/provider-usage.jsonl",
113
- "transcript": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/transcript.json",
114
- "verdict": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/episode.json",
115
- "protocol": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/protocol.json",
116
- "native_goal": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/instructions.json",
117
- "workspace": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/workspace.tar.gz",
118
- "workspace_changes": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/workspace.json",
119
- "owner_journal": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/evidence/episode/evidence/journal/actions.jsonl",
120
- "recording_manifest": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/evidence/episode/evidence/recording/manifest.json",
121
- "usage": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/usage.json",
122
- "analysis": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/analysis.json",
123
- "provenance": "attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/provenance.json"
124
  }
125
  }
126
  ]
@@ -659,20 +659,20 @@
659
  "status": "interrupted"
660
  },
661
  "links": {
662
- "native_session": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/session.jsonl",
663
- "trace": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/trajectory.json",
664
- "provider_usage": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/provider-usage.jsonl",
665
- "transcript": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/transcript.json",
666
- "verdict": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/episode.json",
667
- "protocol": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/protocol.json",
668
- "native_goal": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/instructions.json",
669
- "workspace": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/workspace.tar.gz",
670
- "workspace_changes": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/workspace.json",
671
- "owner_journal": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/evidence/episode/evidence/journal/actions.jsonl",
672
- "recording_manifest": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/evidence/episode/evidence/recording/manifest.json",
673
- "usage": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/usage.json",
674
- "analysis": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/analysis.json",
675
- "provenance": "attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/provenance.json"
676
  }
677
  }
678
  ]
@@ -1418,8 +1418,8 @@
1418
  "catalog_instruction": "Hold the pen holder with one hand and insert all the pens with the other, writing ends down and fully seated inside. Put the holder down upright, release it, and return both arms to their starting poses.",
1419
  "native_instruction": "Hold the pen holder with one hand, place all pens into it with the other hand, then put it back down.",
1420
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
1421
- "status": "pending",
1422
- "episode_id": null,
1423
  "planned_protocol": {
1424
  "episodes": 1,
1425
  "seed": 0,
@@ -1550,8 +1550,8 @@
1550
  },
1551
  "display_slot": "16",
1552
  "display_key": "task04/16",
1553
- "run_status": "running",
1554
- "status_note": "Currently running.",
1555
  "attempt_history": []
1556
  },
1557
  {
@@ -2268,8 +2268,8 @@
2268
  "catalog_instruction": "Place all four objects inside the bottom of the box, with their front sides facing left from the robot's viewpoint and their lengths aligned along the box. Keep the box upright with its long sides running left to right across the table. Release the objects and return both arms to their starting poses.",
2269
  "native_instruction": "Place all the objects into the box with their front sides facing left.",
2270
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2271
- "status": "pending",
2272
- "episode_id": null,
2273
  "planned_protocol": {
2274
  "episodes": 1,
2275
  "seed": 0,
@@ -2332,8 +2332,8 @@
2332
  },
2333
  "display_slot": "25",
2334
  "display_key": "task04/25",
2335
- "run_status": "running",
2336
- "status_note": "Currently running.",
2337
  "attempt_history": []
2338
  },
2339
  {
@@ -2699,8 +2699,8 @@
2699
  "catalog_instruction": "Complete the benchmark task: pour by language.",
2700
  "native_instruction": "Pour the liquid from the first violet bottle into the first black bowl, from the second red bottle into the second white bowl, and from the third turquoise bottle into the third brown bowl. Then reset the robot arm.",
2701
  "instruction_source": "runtime native task.instruction",
2702
- "status": "pending",
2703
- "episode_id": null,
2704
  "planned_protocol": {
2705
  "episodes": 1,
2706
  "seed": 0,
@@ -2712,8 +2712,8 @@
2712
  },
2713
  "display_slot": "30",
2714
  "display_key": "task04/30",
2715
- "run_status": "running",
2716
- "status_note": "Currently running.",
2717
  "attempt_history": []
2718
  },
2719
  {
@@ -2725,8 +2725,8 @@
2725
  "catalog_instruction": "Pour nearly all the liquid into the cup with almost no spillage. Keep the bottle tilted over the cup until the flow stops and the liquid settles, then return it upright, leaving only a small residue inside.",
2726
  "native_instruction": "Pour the liquid from the bottle into the cup.",
2727
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2728
- "status": "pending",
2729
- "episode_id": null,
2730
  "planned_protocol": {
2731
  "episodes": 1,
2732
  "seed": 0,
@@ -2792,8 +2792,8 @@
2792
  },
2793
  "display_slot": "31",
2794
  "display_key": "task04/31",
2795
- "run_status": "running",
2796
- "status_note": "Currently running.",
2797
  "attempt_history": []
2798
  },
2799
  {
@@ -2805,8 +2805,8 @@
2805
  "catalog_instruction": "Starting with the left red button, press and release each red button the number of times shown on its card, confirming that button's count with a press and release of the blue button before moving to the next. Return both arms to their starting poses when finished.",
2806
  "native_instruction": "Press the two red buttons the required number of times according to the number cards, then press the blue button to confirm.",
2807
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2808
- "status": "pending",
2809
- "episode_id": null,
2810
  "planned_protocol": {
2811
  "episodes": 1,
2812
  "seed": 0,
@@ -2957,8 +2957,8 @@
2957
  },
2958
  "display_slot": "32",
2959
  "display_key": "task04/32",
2960
- "run_status": "running",
2961
- "status_note": "Currently running.",
2962
  "attempt_history": []
2963
  },
2964
  {
@@ -2970,8 +2970,8 @@
2970
  "catalog_instruction": "Slide the T-shaped block along the table until it neatly matches the gray T-shaped pad in position and orientation. Keep the block on the table and return both arms to their starting poses when finished.",
2971
  "native_instruction": "Push the T-shaped block to align it precisely with the gray T-shaped pad.",
2972
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2973
- "status": "pending",
2974
- "episode_id": null,
2975
  "planned_protocol": {
2976
  "episodes": 1,
2977
  "seed": 0,
@@ -3062,7 +3062,7 @@
3062
  },
3063
  "display_slot": "33",
3064
  "display_key": "task04/33",
3065
- "run_status": "queued",
3066
  "status_note": "Queued for evaluation.",
3067
  "attempt_history": []
3068
  },
@@ -3154,8 +3154,8 @@
3154
  },
3155
  "display_slot": "34",
3156
  "display_key": "task04/34",
3157
- "run_status": "queued",
3158
- "status_note": "Queued for evaluation.",
3159
  "attempt_history": []
3160
  },
3161
  {
@@ -3180,8 +3180,8 @@
3180
  },
3181
  "display_slot": "35",
3182
  "display_key": "task04/35",
3183
- "run_status": "queued",
3184
- "status_note": "Queued for evaluation.",
3185
  "attempt_history": []
3186
  },
3187
  {
@@ -3257,8 +3257,8 @@
3257
  },
3258
  "display_slot": "36",
3259
  "display_key": "task04/36",
3260
- "run_status": "queued",
3261
- "status_note": "Queued for evaluation.",
3262
  "attempt_history": []
3263
  },
3264
  {
@@ -3354,8 +3354,8 @@
3354
  },
3355
  "display_slot": "37",
3356
  "display_key": "task04/37",
3357
- "run_status": "queued",
3358
- "status_note": "Queued for evaluation.",
3359
  "attempt_history": []
3360
  },
3361
  {
@@ -3380,8 +3380,8 @@
3380
  },
3381
  "display_slot": "38",
3382
  "display_key": "task04/38",
3383
- "run_status": "queued",
3384
- "status_note": "Queued for evaluation.",
3385
  "attempt_history": []
3386
  },
3387
  {
@@ -3427,8 +3427,8 @@
3427
  },
3428
  "display_slot": "39",
3429
  "display_key": "task04/39",
3430
- "run_status": "queued",
3431
- "status_note": "Queued for evaluation.",
3432
  "attempt_history": []
3433
  },
3434
  {
 
107
  "status": "interrupted"
108
  },
109
  "links": {
110
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/session.jsonl",
111
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/trajectory.json",
112
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/agent/provider-usage.jsonl",
113
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/transcript.json",
114
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/episode.json",
115
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/protocol.json",
116
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/instructions.json",
117
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/workspace.tar.gz",
118
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/verifier/workspace.json",
119
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/evidence/episode/evidence/journal/actions.jsonl",
120
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/native/evidence/episode/evidence/recording/manifest.json",
121
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/usage.json",
122
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/analysis.json",
123
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/01/01-robodojo-make-toast-codex-seed0-attempt01/provenance.json"
124
  }
125
  }
126
  ]
 
659
  "status": "interrupted"
660
  },
661
  "links": {
662
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/session.jsonl",
663
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/trajectory.json",
664
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/agent/provider-usage.jsonl",
665
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/transcript.json",
666
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/episode.json",
667
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/protocol.json",
668
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/instructions.json",
669
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/workspace.tar.gz",
670
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/verifier/workspace.json",
671
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/evidence/episode/evidence/journal/actions.jsonl",
672
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/native/evidence/episode/evidence/recording/manifest.json",
673
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/usage.json",
674
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/analysis.json",
675
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/attempts/task04/07/07-robodojo-insert-tubes-codex-seed0-attempt01/provenance.json"
676
  }
677
  }
678
  ]
 
1418
  "catalog_instruction": "Hold the pen holder with one hand and insert all the pens with the other, writing ends down and fully seated inside. Put the holder down upright, release it, and return both arms to their starting poses.",
1419
  "native_instruction": "Hold the pen holder with one hand, place all pens into it with the other hand, then put it back down.",
1420
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
1421
+ "status": "completed",
1422
+ "episode_id": "task04-17-seed0-formal",
1423
  "planned_protocol": {
1424
  "episodes": 1,
1425
  "seed": 0,
 
1550
  },
1551
  "display_slot": "16",
1552
  "display_key": "task04/16",
1553
+ "run_status": "finished",
1554
+ "status_note": "Queued for evaluation.",
1555
  "attempt_history": []
1556
  },
1557
  {
 
2268
  "catalog_instruction": "Place all four objects inside the bottom of the box, with their front sides facing left from the robot's viewpoint and their lengths aligned along the box. Keep the box upright with its long sides running left to right across the table. Release the objects and return both arms to their starting poses.",
2269
  "native_instruction": "Place all the objects into the box with their front sides facing left.",
2270
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2271
+ "status": "completed",
2272
+ "episode_id": "task04-29-seed0-formal",
2273
  "planned_protocol": {
2274
  "episodes": 1,
2275
  "seed": 0,
 
2332
  },
2333
  "display_slot": "25",
2334
  "display_key": "task04/25",
2335
+ "run_status": "finished",
2336
+ "status_note": "Queued for evaluation.",
2337
  "attempt_history": []
2338
  },
2339
  {
 
2699
  "catalog_instruction": "Complete the benchmark task: pour by language.",
2700
  "native_instruction": "Pour the liquid from the first violet bottle into the first black bowl, from the second red bottle into the second white bowl, and from the third turquoise bottle into the third brown bowl. Then reset the robot arm.",
2701
  "instruction_source": "runtime native task.instruction",
2702
+ "status": "completed",
2703
+ "episode_id": "task04-35-seed0-formal",
2704
  "planned_protocol": {
2705
  "episodes": 1,
2706
  "seed": 0,
 
2712
  },
2713
  "display_slot": "30",
2714
  "display_key": "task04/30",
2715
+ "run_status": "finished",
2716
+ "status_note": "Queued for evaluation.",
2717
  "attempt_history": []
2718
  },
2719
  {
 
2725
  "catalog_instruction": "Pour nearly all the liquid into the cup with almost no spillage. Keep the bottle tilted over the cup until the flow stops and the liquid settles, then return it upright, leaving only a small residue inside.",
2726
  "native_instruction": "Pour the liquid from the bottle into the cup.",
2727
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2728
+ "status": "completed",
2729
+ "episode_id": "task04-36-seed0-formal",
2730
  "planned_protocol": {
2731
  "episodes": 1,
2732
  "seed": 0,
 
2792
  },
2793
  "display_slot": "31",
2794
  "display_key": "task04/31",
2795
+ "run_status": "finished",
2796
+ "status_note": "Queued for evaluation.",
2797
  "attempt_history": []
2798
  },
2799
  {
 
2805
  "catalog_instruction": "Starting with the left red button, press and release each red button the number of times shown on its card, confirming that button's count with a press and release of the blue button before moving to the next. Return both arms to their starting poses when finished.",
2806
  "native_instruction": "Press the two red buttons the required number of times according to the number cards, then press the blue button to confirm.",
2807
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2808
+ "status": "completed",
2809
+ "episode_id": "task04-38-seed0-formal",
2810
  "planned_protocol": {
2811
  "episodes": 1,
2812
  "seed": 0,
 
2957
  },
2958
  "display_slot": "32",
2959
  "display_key": "task04/32",
2960
+ "run_status": "finished",
2961
+ "status_note": "Queued for evaluation.",
2962
  "attempt_history": []
2963
  },
2964
  {
 
2970
  "catalog_instruction": "Slide the T-shaped block along the table until it neatly matches the gray T-shaped pad in position and orientation. Keep the block on the table and return both arms to their starting poses when finished.",
2971
  "native_instruction": "Push the T-shaped block to align it precisely with the gray T-shaped pad.",
2972
  "instruction_source": "Current reviewed benchmark instruction; each recording retains its measured instruction.",
2973
+ "status": "completed",
2974
+ "episode_id": "task04-39-seed0-formal",
2975
  "planned_protocol": {
2976
  "episodes": 1,
2977
  "seed": 0,
 
3062
  },
3063
  "display_slot": "33",
3064
  "display_key": "task04/33",
3065
+ "run_status": "finished",
3066
  "status_note": "Queued for evaluation.",
3067
  "attempt_history": []
3068
  },
 
3154
  },
3155
  "display_slot": "34",
3156
  "display_key": "task04/34",
3157
+ "run_status": "running",
3158
+ "status_note": "Currently running.",
3159
  "attempt_history": []
3160
  },
3161
  {
 
3180
  },
3181
  "display_slot": "35",
3182
  "display_key": "task04/35",
3183
+ "run_status": "finished",
3184
+ "status_note": "Evaluation finished; media preparation is in progress.",
3185
  "attempt_history": []
3186
  },
3187
  {
 
3257
  },
3258
  "display_slot": "36",
3259
  "display_key": "task04/36",
3260
+ "run_status": "running",
3261
+ "status_note": "Currently running.",
3262
  "attempt_history": []
3263
  },
3264
  {
 
3354
  },
3355
  "display_slot": "37",
3356
  "display_key": "task04/37",
3357
+ "run_status": "running",
3358
+ "status_note": "Currently running.",
3359
  "attempt_history": []
3360
  },
3361
  {
 
3380
  },
3381
  "display_slot": "38",
3382
  "display_key": "task04/38",
3383
+ "run_status": "running",
3384
+ "status_note": "Currently running.",
3385
  "attempt_history": []
3386
  },
3387
  {
 
3427
  },
3428
  "display_slot": "39",
3429
  "display_key": "task04/39",
3430
+ "run_status": "running",
3431
+ "status_note": "Currently running.",
3432
  "attempt_history": []
3433
  },
3434
  {