yzzhao commited on
Commit
44ec674
·
verified ·
1 Parent(s): 414d668

Publish codex SIMPLE seed-0 evaluation evidence

Browse files
Files changed (10) hide show
  1. README.md +2 -2
  2. REPORT.md +2 -2
  3. data.json +299 -33
  4. episodes.csv +1 -1
  5. episodes.json +266 -0
  6. manifest.json +8 -8
  7. protocol.json +10 -10
  8. publication-manifest.json +15 -15
  9. summary.json +17 -17
  10. task-index.json +5 -5
README.md CHANGED
@@ -10,7 +10,7 @@ pinned: false
10
 
11
  # Codex Benchmark
12
 
13
- 79 published results across 5 simulator families. This snapshot adds 7 of 12 authorized SIMPLE G1 results.
14
 
15
  | Environment | Published / selected | Native successes / valid results |
16
  | --- | ---: | ---: |
@@ -18,7 +18,7 @@ pinned: false
18
  | LIBERO Long | 10 / 10 | 9 / 10 |
19
  | RoboTwin | 10 / 10 | 7 / 10 |
20
  | RoboDojo | 42 / 42 | 31 / 42 |
21
- | SIMPLE G1 | 7 / 12 | 2 / 7 |
22
 
23
  SIMPLE uses GPT-6 Astra with high reasoning effort and default service tier, one fresh seed-0 episode per task, 10000 native control intervals at 50 Hz, stepped MuJoCo physics with Isaac rendering, domain randomization level 0, arm gravity compensation, stop-on-success, recording and an 8-hour wall-clock timeout. Request retries remain within the same session; whole episodes are never retried automatically.
24
 
 
10
 
11
  # Codex Benchmark
12
 
13
+ 80 published results across 5 simulator families. This snapshot adds 8 of 12 authorized SIMPLE G1 results.
14
 
15
  | Environment | Published / selected | Native successes / valid results |
16
  | --- | ---: | ---: |
 
18
  | LIBERO Long | 10 / 10 | 9 / 10 |
19
  | RoboTwin | 10 / 10 | 7 / 10 |
20
  | RoboDojo | 42 / 42 | 31 / 42 |
21
+ | SIMPLE G1 | 8 / 12 | 2 / 8 |
22
 
23
  SIMPLE uses GPT-6 Astra with high reasoning effort and default service tier, one fresh seed-0 episode per task, 10000 native control intervals at 50 Hz, stepped MuJoCo physics with Isaac rendering, domain randomization level 0, arm gravity compensation, stop-on-success, recording and an 8-hour wall-clock timeout. Request retries remain within the same session; whole episodes are never retried automatically.
24
 
REPORT.md CHANGED
@@ -1,6 +1,6 @@
1
  # Codex Benchmark
2
 
3
- 79 published results across 5 simulator families. This snapshot adds 7 of 12 authorized SIMPLE G1 results.
4
 
5
  | Environment | Published / selected | Native successes / valid results |
6
  | --- | ---: | ---: |
@@ -8,7 +8,7 @@
8
  | LIBERO Long | 10 / 10 | 9 / 10 |
9
  | RoboTwin | 10 / 10 | 7 / 10 |
10
  | RoboDojo | 42 / 42 | 31 / 42 |
11
- | SIMPLE G1 | 7 / 12 | 2 / 7 |
12
 
13
  SIMPLE uses GPT-6 Astra with high reasoning effort and default service tier, one fresh seed-0 episode per task, 10000 native control intervals at 50 Hz, stepped MuJoCo physics with Isaac rendering, domain randomization level 0, arm gravity compensation, stop-on-success, recording and an 8-hour wall-clock timeout. Request retries remain within the same session; whole episodes are never retried automatically.
14
 
 
1
  # Codex Benchmark
2
 
3
+ 80 published results across 5 simulator families. This snapshot adds 8 of 12 authorized SIMPLE G1 results.
4
 
5
  | Environment | Published / selected | Native successes / valid results |
6
  | --- | ---: | ---: |
 
8
  | LIBERO Long | 10 / 10 | 9 / 10 |
9
  | RoboTwin | 10 / 10 | 7 / 10 |
10
  | RoboDojo | 42 / 42 | 31 / 42 |
11
+ | SIMPLE G1 | 8 / 12 | 2 / 8 |
12
 
13
  SIMPLE uses GPT-6 Astra with high reasoning effort and default service tier, one fresh seed-0 episode per task, 10000 native control intervals at 50 Hz, stepped MuJoCo physics with Isaac rendering, domain randomization level 0, arm gravity compensation, stop-on-success, recording and an 8-hour wall-clock timeout. Request retries remain within the same session; whole episodes are never retried automatically.
14
 
data.json CHANGED
@@ -6,7 +6,7 @@
6
  "benchmark_complete": false,
7
  "edition": "codex-simple-astra-high-seed0",
8
  "created_at": "2026-10-09T00:09:42.989838+00:00",
9
- "updated_at": "2026-10-10T01:18:00.915385+00:00",
10
  "model": "gpt-6-astra",
11
  "effort": "high",
12
  "seed": 0,
@@ -17,8 +17,8 @@
17
  },
18
  "summary": {
19
  "planned_tasks": 84,
20
- "published_results": 79,
21
- "pending_tasks": 5,
22
  "families": [
23
  {
24
  "id": "task01",
@@ -105,40 +105,40 @@
105
  "id": "task06",
106
  "name": "SIMPLE G1",
107
  "total": 12,
108
- "completed": 7,
109
- "pending": 5,
110
  "successes": 2,
111
- "valid_results": 7,
112
- "success_rate": 0.2857142857142857,
113
- "usage_complete": 7,
114
  "control_frequency_hz": 50,
115
  "max_control_steps": 10000,
116
  "preflight_results": 0,
117
- "formal_results": 7,
118
  "modified_results": 0,
119
- "original_results": 7,
120
- "input_tokens": 91436538,
121
- "cached_input_tokens": 90472576,
122
- "output_tokens": 245134
123
  }
124
  ],
125
  "interrupted_attempts": 4,
126
  "usage_incomplete_tasks": [],
127
  "execution_incomplete_tasks": [],
128
  "deferred_tasks": [],
129
- "new_evaluations": 7,
130
  "progress": {
131
- "finished": 79,
132
  "running": 3,
133
- "queued": 2,
134
  "needs_review": 0,
135
  "interrupted": 0,
136
  "native_successes": 51,
137
- "native_failures": 28
138
  },
139
  "simple_campaign": {
140
  "planned": 12,
141
- "published": 7,
142
  "agent": "codex",
143
  "model": "gpt-6-astra",
144
  "effort": "high",
@@ -232,21 +232,21 @@
232
  "id": "task06",
233
  "name": "SIMPLE G1",
234
  "total": 12,
235
- "completed": 7,
236
- "pending": 5,
237
  "successes": 2,
238
- "valid_results": 7,
239
- "success_rate": 0.2857142857142857,
240
- "usage_complete": 7,
241
  "control_frequency_hz": 50,
242
  "max_control_steps": 10000,
243
  "preflight_results": 0,
244
- "formal_results": 7,
245
  "modified_results": 0,
246
- "original_results": 7,
247
- "input_tokens": 91436538,
248
- "cached_input_tokens": 90472576,
249
- "output_tokens": 245134
250
  }
251
  ],
252
  "tasks": [
@@ -6483,10 +6483,10 @@
6483
  "catalog_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
6484
  "native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
6485
  "instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
6486
- "status": "pending",
6487
- "episode_id": null,
6488
- "run_status": "running",
6489
- "status_note": "Authorized episode is queued, running, or awaiting evidence review.",
6490
  "planned_protocol": {
6491
  "episodes": 1,
6492
  "seed": 0,
@@ -6605,7 +6605,7 @@
6605
  "instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
6606
  "status": "pending",
6607
  "episode_id": null,
6608
- "run_status": "queued",
6609
  "status_note": "Authorized episode is queued, running, or awaiting evidence review.",
6610
  "planned_protocol": {
6611
  "episodes": 1,
@@ -26741,6 +26741,272 @@
26741
  "selected_for_formal_metrics": true,
26742
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/06/seed-0/"
26743
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
26744
  {
26745
  "id": "task06-08-seed0-formal",
26746
  "task_key": "task06/08",
 
6
  "benchmark_complete": false,
7
  "edition": "codex-simple-astra-high-seed0",
8
  "created_at": "2026-10-09T00:09:42.989838+00:00",
9
+ "updated_at": "2026-10-10T01:24:39.237016+00:00",
10
  "model": "gpt-6-astra",
11
  "effort": "high",
12
  "seed": 0,
 
17
  },
18
  "summary": {
19
  "planned_tasks": 84,
20
+ "published_results": 80,
21
+ "pending_tasks": 4,
22
  "families": [
23
  {
24
  "id": "task01",
 
105
  "id": "task06",
106
  "name": "SIMPLE G1",
107
  "total": 12,
108
+ "completed": 8,
109
+ "pending": 4,
110
  "successes": 2,
111
+ "valid_results": 8,
112
+ "success_rate": 0.25,
113
+ "usage_complete": 8,
114
  "control_frequency_hz": 50,
115
  "max_control_steps": 10000,
116
  "preflight_results": 0,
117
+ "formal_results": 8,
118
  "modified_results": 0,
119
+ "original_results": 8,
120
+ "input_tokens": 108175382,
121
+ "cached_input_tokens": 106998016,
122
+ "output_tokens": 283858
123
  }
124
  ],
125
  "interrupted_attempts": 4,
126
  "usage_incomplete_tasks": [],
127
  "execution_incomplete_tasks": [],
128
  "deferred_tasks": [],
129
+ "new_evaluations": 8,
130
  "progress": {
131
+ "finished": 80,
132
  "running": 3,
133
+ "queued": 1,
134
  "needs_review": 0,
135
  "interrupted": 0,
136
  "native_successes": 51,
137
+ "native_failures": 29
138
  },
139
  "simple_campaign": {
140
  "planned": 12,
141
+ "published": 8,
142
  "agent": "codex",
143
  "model": "gpt-6-astra",
144
  "effort": "high",
 
232
  "id": "task06",
233
  "name": "SIMPLE G1",
234
  "total": 12,
235
+ "completed": 8,
236
+ "pending": 4,
237
  "successes": 2,
238
+ "valid_results": 8,
239
+ "success_rate": 0.25,
240
+ "usage_complete": 8,
241
  "control_frequency_hz": 50,
242
  "max_control_steps": 10000,
243
  "preflight_results": 0,
244
+ "formal_results": 8,
245
  "modified_results": 0,
246
+ "original_results": 8,
247
+ "input_tokens": 108175382,
248
+ "cached_input_tokens": 106998016,
249
+ "output_tokens": 283858
250
  }
251
  ],
252
  "tasks": [
 
6483
  "catalog_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
6484
  "native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
6485
  "instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
6486
+ "status": "completed",
6487
+ "episode_id": "task06-07-seed0-formal",
6488
+ "run_status": "finished",
6489
+ "status_note": "Native result, execution and usage complete.",
6490
  "planned_protocol": {
6491
  "episodes": 1,
6492
  "seed": 0,
 
6605
  "instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
6606
  "status": "pending",
6607
  "episode_id": null,
6608
+ "run_status": "running",
6609
  "status_note": "Authorized episode is queued, running, or awaiting evidence review.",
6610
  "planned_protocol": {
6611
  "episodes": 1,
 
26741
  "selected_for_formal_metrics": true,
26742
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/06/seed-0/"
26743
  },
26744
+ {
26745
+ "id": "task06-07-seed0-formal",
26746
+ "task_key": "task06/07",
26747
+ "family": "task06",
26748
+ "slot": "07",
26749
+ "seed": 0,
26750
+ "episode": 1,
26751
+ "phase": "formal",
26752
+ "status": "completed",
26753
+ "success": false,
26754
+ "native_reward": 0.0,
26755
+ "valid": true,
26756
+ "execution": {
26757
+ "reason": null,
26758
+ "status": "finished"
26759
+ },
26760
+ "verdict": {
26761
+ "evidence_valid": true,
26762
+ "steps": 8645,
26763
+ "success": false,
26764
+ "termination": "stopped"
26765
+ },
26766
+ "steps": 8645,
26767
+ "simulation_time_s": null,
26768
+ "wall_time_s": 2551.587007,
26769
+ "model": "gpt-6-astra",
26770
+ "effort": "high",
26771
+ "harness": "codex",
26772
+ "codex_version": "0.160.0",
26773
+ "native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
26774
+ "instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
26775
+ "instruction_policy": "original_native",
26776
+ "usage": {
26777
+ "accounting": "reported-responses",
26778
+ "audit_complete": true,
26779
+ "cache_hit_rate": 0.9872509714529868,
26780
+ "cache_reported_input_tokens": 16738844,
26781
+ "cache_write_input_tokens": 0,
26782
+ "cache_write_reported_input_tokens": 16738844,
26783
+ "cached_input_tokens": 16525440,
26784
+ "completed_turns": 1,
26785
+ "cost_usd": null,
26786
+ "failed_turns": 0,
26787
+ "input_tokens": 16738844,
26788
+ "known_cache_write_input_tokens": 0,
26789
+ "known_cached_input_tokens": 16525440,
26790
+ "known_input_tokens": 16738844,
26791
+ "known_output_tokens": 38724,
26792
+ "known_reasoning_output_tokens": 20279,
26793
+ "output_tokens": 38724,
26794
+ "reasoning_output_tokens": 20279,
26795
+ "reasoning_reported_output_tokens": 38724,
26796
+ "reported_responses": {
26797
+ "cache_reported_input_tokens": 192,
26798
+ "cache_write_input_tokens": 192,
26799
+ "cache_write_reported_input_tokens": 192,
26800
+ "cached_input_tokens": 192,
26801
+ "input_tokens": 192,
26802
+ "output_tokens": 192,
26803
+ "reasoning_output_tokens": 192,
26804
+ "reasoning_reported_output_tokens": 192
26805
+ },
26806
+ "response_count": 192,
26807
+ "response_ids_complete": true,
26808
+ "schema": "rlebench/token-usage/1",
26809
+ "source": "Codex token_usage_record per response",
26810
+ "uncached_input_tokens": 213404,
26811
+ "unidentified_usage_records": 0
26812
+ },
26813
+ "call_activity": {
26814
+ "model_tool_calls": 191,
26815
+ "model_tool_calls_by_name": {
26816
+ "exec": 191
26817
+ },
26818
+ "nested_python_tool_invocations": null,
26819
+ "python_device_rpc_attempts": null,
26820
+ "python_device_rpc_attempts_by_action": null,
26821
+ "python_device_rpc_errors": null,
26822
+ "python_instrumented_model_tool_calls": null,
26823
+ "python_tool_invocations": null,
26824
+ "python_tool_invocations_by_origin": null,
26825
+ "schema": "rlebench/call-activity/1",
26826
+ "source": "Codex native sessions"
26827
+ },
26828
+ "media": {
26829
+ "passed": true,
26830
+ "width": 1280,
26831
+ "height": 360,
26832
+ "duration_s": 43.25,
26833
+ "speed": 4,
26834
+ "source_fps": 10,
26835
+ "output_fps": 20,
26836
+ "recording": {
26837
+ "accepted_samples": 1730,
26838
+ "captured_samples": 1730,
26839
+ "clock": "simulation",
26840
+ "dropped_samples": 0,
26841
+ "encoded_frames": 1730,
26842
+ "end_time_s": 172.90000000001464,
26843
+ "error": null,
26844
+ "experimental": true,
26845
+ "fps": 10,
26846
+ "received_samples": 1730,
26847
+ "schema": "roboenv/recording/1",
26848
+ "state": "closed",
26849
+ "status": "complete",
26850
+ "views": [
26851
+ {
26852
+ "fov_y": 45.0,
26853
+ "height": 360,
26854
+ "name": "camera_head_left",
26855
+ "pose": null,
26856
+ "source": "camera_head_left",
26857
+ "width": 640
26858
+ },
26859
+ {
26860
+ "fov_y": 45.0,
26861
+ "height": 360,
26862
+ "name": "camera_head_right",
26863
+ "pose": null,
26864
+ "source": "camera_head_right",
26865
+ "width": 640
26866
+ }
26867
+ ]
26868
+ },
26869
+ "view_names": [
26870
+ "camera_head_left",
26871
+ "camera_head_right"
26872
+ ],
26873
+ "sha256": "95adbcd6adbca86aa7e6a6cc90c6fe2930fa1d430ac07da60e5756b01ad3dd3b",
26874
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
26875
+ },
26876
+ "analysis": {
26877
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
26878
+ "duration": "\u4f7f\u7528 8,645 / 10,000 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
26879
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
26880
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
26881
+ },
26882
+ "provenance": {
26883
+ "sources": {
26884
+ "RLE-Bench-inhouse": {
26885
+ "build_inputs": [
26886
+ "pyproject.toml",
26887
+ "src",
26888
+ "tasks",
26889
+ "README.md",
26890
+ "Makefile",
26891
+ "tests",
26892
+ "docs"
26893
+ ],
26894
+ "content_sha256": "5b1428b96104cb17b52215bce0ec33b880c5c591d01d4ecc2867cdc647940bda",
26895
+ "dirty": true,
26896
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
26897
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
26898
+ },
26899
+ "RoboEnv": {
26900
+ "build_inputs": [
26901
+ "pyproject.toml",
26902
+ "README.md",
26903
+ "src",
26904
+ "runtime/pyproject.toml",
26905
+ "runtime/README.md",
26906
+ "runtime/src",
26907
+ "runtime/environments.json",
26908
+ "runtime/locks",
26909
+ "catalog",
26910
+ "upstreams.lock.json",
26911
+ "third_party/patches",
26912
+ "docs/validation"
26913
+ ],
26914
+ "content_sha256": "5e90464ff5b67edba2a8e653e68cba5f7065c19f92f34a29689703d61e04d305",
26915
+ "dirty": true,
26916
+ "revision": "b73356c09507eeffaf9ca9e90b668ef0b7f1fa92",
26917
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
26918
+ },
26919
+ "kinex": {
26920
+ "build_inputs": [
26921
+ "package.json",
26922
+ "package-lock.json",
26923
+ ".nvmrc",
26924
+ "tsconfig.json",
26925
+ "VERSION",
26926
+ "src",
26927
+ "packages/core",
26928
+ "packages/setup",
26929
+ "script",
26930
+ "assets",
26931
+ "bin"
26932
+ ],
26933
+ "content_sha256": "70f5107905efa45cace74e04b630660aaee4c80e1db2e3d540d6c81e0f4d0d38",
26934
+ "dirty": false,
26935
+ "revision": "caac19a8a36272972f762e0f74cfe381e8500048",
26936
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/kinex"
26937
+ }
26938
+ },
26939
+ "job": "task06-07-pick-and-place-and-hug-container-codex-seed0-attempt01",
26940
+ "attempt": 1,
26941
+ "harness": "stock Codex CLI",
26942
+ "control_interface": "public RoboEnv SDK and CLI",
26943
+ "kinex_agent_runtime_used": false,
26944
+ "gpu_index": 3,
26945
+ "gpu_model": "NVIDIA L40S",
26946
+ "campaign_dispatch_concurrency_limit": 5,
26947
+ "concurrency_note": "One GPU per active SIMPLE episode; Kinex and Codex share a five-GPU dispatch pool.",
26948
+ "codex_version": "0.160.0",
26949
+ "model": "gpt-6-astra",
26950
+ "effort": "high",
26951
+ "service_tier": "default",
26952
+ "fresh_session": true,
26953
+ "source_jobs": [],
26954
+ "resume_trajectory": false,
26955
+ "imported_skills": [],
26956
+ "automatic_harbor_retries": 0,
26957
+ "request_policy": {
26958
+ "max_request_retries": 50,
26959
+ "configuration": "explicit retry50 SSE",
26960
+ "usage_accounting": "reported-responses"
26961
+ },
26962
+ "classification": "formal",
26963
+ "measured_images": {
26964
+ "agent": "sha256:3fb135c4db8092900969023cfe52e4bddfc8960b6296318fda47c72e80e5f667",
26965
+ "task": "sha256:730c5fc8ca66ce7bb07635943de71bd443d99c4f6ce962e01d35e90a3aec5c28"
26966
+ },
26967
+ "session_original_sha256": "f57a16755d9b599e6a03f0620b4092819f73f48643c586385c5d54ffa8ff4115",
26968
+ "protocol_sha256": "6e24ceebccc59c6e91caf708994c458548bbb4c3539b206b9d85eebcb7ea6266"
26969
+ },
26970
+ "links": {
26971
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/session.jsonl",
26972
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/trajectory.json",
26973
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/provider-usage.jsonl",
26974
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/transcript.json",
26975
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/episode.json",
26976
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/protocol.json",
26977
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/instructions.json",
26978
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/workspace.tar.gz",
26979
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/workspace.json",
26980
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
26981
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/recording/manifest.json",
26982
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/usage.json",
26983
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/analysis.json",
26984
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/provenance.json",
26985
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/video.mp4",
26986
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/poster.jpg",
26987
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/media-validation.json",
26988
+ "final_observation": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/final-observation.json"
26989
+ },
26990
+ "resources": [
26991
+ {
26992
+ "name": "tools/robot.py",
26993
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/resources/tools/robot.py",
26994
+ "kind": "Created during this episode; final workspace snapshot."
26995
+ },
26996
+ {
26997
+ "name": "memos/g1_simple_manipulation.md",
26998
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/resources/memos/g1_simple_manipulation.md",
26999
+ "kind": "Created during this episode; final workspace snapshot."
27000
+ }
27001
+ ],
27002
+ "session_counts": {
27003
+ "visible_events": 414,
27004
+ "observed_images": 108,
27005
+ "tool_errors": 1
27006
+ },
27007
+ "selected_for_formal_metrics": true,
27008
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/"
27009
+ },
27010
  {
27011
  "id": "task06-08-seed0-formal",
27012
  "task_key": "task06/08",
episodes.csv CHANGED
@@ -77,7 +77,7 @@ task06/03,simple/bend-pick,completed,0,True,False,9768,3956.495044,24248209,2404
77
  task06/04,simple/bend-pick-and-place,completed,0,True,False,8560,3261.927668,31457829,31121920,57128
78
  task06/05,simple/bend-handover,completed,0,True,False,9150,3541.240323,24778430,24583808,75261
79
  task06/06,simple/handover,completed,0,True,False,4040,1176.656816,5562825,5475584,17747
80
- task06/07,simple/pick-and-place-and-hug-container,pending,0,,,,,,,
81
  task06/08,simple/close-door,completed,0,True,True,1277,369.21498,1340798,1294208,6729
82
  task06/09,simple/open-oven,pending,0,,,,,,,
83
  task06/10,simple/open-faucet,pending,0,,,,,,,
 
77
  task06/04,simple/bend-pick-and-place,completed,0,True,False,8560,3261.927668,31457829,31121920,57128
78
  task06/05,simple/bend-handover,completed,0,True,False,9150,3541.240323,24778430,24583808,75261
79
  task06/06,simple/handover,completed,0,True,False,4040,1176.656816,5562825,5475584,17747
80
+ task06/07,simple/pick-and-place-and-hug-container,completed,0,True,False,8645,2551.587007,16738844,16525440,38724
81
  task06/08,simple/close-door,completed,0,True,True,1277,369.21498,1340798,1294208,6729
82
  task06/09,simple/open-oven,pending,0,,,,,,,
83
  task06/10,simple/open-faucet,pending,0,,,,,,,
episodes.json CHANGED
@@ -20085,6 +20085,272 @@
20085
  "selected_for_formal_metrics": true,
20086
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/06/seed-0/"
20087
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
20088
  {
20089
  "id": "task06-08-seed0-formal",
20090
  "task_key": "task06/08",
 
20085
  "selected_for_formal_metrics": true,
20086
  "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/06/seed-0/"
20087
  },
20088
+ {
20089
+ "id": "task06-07-seed0-formal",
20090
+ "task_key": "task06/07",
20091
+ "family": "task06",
20092
+ "slot": "07",
20093
+ "seed": 0,
20094
+ "episode": 1,
20095
+ "phase": "formal",
20096
+ "status": "completed",
20097
+ "success": false,
20098
+ "native_reward": 0.0,
20099
+ "valid": true,
20100
+ "execution": {
20101
+ "reason": null,
20102
+ "status": "finished"
20103
+ },
20104
+ "verdict": {
20105
+ "evidence_valid": true,
20106
+ "steps": 8645,
20107
+ "success": false,
20108
+ "termination": "stopped"
20109
+ },
20110
+ "steps": 8645,
20111
+ "simulation_time_s": null,
20112
+ "wall_time_s": 2551.587007,
20113
+ "model": "gpt-6-astra",
20114
+ "effort": "high",
20115
+ "harness": "codex",
20116
+ "codex_version": "0.160.0",
20117
+ "native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
20118
+ "instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
20119
+ "instruction_policy": "original_native",
20120
+ "usage": {
20121
+ "accounting": "reported-responses",
20122
+ "audit_complete": true,
20123
+ "cache_hit_rate": 0.9872509714529868,
20124
+ "cache_reported_input_tokens": 16738844,
20125
+ "cache_write_input_tokens": 0,
20126
+ "cache_write_reported_input_tokens": 16738844,
20127
+ "cached_input_tokens": 16525440,
20128
+ "completed_turns": 1,
20129
+ "cost_usd": null,
20130
+ "failed_turns": 0,
20131
+ "input_tokens": 16738844,
20132
+ "known_cache_write_input_tokens": 0,
20133
+ "known_cached_input_tokens": 16525440,
20134
+ "known_input_tokens": 16738844,
20135
+ "known_output_tokens": 38724,
20136
+ "known_reasoning_output_tokens": 20279,
20137
+ "output_tokens": 38724,
20138
+ "reasoning_output_tokens": 20279,
20139
+ "reasoning_reported_output_tokens": 38724,
20140
+ "reported_responses": {
20141
+ "cache_reported_input_tokens": 192,
20142
+ "cache_write_input_tokens": 192,
20143
+ "cache_write_reported_input_tokens": 192,
20144
+ "cached_input_tokens": 192,
20145
+ "input_tokens": 192,
20146
+ "output_tokens": 192,
20147
+ "reasoning_output_tokens": 192,
20148
+ "reasoning_reported_output_tokens": 192
20149
+ },
20150
+ "response_count": 192,
20151
+ "response_ids_complete": true,
20152
+ "schema": "rlebench/token-usage/1",
20153
+ "source": "Codex token_usage_record per response",
20154
+ "uncached_input_tokens": 213404,
20155
+ "unidentified_usage_records": 0
20156
+ },
20157
+ "call_activity": {
20158
+ "model_tool_calls": 191,
20159
+ "model_tool_calls_by_name": {
20160
+ "exec": 191
20161
+ },
20162
+ "nested_python_tool_invocations": null,
20163
+ "python_device_rpc_attempts": null,
20164
+ "python_device_rpc_attempts_by_action": null,
20165
+ "python_device_rpc_errors": null,
20166
+ "python_instrumented_model_tool_calls": null,
20167
+ "python_tool_invocations": null,
20168
+ "python_tool_invocations_by_origin": null,
20169
+ "schema": "rlebench/call-activity/1",
20170
+ "source": "Codex native sessions"
20171
+ },
20172
+ "media": {
20173
+ "passed": true,
20174
+ "width": 1280,
20175
+ "height": 360,
20176
+ "duration_s": 43.25,
20177
+ "speed": 4,
20178
+ "source_fps": 10,
20179
+ "output_fps": 20,
20180
+ "recording": {
20181
+ "accepted_samples": 1730,
20182
+ "captured_samples": 1730,
20183
+ "clock": "simulation",
20184
+ "dropped_samples": 0,
20185
+ "encoded_frames": 1730,
20186
+ "end_time_s": 172.90000000001464,
20187
+ "error": null,
20188
+ "experimental": true,
20189
+ "fps": 10,
20190
+ "received_samples": 1730,
20191
+ "schema": "roboenv/recording/1",
20192
+ "state": "closed",
20193
+ "status": "complete",
20194
+ "views": [
20195
+ {
20196
+ "fov_y": 45.0,
20197
+ "height": 360,
20198
+ "name": "camera_head_left",
20199
+ "pose": null,
20200
+ "source": "camera_head_left",
20201
+ "width": 640
20202
+ },
20203
+ {
20204
+ "fov_y": 45.0,
20205
+ "height": 360,
20206
+ "name": "camera_head_right",
20207
+ "pose": null,
20208
+ "source": "camera_head_right",
20209
+ "width": 640
20210
+ }
20211
+ ]
20212
+ },
20213
+ "view_names": [
20214
+ "camera_head_left",
20215
+ "camera_head_right"
20216
+ ],
20217
+ "sha256": "95adbcd6adbca86aa7e6a6cc90c6fe2930fa1d430ac07da60e5756b01ad3dd3b",
20218
+ "scope": "Native spectator recording postprocessed at 4x; no replay or rerender"
20219
+ },
20220
+ "analysis": {
20221
+ "outcome": "\u539f\u751f\u5224\u5b9a\u672a\u6210\u529f\u3002",
20222
+ "duration": "\u4f7f\u7528 8,645 / 10,000 \u4e2a\u52a8\u4f5c\u5e27\uff1b\u7ed3\u675f\u539f\u56e0\uff1astopped\u3002",
20223
+ "execution": "\u6267\u884c\u6b63\u5e38\u5b8c\u6210\uff0c\u7528\u91cf\u8bb0\u5f55\u5b8c\u6574\u3002",
20224
+ "cause": "\u539f\u751f\u5931\u8d25\u5df2\u7ecf\u786e\u8ba4\uff1b\u5177\u4f53\u5931\u8d25\u673a\u5236\u5c1a\u672a\u9010\u9879\u5b9a\u4f4d\uff0c\u8bf7\u7ed3\u5408\u7ec8\u6001\u8bc1\u636e\u548c\u5b8c\u6574 session \u67e5\u770b\u3002"
20225
+ },
20226
+ "provenance": {
20227
+ "sources": {
20228
+ "RLE-Bench-inhouse": {
20229
+ "build_inputs": [
20230
+ "pyproject.toml",
20231
+ "src",
20232
+ "tasks",
20233
+ "README.md",
20234
+ "Makefile",
20235
+ "tests",
20236
+ "docs"
20237
+ ],
20238
+ "content_sha256": "5b1428b96104cb17b52215bce0ec33b880c5c591d01d4ecc2867cdc647940bda",
20239
+ "dirty": true,
20240
+ "revision": "6e75c8cdf4d737087d727c6627645d0aa7e2d1dd",
20241
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RLE-Bench-inhouse"
20242
+ },
20243
+ "RoboEnv": {
20244
+ "build_inputs": [
20245
+ "pyproject.toml",
20246
+ "README.md",
20247
+ "src",
20248
+ "runtime/pyproject.toml",
20249
+ "runtime/README.md",
20250
+ "runtime/src",
20251
+ "runtime/environments.json",
20252
+ "runtime/locks",
20253
+ "catalog",
20254
+ "upstreams.lock.json",
20255
+ "third_party/patches",
20256
+ "docs/validation"
20257
+ ],
20258
+ "content_sha256": "5e90464ff5b67edba2a8e653e68cba5f7065c19f92f34a29689703d61e04d305",
20259
+ "dirty": true,
20260
+ "revision": "b73356c09507eeffaf9ca9e90b668ef0b7f1fa92",
20261
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/RoboEnv"
20262
+ },
20263
+ "kinex": {
20264
+ "build_inputs": [
20265
+ "package.json",
20266
+ "package-lock.json",
20267
+ ".nvmrc",
20268
+ "tsconfig.json",
20269
+ "VERSION",
20270
+ "src",
20271
+ "packages/core",
20272
+ "packages/setup",
20273
+ "script",
20274
+ "assets",
20275
+ "bin"
20276
+ ],
20277
+ "content_sha256": "70f5107905efa45cace74e04b630660aaee4c80e1db2e3d540d6c81e0f4d0d38",
20278
+ "dirty": false,
20279
+ "revision": "caac19a8a36272972f762e0f74cfe381e8500048",
20280
+ "source": "/home/evaluator/PhyRSI/benchmark-v0.10.0/kinex"
20281
+ }
20282
+ },
20283
+ "job": "task06-07-pick-and-place-and-hug-container-codex-seed0-attempt01",
20284
+ "attempt": 1,
20285
+ "harness": "stock Codex CLI",
20286
+ "control_interface": "public RoboEnv SDK and CLI",
20287
+ "kinex_agent_runtime_used": false,
20288
+ "gpu_index": 3,
20289
+ "gpu_model": "NVIDIA L40S",
20290
+ "campaign_dispatch_concurrency_limit": 5,
20291
+ "concurrency_note": "One GPU per active SIMPLE episode; Kinex and Codex share a five-GPU dispatch pool.",
20292
+ "codex_version": "0.160.0",
20293
+ "model": "gpt-6-astra",
20294
+ "effort": "high",
20295
+ "service_tier": "default",
20296
+ "fresh_session": true,
20297
+ "source_jobs": [],
20298
+ "resume_trajectory": false,
20299
+ "imported_skills": [],
20300
+ "automatic_harbor_retries": 0,
20301
+ "request_policy": {
20302
+ "max_request_retries": 50,
20303
+ "configuration": "explicit retry50 SSE",
20304
+ "usage_accounting": "reported-responses"
20305
+ },
20306
+ "classification": "formal",
20307
+ "measured_images": {
20308
+ "agent": "sha256:3fb135c4db8092900969023cfe52e4bddfc8960b6296318fda47c72e80e5f667",
20309
+ "task": "sha256:730c5fc8ca66ce7bb07635943de71bd443d99c4f6ce962e01d35e90a3aec5c28"
20310
+ },
20311
+ "session_original_sha256": "f57a16755d9b599e6a03f0620b4092819f73f48643c586385c5d54ffa8ff4115",
20312
+ "protocol_sha256": "6e24ceebccc59c6e91caf708994c458548bbb4c3539b206b9d85eebcb7ea6266"
20313
+ },
20314
+ "links": {
20315
+ "native_session": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/session.jsonl",
20316
+ "trace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/trajectory.json",
20317
+ "provider_usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/agent/provider-usage.jsonl",
20318
+ "transcript": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/transcript.json",
20319
+ "verdict": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/episode.json",
20320
+ "protocol": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/protocol.json",
20321
+ "native_goal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/instructions.json",
20322
+ "workspace": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/workspace.tar.gz",
20323
+ "workspace_changes": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/verifier/workspace.json",
20324
+ "owner_journal": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/journal/actions.jsonl",
20325
+ "recording_manifest": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/recording/manifest.json",
20326
+ "usage": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/usage.json",
20327
+ "analysis": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/analysis.json",
20328
+ "provenance": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/provenance.json",
20329
+ "video": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/video.mp4",
20330
+ "poster": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/poster.jpg",
20331
+ "media_check": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/media-validation.json",
20332
+ "final_observation": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/native/evidence/episode/evidence/final-observation.json"
20333
+ },
20334
+ "resources": [
20335
+ {
20336
+ "name": "tools/robot.py",
20337
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/resources/tools/robot.py",
20338
+ "kind": "Created during this episode; final workspace snapshot."
20339
+ },
20340
+ {
20341
+ "name": "memos/g1_simple_manipulation.md",
20342
+ "file": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/resources/memos/g1_simple_manipulation.md",
20343
+ "kind": "Created during this episode; final workspace snapshot."
20344
+ }
20345
+ ],
20346
+ "session_counts": {
20347
+ "visible_events": 414,
20348
+ "observed_images": 108,
20349
+ "tool_errors": 1
20350
+ },
20351
+ "selected_for_formal_metrics": true,
20352
+ "asset_base": "https://huggingface.co/datasets/RLE-Bench/codex-benchmark/resolve/main/episodes/task06/07/seed-0/"
20353
+ },
20354
  {
20355
  "id": "task06-08-seed0-formal",
20356
  "task_key": "task06/08",
manifest.json CHANGED
@@ -1,8 +1,8 @@
1
  {
2
  ".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
3
  "DATA_FORMAT.md": "e2fd4f6b5c8be9d9e008db7d0c1f5d52b419d68328a1c86625c9f67c49b03403",
4
- "README.md": "758ffe296585c8984d3e8444de92d9816752217e2dcdeaef8f608e258e88d552",
5
- "REPORT.md": "dffa31e9f6c868a6916559fc813f56e62b35808e007f5b964d24e79772e25c06",
6
  "THIRD_PARTY_NOTICES.md": "8aa8e3db100d7a570931c5d87f44f78705b42cf17e8147016b35e9aeddb06bba",
7
  "app.js": "eb013cc9d24382410953a1a8c81e446a6592a81b32887749d17bfc863936556b",
8
  "attempt-history.json": "8555f458c1dd362f5a07447342994fddf265ff23bbc40e9bd6f9984a892ddb4e",
@@ -19,9 +19,9 @@
19
  "catalog/task06/11-push-office-chair/task.yaml": "ab19bf9b7cec8992e2dcc1fb30ad9f6ce0f8343063b44777b2b68afc39d7f426",
20
  "catalog/task06/12-open-trash-can/task.yaml": "8d8b4b0ec848f91283e73416d257f5b0f816ac4b872fccd10b213e74d953085c",
21
  "comparison.json": "487d479f3d03e72efaf3c19c1e63f5894bcfea2954cb1cc46327c379b5315ad3",
22
- "data.json": "778e49376f6133719890725e55fa81fde807327874d39f9d7741dad7a5808d8f",
23
- "episodes.csv": "3d88b62d09ca6f64a6932942d21bbd46142f44c5403ab465abcbaf70227c3cd5",
24
- "episodes.json": "66b5a163d5b639b1d3bd2e4232083e030d5b6e5e41caccf560a8617e973de2ee",
25
  "failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
26
  "favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
27
  "index.html": "524b3f71cc36152a08125212ad1c9ae2ca30601907b378f69cce4d34390d0ee3",
@@ -30,8 +30,8 @@
30
  "licenses/RoboTwin-LICENSE.txt": "c695d421718e54e6f3a60858f0c601125c3ecc60c199332c15947360561e2a9f",
31
  "licenses/robocasa-LICENSE.txt": "5da18670b3f00c59847b1ded9c28dee59940d963b1e03b528b0108d9c5a09885",
32
  "licenses/robosuite-LICENSE.txt": "177978cbece0a4c454c2aaec5b3f145b39270814874c43109da9e829c39d9cba",
33
- "protocol.json": "a3288345609e793384a1cf99eea6b07a451a302282082dc4c1e0714ea0b35859",
34
  "style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
35
- "summary.json": "71ec7e4438a0333ea048aad0d3bc8c4e5f558b03a885afdda327d7a74ce7b5c3",
36
- "task-index.json": "26fea0071532a894ffee5fe88af6a7e80a8320bd0b1c6026c742ef5947a04f72"
37
  }
 
1
  {
2
  ".gitattributes": "51dec59a208c8656ce7d6614b7b547022f7a4720b1cdad6a20aeeda717ec090b",
3
  "DATA_FORMAT.md": "e2fd4f6b5c8be9d9e008db7d0c1f5d52b419d68328a1c86625c9f67c49b03403",
4
+ "README.md": "a69c662b67b32d762b71db8bfb458f33823a2ac65814cb13bf998f99c90cf53a",
5
+ "REPORT.md": "e8c55b5928c2ded32606f1552fd73634247d7921658cab0edbc320e0ffc84445",
6
  "THIRD_PARTY_NOTICES.md": "8aa8e3db100d7a570931c5d87f44f78705b42cf17e8147016b35e9aeddb06bba",
7
  "app.js": "eb013cc9d24382410953a1a8c81e446a6592a81b32887749d17bfc863936556b",
8
  "attempt-history.json": "8555f458c1dd362f5a07447342994fddf265ff23bbc40e9bd6f9984a892ddb4e",
 
19
  "catalog/task06/11-push-office-chair/task.yaml": "ab19bf9b7cec8992e2dcc1fb30ad9f6ce0f8343063b44777b2b68afc39d7f426",
20
  "catalog/task06/12-open-trash-can/task.yaml": "8d8b4b0ec848f91283e73416d257f5b0f816ac4b872fccd10b213e74d953085c",
21
  "comparison.json": "487d479f3d03e72efaf3c19c1e63f5894bcfea2954cb1cc46327c379b5315ad3",
22
+ "data.json": "ac940eeb5e78302389d387aff8cccde4e81ecdf4ccea4d6fbb7b54f43e1ccd51",
23
+ "episodes.csv": "7f4d5f46d65689ff303609a98ffa8a34ad875e7cc9b6018628fc368e87070ec9",
24
+ "episodes.json": "41c1c467c7c36745454b1f1da54e2cbd25823af225581cfa168472878f4ea0f9",
25
  "failure-reviews.json": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
26
  "favicon.svg": "46063791218655f70c828ef71a1232a34fd38181da938338badec929ad226819",
27
  "index.html": "524b3f71cc36152a08125212ad1c9ae2ca30601907b378f69cce4d34390d0ee3",
 
30
  "licenses/RoboTwin-LICENSE.txt": "c695d421718e54e6f3a60858f0c601125c3ecc60c199332c15947360561e2a9f",
31
  "licenses/robocasa-LICENSE.txt": "5da18670b3f00c59847b1ded9c28dee59940d963b1e03b528b0108d9c5a09885",
32
  "licenses/robosuite-LICENSE.txt": "177978cbece0a4c454c2aaec5b3f145b39270814874c43109da9e829c39d9cba",
33
+ "protocol.json": "96be82ffa28eedefaa70bff4f7d4daa9e56f70fed407c1beac1ce8b725a00861",
34
  "style.css": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
35
+ "summary.json": "dbb20c916a74b687e14aece840ffe67138889d90d3bd4b7d389593d497c9aede",
36
+ "task-index.json": "1f9542d87396e5e80c842c021f081077ab3f8cbace59c1bf61593b587d9020bc"
37
  }
protocol.json CHANGED
@@ -95,21 +95,21 @@
95
  "id": "task06",
96
  "name": "SIMPLE G1",
97
  "total": 12,
98
- "completed": 7,
99
- "pending": 5,
100
  "successes": 2,
101
- "valid_results": 7,
102
- "success_rate": 0.2857142857142857,
103
- "usage_complete": 7,
104
  "control_frequency_hz": 50,
105
  "max_control_steps": 10000,
106
  "preflight_results": 0,
107
- "formal_results": 7,
108
  "modified_results": 0,
109
- "original_results": 7,
110
- "input_tokens": 91436538,
111
- "cached_input_tokens": 90472576,
112
- "output_tokens": 245134
113
  }
114
  ],
115
  "usage_accounting": "reported-responses",
 
95
  "id": "task06",
96
  "name": "SIMPLE G1",
97
  "total": 12,
98
+ "completed": 8,
99
+ "pending": 4,
100
  "successes": 2,
101
+ "valid_results": 8,
102
+ "success_rate": 0.25,
103
+ "usage_complete": 8,
104
  "control_frequency_hz": 50,
105
  "max_control_steps": 10000,
106
  "preflight_results": 0,
107
+ "formal_results": 8,
108
  "modified_results": 0,
109
+ "original_results": 8,
110
+ "input_tokens": 108175382,
111
+ "cached_input_tokens": 106998016,
112
+ "output_tokens": 283858
113
  }
114
  ],
115
  "usage_accounting": "reported-responses",
publication-manifest.json CHANGED
@@ -10,11 +10,11 @@
10
  "bytes": 1129
11
  },
12
  "README.md": {
13
- "sha256": "758ffe296585c8984d3e8444de92d9816752217e2dcdeaef8f608e258e88d552",
14
  "bytes": 1544
15
  },
16
  "REPORT.md": {
17
- "sha256": "dffa31e9f6c868a6916559fc813f56e62b35808e007f5b964d24e79772e25c06",
18
  "bytes": 1421
19
  },
20
  "THIRD_PARTY_NOTICES.md": {
@@ -82,16 +82,16 @@
82
  "bytes": 134701
83
  },
84
  "data.json": {
85
- "sha256": "778e49376f6133719890725e55fa81fde807327874d39f9d7741dad7a5808d8f",
86
- "bytes": 1154073
87
  },
88
  "episodes.csv": {
89
- "sha256": "3d88b62d09ca6f64a6932942d21bbd46142f44c5403ab465abcbaf70227c3cd5",
90
- "bytes": 8064
91
  },
92
  "episodes.json": {
93
- "sha256": "66b5a163d5b639b1d3bd2e4232083e030d5b6e5e41caccf560a8617e973de2ee",
94
- "bytes": 863439
95
  },
96
  "failure-reviews.json": {
97
  "sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
@@ -126,23 +126,23 @@
126
  "bytes": 1474
127
  },
128
  "protocol.json": {
129
- "sha256": "a3288345609e793384a1cf99eea6b07a451a302282082dc4c1e0714ea0b35859",
130
- "bytes": 5450
131
  },
132
  "style.css": {
133
  "sha256": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
134
  "bytes": 16560
135
  },
136
  "summary.json": {
137
- "sha256": "71ec7e4438a0333ea048aad0d3bc8c4e5f558b03a885afdda327d7a74ce7b5c3",
138
- "bytes": 3221
139
  },
140
  "task-index.json": {
141
- "sha256": "26fea0071532a894ffee5fe88af6a7e80a8320bd0b1c6026c742ef5947a04f72",
142
- "bytes": 229779
143
  },
144
  "manifest.json": {
145
- "sha256": "887c8f7adde67c1b8241fc55b1a7989ec471bc81b41f3318d98864172bf09f9f",
146
  "bytes": 3481
147
  }
148
  }
 
10
  "bytes": 1129
11
  },
12
  "README.md": {
13
+ "sha256": "a69c662b67b32d762b71db8bfb458f33823a2ac65814cb13bf998f99c90cf53a",
14
  "bytes": 1544
15
  },
16
  "REPORT.md": {
17
+ "sha256": "e8c55b5928c2ded32606f1552fd73634247d7921658cab0edbc320e0ffc84445",
18
  "bytes": 1421
19
  },
20
  "THIRD_PARTY_NOTICES.md": {
 
82
  "bytes": 134701
83
  },
84
  "data.json": {
85
+ "sha256": "ac940eeb5e78302389d387aff8cccde4e81ecdf4ccea4d6fbb7b54f43e1ccd51",
86
+ "bytes": 1166026
87
  },
88
  "episodes.csv": {
89
+ "sha256": "7f4d5f46d65689ff303609a98ffa8a34ad875e7cc9b6018628fc368e87070ec9",
90
+ "bytes": 8111
91
  },
92
  "episodes.json": {
93
+ "sha256": "41c1c467c7c36745454b1f1da54e2cbd25823af225581cfa168472878f4ea0f9",
94
+ "bytes": 874883
95
  },
96
  "failure-reviews.json": {
97
  "sha256": "c66a997c9ecd6f7f43f130a8a33cd64c7a5b2049033e9ccde653ee30e4796f23",
 
126
  "bytes": 1474
127
  },
128
  "protocol.json": {
129
+ "sha256": "96be82ffa28eedefaa70bff4f7d4daa9e56f70fed407c1beac1ce8b725a00861",
130
+ "bytes": 5438
131
  },
132
  "style.css": {
133
  "sha256": "dd31b30951c220ebb82bc96224b5e640913945679e42b2e0a42e2c373a329e6c",
134
  "bytes": 16560
135
  },
136
  "summary.json": {
137
+ "sha256": "dbb20c916a74b687e14aece840ffe67138889d90d3bd4b7d389593d497c9aede",
138
+ "bytes": 3209
139
  },
140
  "task-index.json": {
141
+ "sha256": "1f9542d87396e5e80c842c021f081077ab3f8cbace59c1bf61593b587d9020bc",
142
+ "bytes": 229780
143
  },
144
  "manifest.json": {
145
+ "sha256": "f30322dfb31fe1049111c2bd2c2f026804ce81f9a7d025c7c90904e505fa5261",
146
  "bytes": 3481
147
  }
148
  }
summary.json CHANGED
@@ -1,7 +1,7 @@
1
  {
2
  "planned_tasks": 84,
3
- "published_results": 79,
4
- "pending_tasks": 5,
5
  "families": [
6
  {
7
  "id": "task01",
@@ -88,40 +88,40 @@
88
  "id": "task06",
89
  "name": "SIMPLE G1",
90
  "total": 12,
91
- "completed": 7,
92
- "pending": 5,
93
  "successes": 2,
94
- "valid_results": 7,
95
- "success_rate": 0.2857142857142857,
96
- "usage_complete": 7,
97
  "control_frequency_hz": 50,
98
  "max_control_steps": 10000,
99
  "preflight_results": 0,
100
- "formal_results": 7,
101
  "modified_results": 0,
102
- "original_results": 7,
103
- "input_tokens": 91436538,
104
- "cached_input_tokens": 90472576,
105
- "output_tokens": 245134
106
  }
107
  ],
108
  "interrupted_attempts": 4,
109
  "usage_incomplete_tasks": [],
110
  "execution_incomplete_tasks": [],
111
  "deferred_tasks": [],
112
- "new_evaluations": 7,
113
  "progress": {
114
- "finished": 79,
115
  "running": 3,
116
- "queued": 2,
117
  "needs_review": 0,
118
  "interrupted": 0,
119
  "native_successes": 51,
120
- "native_failures": 28
121
  },
122
  "simple_campaign": {
123
  "planned": 12,
124
- "published": 7,
125
  "agent": "codex",
126
  "model": "gpt-6-astra",
127
  "effort": "high",
 
1
  {
2
  "planned_tasks": 84,
3
+ "published_results": 80,
4
+ "pending_tasks": 4,
5
  "families": [
6
  {
7
  "id": "task01",
 
88
  "id": "task06",
89
  "name": "SIMPLE G1",
90
  "total": 12,
91
+ "completed": 8,
92
+ "pending": 4,
93
  "successes": 2,
94
+ "valid_results": 8,
95
+ "success_rate": 0.25,
96
+ "usage_complete": 8,
97
  "control_frequency_hz": 50,
98
  "max_control_steps": 10000,
99
  "preflight_results": 0,
100
+ "formal_results": 8,
101
  "modified_results": 0,
102
+ "original_results": 8,
103
+ "input_tokens": 108175382,
104
+ "cached_input_tokens": 106998016,
105
+ "output_tokens": 283858
106
  }
107
  ],
108
  "interrupted_attempts": 4,
109
  "usage_incomplete_tasks": [],
110
  "execution_incomplete_tasks": [],
111
  "deferred_tasks": [],
112
+ "new_evaluations": 8,
113
  "progress": {
114
+ "finished": 80,
115
  "running": 3,
116
+ "queued": 1,
117
  "needs_review": 0,
118
  "interrupted": 0,
119
  "native_successes": 51,
120
+ "native_failures": 29
121
  },
122
  "simple_campaign": {
123
  "planned": 12,
124
+ "published": 8,
125
  "agent": "codex",
126
  "model": "gpt-6-astra",
127
  "effort": "high",
task-index.json CHANGED
@@ -6232,10 +6232,10 @@
6232
  "catalog_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
6233
  "native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
6234
  "instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
6235
- "status": "pending",
6236
- "episode_id": null,
6237
- "run_status": "running",
6238
- "status_note": "Authorized episode is queued, running, or awaiting evidence review.",
6239
  "planned_protocol": {
6240
  "episodes": 1,
6241
  "seed": 0,
@@ -6354,7 +6354,7 @@
6354
  "instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
6355
  "status": "pending",
6356
  "episode_id": null,
6357
- "run_status": "queued",
6358
  "status_note": "Authorized episode is queued, running, or awaiting evidence review.",
6359
  "planned_protocol": {
6360
  "episodes": 1,
 
6232
  "catalog_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
6233
  "native_instruction": "pick up the apple from table1,hug the container,walk to table2,and place on table2.",
6234
  "instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
6235
+ "status": "completed",
6236
+ "episode_id": "task06-07-seed0-formal",
6237
+ "run_status": "finished",
6238
+ "status_note": "Native result, execution and usage complete.",
6239
  "planned_protocol": {
6240
  "episodes": 1,
6241
  "seed": 0,
 
6354
  "instruction_source": "Unchanged native SIMPLE instruction for the effective default task configuration.",
6355
  "status": "pending",
6356
  "episode_id": null,
6357
+ "run_status": "running",
6358
  "status_note": "Authorized episode is queued, running, or awaiting evidence review.",
6359
  "planned_protocol": {
6360
  "episodes": 1,