outsider86 commited on
Commit
fa1fd40
·
verified ·
1 Parent(s): e408d9d

Upload folder using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +6 -0
  2. fastumi_pickandplace_qwenPI/checkpoints/steps_10000_pytorch_model.pt +3 -0
  3. fastumi_pickandplace_qwenPI/checkpoints/steps_15000_pytorch_model.pt +3 -0
  4. fastumi_pickandplace_qwenPI/checkpoints/steps_5000_pytorch_model.pt +3 -0
  5. fastumi_pickandplace_qwenPI/config.yaml +70 -0
  6. fastumi_pickandplace_qwenPI/dataset_statistics.json +178 -0
  7. fastumi_pickandplace_qwenPI/summary.jsonl +3 -0
  8. fastumi_pickandplace_qwenPI/wandb/wandb/debug-internal.log +7 -0
  9. fastumi_pickandplace_qwenPI/wandb/wandb/debug.log +0 -0
  10. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/config.yaml +168 -0
  11. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/output.log +25 -0
  12. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/requirements.txt +154 -0
  13. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/wandb-metadata.json +159 -0
  14. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/wandb-summary.json +1 -0
  15. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/logs/debug-core.log +13 -0
  16. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/logs/debug-internal.log +8 -0
  17. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/logs/debug.log +0 -0
  18. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/run-hshxz32s.wandb +0 -0
  19. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/config.yaml +169 -0
  20. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/output.log +90 -0
  21. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/requirements.txt +154 -0
  22. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/wandb-metadata.json +159 -0
  23. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/wandb-summary.json +1 -0
  24. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/logs/debug-core.log +13 -0
  25. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/logs/debug-internal.log +8 -0
  26. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/logs/debug.log +0 -0
  27. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/run-9547dyq0.wandb +3 -0
  28. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/files/output.log +133 -0
  29. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/files/requirements.txt +154 -0
  30. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/files/wandb-metadata.json +159 -0
  31. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/logs/debug-core.log +8 -0
  32. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/logs/debug-internal.log +6 -0
  33. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/logs/debug.log +0 -0
  34. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/run-pxng7bwb.wandb +3 -0
  35. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/files/output.log +1 -0
  36. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/files/requirements.txt +154 -0
  37. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/files/wandb-metadata.json +161 -0
  38. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/logs/debug-core.log +7 -0
  39. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/logs/debug-internal.log +6 -0
  40. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/logs/debug.log +0 -0
  41. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/run-rqa9o89w.wandb +0 -0
  42. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/config.yaml +171 -0
  43. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/output.log +102 -0
  44. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/requirements.txt +154 -0
  45. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/wandb-metadata.json +161 -0
  46. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/wandb-summary.json +1 -0
  47. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/logs/debug-core.log +13 -0
  48. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/logs/debug-internal.log +8 -0
  49. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/logs/debug.log +0 -0
  50. fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/run-va5ln8ez.wandb +3 -0
.gitattributes CHANGED
@@ -37,3 +37,9 @@ fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_07354
37
  fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/run-so4tb51q.wandb filter=lfs diff=lfs merge=lfs -text
38
  fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/run-do490bua.wandb filter=lfs diff=lfs merge=lfs -text
39
  fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/run-3mz29zso.wandb filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
37
  fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/run-so4tb51q.wandb filter=lfs diff=lfs merge=lfs -text
38
  fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/run-do490bua.wandb filter=lfs diff=lfs merge=lfs -text
39
  fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/run-3mz29zso.wandb filter=lfs diff=lfs merge=lfs -text
40
+ fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/run-9547dyq0.wandb filter=lfs diff=lfs merge=lfs -text
41
+ fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/run-pxng7bwb.wandb filter=lfs diff=lfs merge=lfs -text
42
+ fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/run-va5ln8ez.wandb filter=lfs diff=lfs merge=lfs -text
43
+ fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_082405-hh8947p1/run-hh8947p1.wandb filter=lfs diff=lfs merge=lfs -text
44
+ fastumi_pickandplace_qwenPI/wandb/wandb/run-20260314_021141-uhbkzqjc/run-uhbkzqjc.wandb filter=lfs diff=lfs merge=lfs -text
45
+ fastumi_pickandplace_qwenPI/wandb/wandb/run-20260315_072213-nyjyyu3l/run-nyjyyu3l.wandb filter=lfs diff=lfs merge=lfs -text
fastumi_pickandplace_qwenPI/checkpoints/steps_10000_pytorch_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f7f9faf75dad85aa79ec82e2312a32c46158d01f80341d28bcbfceb1d024200b
3
+ size 12444522384
fastumi_pickandplace_qwenPI/checkpoints/steps_15000_pytorch_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f4e6f9febe6c1795a63a469f262362f87ce7ef27697246af968fe9a81cb26672
3
+ size 12444522384
fastumi_pickandplace_qwenPI/checkpoints/steps_5000_pytorch_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b748948f95cbe1ca3cd8c5cefdeb36455303833569a15d9ac695892eb2e043e
3
+ size 12444521025
fastumi_pickandplace_qwenPI/config.yaml ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ datasets:
2
+ vla_data:
3
+ CoT_prompt: Your task is {instruction}. To identify the key objects for your task.
4
+ Locate their bounding boxes in [x1,y1,x2,y2] format.
5
+ data_mix: fastumi_pickandplace_ur5_0314
6
+ data_root_dir: playground/Datasets/FastUMI
7
+ dataset_py: lerobot_datasets
8
+ per_device_batch_size: 8
9
+ video_backend: torchvision_av
10
+ framework:
11
+ action_model:
12
+ action_dim: 10
13
+ add_pos_embed: true
14
+ diffusion_model_cfg:
15
+ attention_head_dim: 64
16
+ cross_attention_dim: 2048
17
+ dropout: 0.2
18
+ final_dropout: true
19
+ input_embedding_dim: 2048
20
+ interleave_self_attention: true
21
+ norm_type: ada_norm
22
+ num_attention_heads: 32
23
+ num_layers: 36
24
+ output_dim: 1024
25
+ positional_embeddings: null
26
+ future_action_window_size: 15
27
+ max_seq_len: 1024
28
+ noise_beta_alpha: 1.5
29
+ noise_beta_beta: 1.0
30
+ noise_s: 0.999
31
+ num_inference_timesteps: 4
32
+ num_target_vision_tokens: 32
33
+ num_timestep_buckets: 1000
34
+ past_action_window_size: 0
35
+ state_dim: 10
36
+ name: QwenPI
37
+ qwenvl:
38
+ attn_implementation: flash_attention_2
39
+ base_vlm: playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action
40
+ num_vl_layers: 36
41
+ vl_hidden_dim: 2048
42
+ output_dir: ./results/Checkpoints/fastumi_pickandplace_qwenPI
43
+ run_id: fastumi_pickandplace_qwenPI
44
+ run_root_dir: ./results/Checkpoints
45
+ seed: 42
46
+ trainer:
47
+ eval_interval: 100
48
+ freeze_modules: null
49
+ gradient_accumulation_steps: 1
50
+ gradient_clipping: 1.0
51
+ is_resume: true
52
+ learning_rate:
53
+ action_model: 0.0001
54
+ base: 2.5e-05
55
+ qwen_vl_interface: 1.0e-05
56
+ logging_frequency: 50
57
+ lr_scheduler_type: cosine_with_min_lr
58
+ max_train_steps: 20000
59
+ num_warmup_steps: 5000
60
+ optimizer:
61
+ betas:
62
+ - 0.9
63
+ - 0.95
64
+ eps: 1.0e-08
65
+ weight_decay: 1.0e-08
66
+ save_interval: 5000
67
+ scheduler_specific_kwargs:
68
+ min_lr: 1.0e-06
69
+ wandb_entity: 2200011093-peking-university
70
+ wandb_project: starVLA_FastUMI
fastumi_pickandplace_qwenPI/dataset_statistics.json ADDED
@@ -0,0 +1,178 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "action": {
4
+ "mean": [
5
+ -0.00017731482512317598,
6
+ 6.326655420707539e-05,
7
+ -0.0003823576553259045,
8
+ 0.9999275803565979,
9
+ 0.00015219292254187167,
10
+ -7.193408237071708e-05,
11
+ -0.00015269513824023306,
12
+ 0.9998728632926941,
13
+ -0.0001340110320597887,
14
+ 0.23157350718975067
15
+ ],
16
+ "std": [
17
+ 0.0029254474211484194,
18
+ 0.00266967061907053,
19
+ 0.004045490175485611,
20
+ 0.0004587694129540108,
21
+ 0.011643894948065281,
22
+ 0.0005491027259267867,
23
+ 0.011645006015896797,
24
+ 0.0008708133827583353,
25
+ 0.011052136309444904,
26
+ 0.24192321300506592
27
+ ],
28
+ "max": [
29
+ 0.011658241041004658,
30
+ 0.01042273361235857,
31
+ 0.0110545065253973,
32
+ 1.0000001192092896,
33
+ 0.08344583958387375,
34
+ 0.0038863078225404024,
35
+ 0.08254292607307434,
36
+ 1.0000001192092896,
37
+ 0.079257071018219,
38
+ 0.5245640277862549
39
+ ],
40
+ "min": [
41
+ -0.013589359819889069,
42
+ -0.013344451785087585,
43
+ -0.012630008161067963,
44
+ 0.9965112805366516,
45
+ -0.08266565203666687,
46
+ -0.01173458807170391,
47
+ -0.08334038406610489,
48
+ 0.9939950704574585,
49
+ -0.07899841666221619,
50
+ 0.009166000410914421
51
+ ],
52
+ "q01": [
53
+ -0.007945208344608545,
54
+ -0.007611272423528135,
55
+ -0.009110004445537924,
56
+ 0.9968565315008163,
57
+ -0.000308865460101515,
58
+ -0.0032842739787884052,
59
+ -0.07796837285161018,
60
+ 0.9940040111541748,
61
+ -0.07265709199011326,
62
+ 0.009802999906241894
63
+ ],
64
+ "q99": [
65
+ 0.007261873693205412,
66
+ 0.007667668941430744,
67
+ 0.008452737815678119,
68
+ 1.0,
69
+ 0.07798104047775269,
70
+ 0.0002730701782274991,
71
+ 0.0003088487408240318,
72
+ 1.0,
73
+ 0.0004576626507332527,
74
+ 0.5240129828453064
75
+ ],
76
+ "mask": [
77
+ true,
78
+ true,
79
+ true,
80
+ true,
81
+ true,
82
+ true,
83
+ true,
84
+ true,
85
+ true,
86
+ false
87
+ ],
88
+ "norm_modes": [
89
+ "min_max",
90
+ "min_max",
91
+ "min_max",
92
+ "none",
93
+ "none",
94
+ "none",
95
+ "none",
96
+ "none",
97
+ "none",
98
+ "binary"
99
+ ]
100
+ },
101
+ "state": {
102
+ "mean": [
103
+ 0.34298810362815857,
104
+ -0.29047951102256775,
105
+ 0.08274945616722107,
106
+ 0.9994291067123413,
107
+ -0.0020131091587245464,
108
+ -0.0012421202845871449,
109
+ 0.002004049951210618,
110
+ 0.9983507394790649,
111
+ 0.0019875187426805496,
112
+ 0.22578883171081543
113
+ ],
114
+ "std": [
115
+ 0.04991065338253975,
116
+ 0.0422644317150116,
117
+ 0.06237703189253807,
118
+ 0.00019692142086756705,
119
+ 0.039019376039505005,
120
+ 0.0012851571664214134,
121
+ 0.03903917223215103,
122
+ 0.00019175464691510824,
123
+ 0.03722655400633812,
124
+ 0.24103344976902008
125
+ ],
126
+ "max": [
127
+ 0.49783000349998474,
128
+ -0.10034999996423721,
129
+ 0.32138198614120483,
130
+ 0.9993224143981934,
131
+ 0.04183037206530571,
132
+ 0.004732622765004635,
133
+ 0.04235748574137688,
134
+ 0.9986677765846252,
135
+ 0.03970242291688919,
136
+ 0.5245640277862549
137
+ ],
138
+ "min": [
139
+ 0.19434399902820587,
140
+ -0.44425100088119507,
141
+ 0.0019030000548809767,
142
+ 0.999085009098053,
143
+ -0.042528580874204636,
144
+ -0.005806926172226667,
145
+ -0.04183799773454666,
146
+ 0.9984964728355408,
147
+ -0.03961476683616638,
148
+ 0.009166000410914421
149
+ ],
150
+ "q01": [
151
+ 0.23710880234837534,
152
+ -0.3849276554584503,
153
+ 0.00841386997140944,
154
+ 0.9991415911912918,
155
+ -0.041123956106603146,
156
+ -0.004370019193738699,
157
+ -0.04099516797810793,
158
+ 0.9984980225563049,
159
+ -0.03880806788802147,
160
+ 0.009802999906241894
161
+ ],
162
+ "q99": [
163
+ 0.46919367551803587,
164
+ -0.1585090310871601,
165
+ 0.2566446015238762,
166
+ 0.999306601881981,
167
+ 0.04096023611724377,
168
+ 0.002124080888461314,
169
+ 0.04100460018962622,
170
+ 0.9986270666122437,
171
+ 0.03915046207606793,
172
+ 0.5240129828453064
173
+ ]
174
+ },
175
+ "num_transitions": 26330,
176
+ "num_trajectories": 300
177
+ }
178
+ }
fastumi_pickandplace_qwenPI/summary.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {"steps": 5000}
2
+ {"steps": 10000}
3
+ {"steps": 15000}
fastumi_pickandplace_qwenPI/wandb/wandb/debug-internal.log ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-15T07:22:13.699370785Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-15T07:22:14.143677678Z","level":"INFO","msg":"stream: created new stream","id":"nyjyyu3l"}
3
+ {"time":"2026-03-15T07:22:14.143798108Z","level":"INFO","msg":"handler: started","stream_id":"nyjyyu3l"}
4
+ {"time":"2026-03-15T07:22:14.143934497Z","level":"INFO","msg":"stream: started","id":"nyjyyu3l"}
5
+ {"time":"2026-03-15T07:22:14.143996517Z","level":"INFO","msg":"writer: started","stream_id":"nyjyyu3l"}
6
+ {"time":"2026-03-15T07:22:14.143999787Z","level":"INFO","msg":"sender: started","stream_id":"nyjyyu3l"}
7
+ {"time":"2026-03-15T18:13:36.019181495Z","level":"INFO","msg":"stream: closing","id":"nyjyyu3l"}
fastumi_pickandplace_qwenPI/wandb/wandb/debug.log ADDED
File without changes
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/config.yaml ADDED
@@ -0,0 +1,168 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.25.0
4
+ e:
5
+ 9rs4jt49lzudq17zfe4jqbilhwm7rk8l:
6
+ args:
7
+ - --config_yaml
8
+ - ./examples/calvin/train_files/starvla_train_calvin.yaml
9
+ - --framework.name
10
+ - QwenPI
11
+ - --framework.qwenvl.base_vlm
12
+ - playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action
13
+ - --framework.qwenvl.attn_implementation
14
+ - flash_attention_2
15
+ - --framework.action_model.action_dim
16
+ - "10"
17
+ - --framework.action_model.state_dim
18
+ - "10"
19
+ - --framework.action_model.future_action_window_size
20
+ - "15"
21
+ - --framework.action_model.past_action_window_size
22
+ - "0"
23
+ - --framework.action_model.action_hidden_dim
24
+ - "1024"
25
+ - --framework.action_model.hidden_size
26
+ - "1024"
27
+ - --framework.action_model.action_model_type
28
+ - DiT-B
29
+ - --framework.action_model.add_pos_embed
30
+ - "True"
31
+ - --framework.action_model.max_seq_len
32
+ - "1024"
33
+ - --framework.action_model.noise_beta_alpha
34
+ - "1.5"
35
+ - --framework.action_model.noise_beta_beta
36
+ - "1.0"
37
+ - --framework.action_model.noise_s
38
+ - "0.999"
39
+ - --framework.action_model.num_timestep_buckets
40
+ - "1000"
41
+ - --framework.action_model.num_inference_timesteps
42
+ - "4"
43
+ - --framework.action_model.num_target_vision_tokens
44
+ - "32"
45
+ - --datasets.vla_data.data_root_dir
46
+ - playground/Datasets/FastUMI
47
+ - --datasets.vla_data.data_mix
48
+ - fastumi_pickandplace_real_0307
49
+ - --datasets.vla_data.per_device_batch_size
50
+ - "8"
51
+ - --datasets.vla_data.video_backend
52
+ - torchvision_av
53
+ - --trainer.freeze_modules
54
+ - ""
55
+ - --trainer.max_train_steps
56
+ - "20000"
57
+ - --trainer.save_interval
58
+ - "5000"
59
+ - --trainer.logging_frequency
60
+ - "50"
61
+ - --trainer.eval_interval
62
+ - "100"
63
+ - --trainer.gradient_accumulation_steps
64
+ - "1"
65
+ - --trainer.is_resume
66
+ - "true"
67
+ - --run_root_dir
68
+ - ./results/Checkpoints
69
+ - --run_id
70
+ - fastumi_pickandplace_qwenPI
71
+ - --wandb_project
72
+ - starVLA_FastUMI
73
+ - --wandb_entity
74
+ - 2200011093-peking-university
75
+ codePath: starVLA/training/train_starvla.py
76
+ codePathLocal: starVLA/training/train_starvla.py
77
+ cpu_count: 192
78
+ cpu_count_logical: 384
79
+ cudaVersion: "13.1"
80
+ disk:
81
+ /:
82
+ total: "3776651378688"
83
+ used: "136350265344"
84
+ email: wangpc@berkeley.edu
85
+ executable: /home/wangpc/miniconda3/envs/starVLA/bin/python3.10
86
+ git:
87
+ commit: 66b43863ede17a0e3081f822346e95cec52816d0
88
+ remote: https://github.com/Kaiwen-Hong/starVLA.git
89
+ gpu: NVIDIA RTX PRO 6000 Blackwell Server Edition
90
+ gpu_count: 8
91
+ gpu_nvidia:
92
+ - architecture: Blackwell
93
+ cudaCores: 24064
94
+ memoryTotal: "102641958912"
95
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
96
+ uuid: GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673
97
+ - architecture: Blackwell
98
+ cudaCores: 24064
99
+ memoryTotal: "102641958912"
100
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
101
+ uuid: GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5
102
+ - architecture: Blackwell
103
+ cudaCores: 24064
104
+ memoryTotal: "102641958912"
105
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
106
+ uuid: GPU-1f3c0889-b740-5143-e064-afeb49245756
107
+ - architecture: Blackwell
108
+ cudaCores: 24064
109
+ memoryTotal: "102641958912"
110
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
111
+ uuid: GPU-49955a45-509a-e8af-a468-b9ef0449b005
112
+ - architecture: Blackwell
113
+ cudaCores: 24064
114
+ memoryTotal: "102641958912"
115
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
116
+ uuid: GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd
117
+ - architecture: Blackwell
118
+ cudaCores: 24064
119
+ memoryTotal: "102641958912"
120
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
121
+ uuid: GPU-d16b5acf-a100-2d1b-8138-1661892f3d26
122
+ - architecture: Blackwell
123
+ cudaCores: 24064
124
+ memoryTotal: "102641958912"
125
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
126
+ uuid: GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab
127
+ - architecture: Blackwell
128
+ cudaCores: 24064
129
+ memoryTotal: "102641958912"
130
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
131
+ uuid: GPU-747dc32b-b025-76e9-697d-fd33ef441b47
132
+ host: tams02
133
+ memory:
134
+ total: "1081550508032"
135
+ os: Linux-6.8.0-94-generic-x86_64-with-glibc2.39
136
+ program: /scratch/wangpc/starVLA/starVLA/training/train_starvla.py
137
+ python: CPython 3.10.19
138
+ root: ./results/Checkpoints/fastumi_pickandplace_qwenPI/wandb
139
+ startedAt: "2026-03-13T04:12:22.513938Z"
140
+ writerId: 9rs4jt49lzudq17zfe4jqbilhwm7rk8l
141
+ m: []
142
+ python_version: 3.10.19
143
+ t:
144
+ "1":
145
+ - 1
146
+ - 11
147
+ - 41
148
+ - 49
149
+ - 63
150
+ - 71
151
+ - 80
152
+ - 83
153
+ "2":
154
+ - 1
155
+ - 11
156
+ - 41
157
+ - 49
158
+ - 63
159
+ - 71
160
+ - 80
161
+ - 83
162
+ "3":
163
+ - 13
164
+ "4": 3.10.19
165
+ "5": 0.25.0
166
+ "6": 4.57.0
167
+ "12": 0.25.0
168
+ "13": linux-x86_64
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/output.log ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 03/13 [04:12:23] INFO  | >> ***** Training Configuration ***** ]8;id=98246;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=229258;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#338\338]8;;\
2
+   INFO  | >> Total optimization steps = 20000 ]8;id=208496;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=750800;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#339\339]8;;\
3
+   INFO  | >> Per device batch size = 8 ]8;id=471029;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=617889;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#340\340]8;;\
4
+   INFO  | >> Gradient accumulation steps = 1 ]8;id=844962;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=167414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#341\341]8;;\
5
+   INFO  | >> Total batch size = 64 ]8;id=225772;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=800581;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#342\342]8;;\
6
+ 0%| | 1/20000 [00:04<22:16:25, 4.01s/it, data_times=0.608, model_times=3.402]Traceback (most recent call last):
7
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 440, in <module>
8
+ main(cfg)
9
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 410, in main
10
+ trainer.train()
11
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
12
+ step_metrics = self._train_step(batch_vla)
13
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 363, in _train_step
14
+ "action_dit_loss": action_loss.item(),
15
+ KeyboardInterrupt
16
+ [rank0]: Traceback (most recent call last):
17
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 440, in <module>
18
+ [rank0]: main(cfg)
19
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 410, in main
20
+ [rank0]: trainer.train()
21
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
22
+ [rank0]: step_metrics = self._train_step(batch_vla)
23
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 363, in _train_step
24
+ [rank0]: "action_dit_loss": action_loss.item(),
25
+ [rank0]: KeyboardInterrupt
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/wandb-metadata.json ADDED
@@ -0,0 +1,159 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-13T04:12:22.513938Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "./examples/calvin/train_files/starvla_train_calvin.yaml",
8
+ "--framework.name",
9
+ "QwenPI",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.attn_implementation",
13
+ "flash_attention_2",
14
+ "--framework.action_model.action_dim",
15
+ "10",
16
+ "--framework.action_model.state_dim",
17
+ "10",
18
+ "--framework.action_model.future_action_window_size",
19
+ "15",
20
+ "--framework.action_model.past_action_window_size",
21
+ "0",
22
+ "--framework.action_model.action_hidden_dim",
23
+ "1024",
24
+ "--framework.action_model.hidden_size",
25
+ "1024",
26
+ "--framework.action_model.action_model_type",
27
+ "DiT-B",
28
+ "--framework.action_model.add_pos_embed",
29
+ "True",
30
+ "--framework.action_model.max_seq_len",
31
+ "1024",
32
+ "--framework.action_model.noise_beta_alpha",
33
+ "1.5",
34
+ "--framework.action_model.noise_beta_beta",
35
+ "1.0",
36
+ "--framework.action_model.noise_s",
37
+ "0.999",
38
+ "--framework.action_model.num_timestep_buckets",
39
+ "1000",
40
+ "--framework.action_model.num_inference_timesteps",
41
+ "4",
42
+ "--framework.action_model.num_target_vision_tokens",
43
+ "32",
44
+ "--datasets.vla_data.data_root_dir",
45
+ "playground/Datasets/FastUMI",
46
+ "--datasets.vla_data.data_mix",
47
+ "fastumi_pickandplace_real_0307",
48
+ "--datasets.vla_data.per_device_batch_size",
49
+ "8",
50
+ "--datasets.vla_data.video_backend",
51
+ "torchvision_av",
52
+ "--trainer.freeze_modules",
53
+ "",
54
+ "--trainer.max_train_steps",
55
+ "20000",
56
+ "--trainer.save_interval",
57
+ "5000",
58
+ "--trainer.logging_frequency",
59
+ "50",
60
+ "--trainer.eval_interval",
61
+ "100",
62
+ "--trainer.gradient_accumulation_steps",
63
+ "1",
64
+ "--trainer.is_resume",
65
+ "true",
66
+ "--run_root_dir",
67
+ "./results/Checkpoints",
68
+ "--run_id",
69
+ "fastumi_pickandplace_qwenPI",
70
+ "--wandb_project",
71
+ "starVLA_FastUMI",
72
+ "--wandb_entity",
73
+ "2200011093-peking-university"
74
+ ],
75
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
76
+ "codePath": "starVLA/training/train_starvla.py",
77
+ "codePathLocal": "starVLA/training/train_starvla.py",
78
+ "git": {
79
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
80
+ "commit": "66b43863ede17a0e3081f822346e95cec52816d0"
81
+ },
82
+ "email": "wangpc@berkeley.edu",
83
+ "root": "./results/Checkpoints/fastumi_pickandplace_qwenPI/wandb",
84
+ "host": "tams02",
85
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
86
+ "cpu_count": 192,
87
+ "cpu_count_logical": 384,
88
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
89
+ "gpu_count": 8,
90
+ "disk": {
91
+ "/": {
92
+ "total": "3776651378688",
93
+ "used": "136350265344"
94
+ }
95
+ },
96
+ "memory": {
97
+ "total": "1081550508032"
98
+ },
99
+ "gpu_nvidia": [
100
+ {
101
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
102
+ "memoryTotal": "102641958912",
103
+ "cudaCores": 24064,
104
+ "architecture": "Blackwell",
105
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
106
+ },
107
+ {
108
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
109
+ "memoryTotal": "102641958912",
110
+ "cudaCores": 24064,
111
+ "architecture": "Blackwell",
112
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
113
+ },
114
+ {
115
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
116
+ "memoryTotal": "102641958912",
117
+ "cudaCores": 24064,
118
+ "architecture": "Blackwell",
119
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
120
+ },
121
+ {
122
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
123
+ "memoryTotal": "102641958912",
124
+ "cudaCores": 24064,
125
+ "architecture": "Blackwell",
126
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
127
+ },
128
+ {
129
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
130
+ "memoryTotal": "102641958912",
131
+ "cudaCores": 24064,
132
+ "architecture": "Blackwell",
133
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
134
+ },
135
+ {
136
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
137
+ "memoryTotal": "102641958912",
138
+ "cudaCores": 24064,
139
+ "architecture": "Blackwell",
140
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
141
+ },
142
+ {
143
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
144
+ "memoryTotal": "102641958912",
145
+ "cudaCores": 24064,
146
+ "architecture": "Blackwell",
147
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
148
+ },
149
+ {
150
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
151
+ "memoryTotal": "102641958912",
152
+ "cudaCores": 24064,
153
+ "architecture": "Blackwell",
154
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
155
+ }
156
+ ],
157
+ "cudaVersion": "13.1",
158
+ "writerId": "9rs4jt49lzudq17zfe4jqbilhwm7rk8l"
159
+ }
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":6},"_runtime":6}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/logs/debug-core.log ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T04:12:22.578275025Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmp6hhsvuc9/port-1882769.txt","pid":1882769,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-13T04:12:22.579072958Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":1882769}
3
+ {"time":"2026-03-13T04:12:22.579064158Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-1882769-1889351-4240563926/socket","Net":"unix"}}
4
+ {"time":"2026-03-13T04:12:22.754310046Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-13T04:12:22.757113104Z","level":"INFO","msg":"handleInformInit: received","streamId":"hshxz32s","id":"1(@)"}
6
+ {"time":"2026-03-13T04:12:23.096003393Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"hshxz32s","id":"1(@)"}
7
+ {"time":"2026-03-13T04:12:28.429538362Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"af6ag242d2oa"}
8
+ {"time":"2026-03-13T04:12:29.993215437Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"}
9
+ {"time":"2026-03-13T04:12:29.993286866Z","level":"INFO","msg":"server is shutting down"}
10
+ {"time":"2026-03-13T04:12:29.993282846Z","level":"INFO","msg":"connection: closing","id":"1(@)"}
11
+ {"time":"2026-03-13T04:12:29.993379816Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"}
12
+ {"time":"2026-03-13T04:12:29.993391266Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-1882769-1889351-4240563926/socket","Net":"unix"}}
13
+ {"time":"2026-03-13T04:12:30.299408944Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/logs/debug-internal.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T04:12:22.757194123Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-13T04:12:23.095857965Z","level":"INFO","msg":"stream: created new stream","id":"hshxz32s"}
3
+ {"time":"2026-03-13T04:12:23.095946134Z","level":"INFO","msg":"handler: started","stream_id":"hshxz32s"}
4
+ {"time":"2026-03-13T04:12:23.095998773Z","level":"INFO","msg":"stream: started","id":"hshxz32s"}
5
+ {"time":"2026-03-13T04:12:23.096020003Z","level":"INFO","msg":"writer: started","stream_id":"hshxz32s"}
6
+ {"time":"2026-03-13T04:12:23.096030263Z","level":"INFO","msg":"sender: started","stream_id":"hshxz32s"}
7
+ {"time":"2026-03-13T04:12:29.993293716Z","level":"INFO","msg":"stream: closing","id":"hshxz32s"}
8
+ {"time":"2026-03-13T04:12:30.26757045Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/logs/debug.log ADDED
File without changes
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_041222-hshxz32s/run-hshxz32s.wandb ADDED
Binary file (7 Bytes). View file
 
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/config.yaml ADDED
@@ -0,0 +1,169 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.25.0
4
+ e:
5
+ s3rvfu8shkj8mzdb09dke7i4kcnoimei:
6
+ args:
7
+ - --config_yaml
8
+ - ./examples/calvin/train_files/starvla_train_calvin.yaml
9
+ - --framework.name
10
+ - QwenPI
11
+ - --framework.qwenvl.base_vlm
12
+ - playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action
13
+ - --framework.qwenvl.attn_implementation
14
+ - flash_attention_2
15
+ - --framework.action_model.action_dim
16
+ - "10"
17
+ - --framework.action_model.state_dim
18
+ - "10"
19
+ - --framework.action_model.future_action_window_size
20
+ - "15"
21
+ - --framework.action_model.past_action_window_size
22
+ - "0"
23
+ - --framework.action_model.action_hidden_dim
24
+ - "1024"
25
+ - --framework.action_model.hidden_size
26
+ - "1024"
27
+ - --framework.action_model.action_model_type
28
+ - DiT-B
29
+ - --framework.action_model.add_pos_embed
30
+ - "True"
31
+ - --framework.action_model.max_seq_len
32
+ - "1024"
33
+ - --framework.action_model.noise_beta_alpha
34
+ - "1.5"
35
+ - --framework.action_model.noise_beta_beta
36
+ - "1.0"
37
+ - --framework.action_model.noise_s
38
+ - "0.999"
39
+ - --framework.action_model.num_timestep_buckets
40
+ - "1000"
41
+ - --framework.action_model.num_inference_timesteps
42
+ - "4"
43
+ - --framework.action_model.num_target_vision_tokens
44
+ - "32"
45
+ - --datasets.vla_data.data_root_dir
46
+ - playground/Datasets/FastUMI
47
+ - --datasets.vla_data.data_mix
48
+ - fastumi_pickandplace_real_0307
49
+ - --datasets.vla_data.per_device_batch_size
50
+ - "8"
51
+ - --datasets.vla_data.video_backend
52
+ - torchvision_av
53
+ - --trainer.freeze_modules
54
+ - ""
55
+ - --trainer.max_train_steps
56
+ - "20000"
57
+ - --trainer.save_interval
58
+ - "5000"
59
+ - --trainer.logging_frequency
60
+ - "50"
61
+ - --trainer.eval_interval
62
+ - "100"
63
+ - --trainer.gradient_accumulation_steps
64
+ - "1"
65
+ - --trainer.is_resume
66
+ - "true"
67
+ - --run_root_dir
68
+ - ./results/Checkpoints
69
+ - --run_id
70
+ - fastumi_pickandplace_qwenPI
71
+ - --wandb_project
72
+ - starVLA_FastUMI
73
+ - --wandb_entity
74
+ - 2200011093-peking-university
75
+ codePath: starVLA/training/train_starvla.py
76
+ codePathLocal: starVLA/training/train_starvla.py
77
+ cpu_count: 192
78
+ cpu_count_logical: 384
79
+ cudaVersion: "13.1"
80
+ disk:
81
+ /:
82
+ total: "3776651378688"
83
+ used: "136342257664"
84
+ email: wangpc@berkeley.edu
85
+ executable: /home/wangpc/miniconda3/envs/starVLA/bin/python3.10
86
+ git:
87
+ commit: 66b43863ede17a0e3081f822346e95cec52816d0
88
+ remote: https://github.com/Kaiwen-Hong/starVLA.git
89
+ gpu: NVIDIA RTX PRO 6000 Blackwell Server Edition
90
+ gpu_count: 8
91
+ gpu_nvidia:
92
+ - architecture: Blackwell
93
+ cudaCores: 24064
94
+ memoryTotal: "102641958912"
95
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
96
+ uuid: GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673
97
+ - architecture: Blackwell
98
+ cudaCores: 24064
99
+ memoryTotal: "102641958912"
100
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
101
+ uuid: GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5
102
+ - architecture: Blackwell
103
+ cudaCores: 24064
104
+ memoryTotal: "102641958912"
105
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
106
+ uuid: GPU-1f3c0889-b740-5143-e064-afeb49245756
107
+ - architecture: Blackwell
108
+ cudaCores: 24064
109
+ memoryTotal: "102641958912"
110
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
111
+ uuid: GPU-49955a45-509a-e8af-a468-b9ef0449b005
112
+ - architecture: Blackwell
113
+ cudaCores: 24064
114
+ memoryTotal: "102641958912"
115
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
116
+ uuid: GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd
117
+ - architecture: Blackwell
118
+ cudaCores: 24064
119
+ memoryTotal: "102641958912"
120
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
121
+ uuid: GPU-d16b5acf-a100-2d1b-8138-1661892f3d26
122
+ - architecture: Blackwell
123
+ cudaCores: 24064
124
+ memoryTotal: "102641958912"
125
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
126
+ uuid: GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab
127
+ - architecture: Blackwell
128
+ cudaCores: 24064
129
+ memoryTotal: "102641958912"
130
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
131
+ uuid: GPU-747dc32b-b025-76e9-697d-fd33ef441b47
132
+ host: tams02
133
+ memory:
134
+ total: "1081550508032"
135
+ os: Linux-6.8.0-94-generic-x86_64-with-glibc2.39
136
+ program: /scratch/wangpc/starVLA/starVLA/training/train_starvla.py
137
+ python: CPython 3.10.19
138
+ root: ./results/Checkpoints/fastumi_pickandplace_qwenPI/wandb
139
+ startedAt: "2026-03-13T05:21:28.850487Z"
140
+ writerId: s3rvfu8shkj8mzdb09dke7i4kcnoimei
141
+ m: []
142
+ python_version: 3.10.19
143
+ t:
144
+ "1":
145
+ - 1
146
+ - 11
147
+ - 41
148
+ - 49
149
+ - 63
150
+ - 71
151
+ - 80
152
+ - 83
153
+ "2":
154
+ - 1
155
+ - 11
156
+ - 41
157
+ - 49
158
+ - 63
159
+ - 71
160
+ - 80
161
+ - 83
162
+ "3":
163
+ - 13
164
+ - 61
165
+ "4": 3.10.19
166
+ "5": 0.25.0
167
+ "6": 4.57.0
168
+ "12": 0.25.0
169
+ "13": linux-x86_64
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/output.log ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 03/13 [05:21:29] INFO  | >> ***** Training Configuration ***** ]8;id=98246;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=229258;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#338\338]8;;\
2
+   INFO  | >> Total optimization steps = 20000 ]8;id=208496;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=750800;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#339\339]8;;\
3
+   INFO  | >> Per device batch size = 8 ]8;id=471029;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=617889;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#340\340]8;;\
4
+   INFO  | >> Gradient accumulation steps = 1 ]8;id=844962;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=167414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#341\341]8;;\
5
+   INFO  | >> Total batch size = 64 ]8;id=225772;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=800581;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#342\342]8;;\
6
+ 2%|█ | 500/20000 [21:07<13:36:04, 2.51s/it, data_times=0.005, model_times=2.491]
7
+ 03/13 [05:23:38] INFO  | >> Step 50, Loss: {'action_dit_loss': 202009.875, 'data_time': ]8;id=376417;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=888662;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
8
+   0.00016733817756175995, 'model_time': 2.527299334295094, 'learning_rate':  
9
+   1.0000000000000001e-07, 'epoch': 0.14})  
10
+ 03/13 [05:25:44] INFO  | >> Step 100, Loss: {'action_dit_loss': 133267.859375, 'mse_score': ]8;id=45561;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=765179;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
11
+   2.36893367767334, 'data_time': 0.0032478091306984425, 'model_time':  
12
+   2.539055143017322, 'learning_rate': 2.0000000000000002e-07, 'epoch':  
13
+   0.29})  
14
+ 03/13 [05:27:51] INFO  | >> Step 150, Loss: {'action_dit_loss': 2462.653076171875, 'data_time': ]8;id=396922;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=82627;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
15
+   0.00015402119606733322, 'model_time': 2.4931312440894544, 'learning_rate':  
16
+   3.0000000000000004e-07, 'epoch': 0.43})  
17
+ 03/13 [05:29:57] INFO  | >> Step 200, Loss: {'action_dit_loss': 2822.628662109375, 'mse_score': ]8;id=648564;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=928463;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
18
+   15.090812683105469, 'data_time': 0.0030169920064508915, 'model_time':  
19
+   2.534589695278555, 'learning_rate': 4.0000000000000003e-07, 'epoch':  
20
+   0.57})  
21
+ 03/13 [05:32:03] INFO  | >> Step 250, Loss: {'action_dit_loss': 375.4444580078125, 'data_time': ]8;id=738797;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=72933;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
22
+   0.0033243270590901375, 'model_time': 2.506985930260271, 'learning_rate':  
23
+   5.000000000000001e-07, 'epoch': 0.72})  
24
+ 03/13 [05:34:10] INFO  | >> Step 300, Loss: {'action_dit_loss': 216.59744262695312, 'mse_score': ]8;id=303445;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=83667;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
25
+   6.797264862060547, 'data_time': 0.002378092147409916, 'model_time':  
26
+   2.4954333789646626, 'learning_rate': 6.000000000000001e-07, 'epoch':  
27
+   0.86})  
28
+ 03/13 [05:36:17] INFO  | >> Step 350, Loss: {'action_dit_loss': 118.1450424194336, 'data_time': ]8;id=398591;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=291476;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
29
+   0.004711804911494255, 'model_time': 2.535939588211477, 'learning_rate':  
30
+   7.000000000000001e-07, 'epoch': 1.01})  
31
+ 03/13 [05:38:24] INFO  | >> Step 400, Loss: {'action_dit_loss': 32.82866287231445, 'mse_score': ]8;id=170555;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=388162;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
32
+   3.5609485626220705, 'data_time': 0.0002001640386879444, 'model_time':  
33
+   2.75067617604509, 'learning_rate': 8.000000000000001e-07, 'epoch': 1.15})  
34
+ 03/13 [05:40:30] INFO  | >> Step 450, Loss: {'action_dit_loss': 11.134794235229492, 'data_time': ]8;id=735911;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=982153;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
35
+   0.00020384322851896286, 'model_time': 2.523681683000177, 'learning_rate':  
36
+   9.000000000000001e-07, 'epoch': 1.29})  
37
+ 03/13 [05:42:37] INFO  | >> Step 500, Loss: {'action_dit_loss': 46.9814453125, 'mse_score': ]8;id=665822;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=179451;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
38
+   7.125044250488282, 'data_time': 0.004552784841507673, 'model_time':  
39
+   2.4910148638300598, 'learning_rate': 1.0000000000000002e-06, 'epoch':  
40
+   1.44})  
41
+ 03/13 [05:44:43] INFO  | >> Step 550, Loss: {'action_dit_loss': 170.627685546875, 'data_time': ]8;id=484714;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=397887;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
42
+   0.000180800911039114, 'model_time': 2.5395604190416634, 'learning_rate':  
43
+   1.1e-06, 'epoch': 1.58})  
44
+ 03/13 [05:46:50] INFO  | >> Step 600, Loss: {'action_dit_loss': 914.7533569335938, 'mse_score': ]8;id=584004;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=230283;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
45
+   12.847688293457031, 'data_time': 0.002079134341329336, 'model_time':  
46
+   2.504630303941667, 'learning_rate': 1.2000000000000002e-06, 'epoch':  
47
+   1.72})  
48
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 440, in <module>
49
+ main(cfg)
50
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 410, in main
51
+ trainer.train()
52
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
53
+ step_metrics = self._train_step(batch_vla)
54
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 354, in _train_step
55
+ self.accelerator.backward(total_loss)
56
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/accelerate/accelerator.py", line 2351, in backward
57
+ self.deepspeed_engine_wrapped.backward(loss, **kwargs)
58
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/accelerate/utils/deepspeed.py", line 275, in backward
59
+ self.engine.step()
60
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/runtime/engine.py", line 2382, in step
61
+ self.tput_timer.stop(global_step=self.is_gradient_accumulation_boundary(), report_speed=report_progress)
62
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/utils/timer.py", line 256, in stop
63
+ get_accelerator().synchronize()
64
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/accelerator/cuda_accelerator.py", line 79, in synchronize
65
+ return torch.cuda.synchronize(device_index)
66
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/cuda/__init__.py", line 1040, in synchronize
67
+ return torch._C._cuda_synchronize()
68
+ KeyboardInterrupt
69
+ [rank0]: Traceback (most recent call last):
70
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 440, in <module>
71
+ [rank0]: main(cfg)
72
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 410, in main
73
+ [rank0]: trainer.train()
74
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
75
+ [rank0]: step_metrics = self._train_step(batch_vla)
76
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 354, in _train_step
77
+ [rank0]: self.accelerator.backward(total_loss)
78
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/accelerate/accelerator.py", line 2351, in backward
79
+ [rank0]: self.deepspeed_engine_wrapped.backward(loss, **kwargs)
80
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/accelerate/utils/deepspeed.py", line 275, in backward
81
+ [rank0]: self.engine.step()
82
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/runtime/engine.py", line 2382, in step
83
+ [rank0]: self.tput_timer.stop(global_step=self.is_gradient_accumulation_boundary(), report_speed=report_progress)
84
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/utils/timer.py", line 256, in stop
85
+ [rank0]: get_accelerator().synchronize()
86
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/accelerator/cuda_accelerator.py", line 79, in synchronize
87
+ [rank0]: return torch.cuda.synchronize(device_index)
88
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/cuda/__init__.py", line 1040, in synchronize
89
+ [rank0]: return torch._C._cuda_synchronize()
90
+ [rank0]: KeyboardInterrupt
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/wandb-metadata.json ADDED
@@ -0,0 +1,159 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-13T05:21:28.850487Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "./examples/calvin/train_files/starvla_train_calvin.yaml",
8
+ "--framework.name",
9
+ "QwenPI",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.attn_implementation",
13
+ "flash_attention_2",
14
+ "--framework.action_model.action_dim",
15
+ "10",
16
+ "--framework.action_model.state_dim",
17
+ "10",
18
+ "--framework.action_model.future_action_window_size",
19
+ "15",
20
+ "--framework.action_model.past_action_window_size",
21
+ "0",
22
+ "--framework.action_model.action_hidden_dim",
23
+ "1024",
24
+ "--framework.action_model.hidden_size",
25
+ "1024",
26
+ "--framework.action_model.action_model_type",
27
+ "DiT-B",
28
+ "--framework.action_model.add_pos_embed",
29
+ "True",
30
+ "--framework.action_model.max_seq_len",
31
+ "1024",
32
+ "--framework.action_model.noise_beta_alpha",
33
+ "1.5",
34
+ "--framework.action_model.noise_beta_beta",
35
+ "1.0",
36
+ "--framework.action_model.noise_s",
37
+ "0.999",
38
+ "--framework.action_model.num_timestep_buckets",
39
+ "1000",
40
+ "--framework.action_model.num_inference_timesteps",
41
+ "4",
42
+ "--framework.action_model.num_target_vision_tokens",
43
+ "32",
44
+ "--datasets.vla_data.data_root_dir",
45
+ "playground/Datasets/FastUMI",
46
+ "--datasets.vla_data.data_mix",
47
+ "fastumi_pickandplace_real_0307",
48
+ "--datasets.vla_data.per_device_batch_size",
49
+ "8",
50
+ "--datasets.vla_data.video_backend",
51
+ "torchvision_av",
52
+ "--trainer.freeze_modules",
53
+ "",
54
+ "--trainer.max_train_steps",
55
+ "20000",
56
+ "--trainer.save_interval",
57
+ "5000",
58
+ "--trainer.logging_frequency",
59
+ "50",
60
+ "--trainer.eval_interval",
61
+ "100",
62
+ "--trainer.gradient_accumulation_steps",
63
+ "1",
64
+ "--trainer.is_resume",
65
+ "true",
66
+ "--run_root_dir",
67
+ "./results/Checkpoints",
68
+ "--run_id",
69
+ "fastumi_pickandplace_qwenPI",
70
+ "--wandb_project",
71
+ "starVLA_FastUMI",
72
+ "--wandb_entity",
73
+ "2200011093-peking-university"
74
+ ],
75
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
76
+ "codePath": "starVLA/training/train_starvla.py",
77
+ "codePathLocal": "starVLA/training/train_starvla.py",
78
+ "git": {
79
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
80
+ "commit": "66b43863ede17a0e3081f822346e95cec52816d0"
81
+ },
82
+ "email": "wangpc@berkeley.edu",
83
+ "root": "./results/Checkpoints/fastumi_pickandplace_qwenPI/wandb",
84
+ "host": "tams02",
85
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
86
+ "cpu_count": 192,
87
+ "cpu_count_logical": 384,
88
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
89
+ "gpu_count": 8,
90
+ "disk": {
91
+ "/": {
92
+ "total": "3776651378688",
93
+ "used": "136342257664"
94
+ }
95
+ },
96
+ "memory": {
97
+ "total": "1081550508032"
98
+ },
99
+ "gpu_nvidia": [
100
+ {
101
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
102
+ "memoryTotal": "102641958912",
103
+ "cudaCores": 24064,
104
+ "architecture": "Blackwell",
105
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
106
+ },
107
+ {
108
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
109
+ "memoryTotal": "102641958912",
110
+ "cudaCores": 24064,
111
+ "architecture": "Blackwell",
112
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
113
+ },
114
+ {
115
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
116
+ "memoryTotal": "102641958912",
117
+ "cudaCores": 24064,
118
+ "architecture": "Blackwell",
119
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
120
+ },
121
+ {
122
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
123
+ "memoryTotal": "102641958912",
124
+ "cudaCores": 24064,
125
+ "architecture": "Blackwell",
126
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
127
+ },
128
+ {
129
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
130
+ "memoryTotal": "102641958912",
131
+ "cudaCores": 24064,
132
+ "architecture": "Blackwell",
133
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
134
+ },
135
+ {
136
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
137
+ "memoryTotal": "102641958912",
138
+ "cudaCores": 24064,
139
+ "architecture": "Blackwell",
140
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
141
+ },
142
+ {
143
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
144
+ "memoryTotal": "102641958912",
145
+ "cudaCores": 24064,
146
+ "architecture": "Blackwell",
147
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
148
+ },
149
+ {
150
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
151
+ "memoryTotal": "102641958912",
152
+ "cudaCores": 24064,
153
+ "architecture": "Blackwell",
154
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
155
+ }
156
+ ],
157
+ "cudaVersion": "13.1",
158
+ "writerId": "s3rvfu8shkj8mzdb09dke7i4kcnoimei"
159
+ }
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":1530},"epoch":1.72,"action_dit_loss":914.7533569335938,"learning_rate":1.2000000000000002e-06,"_runtime":1530.762687545,"data_time":0.002079134341329336,"_timestamp":1.7733808103730097e+09,"_step":600,"model_time":2.504630303941667,"mse_score":12.847688293457031}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/logs/debug-core.log ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T05:21:28.912497806Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpu6nvh7we/port-1922847.txt","pid":1922847,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-13T05:21:28.913330215Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":1922847}
3
+ {"time":"2026-03-13T05:21:28.913304855Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-1922847-1932814-919096094/socket","Net":"unix"}}
4
+ {"time":"2026-03-13T05:21:29.089472533Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-13T05:21:29.093429289Z","level":"INFO","msg":"handleInformInit: received","streamId":"9547dyq0","id":"1(@)"}
6
+ {"time":"2026-03-13T05:21:29.323526051Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"9547dyq0","id":"1(@)"}
7
+ {"time":"2026-03-13T05:21:34.786630872Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"2wlkb3en6u2r"}
8
+ {"time":"2026-03-13T05:47:00.423964632Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"}
9
+ {"time":"2026-03-13T05:47:00.424053363Z","level":"INFO","msg":"connection: closing","id":"1(@)"}
10
+ {"time":"2026-03-13T05:47:00.424105823Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"}
11
+ {"time":"2026-03-13T05:47:00.424100473Z","level":"INFO","msg":"server is shutting down"}
12
+ {"time":"2026-03-13T05:47:00.424227904Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-1922847-1932814-919096094/socket","Net":"unix"}}
13
+ {"time":"2026-03-13T05:47:00.847927312Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/logs/debug-internal.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T05:21:29.093650176Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-13T05:21:29.323312054Z","level":"INFO","msg":"stream: created new stream","id":"9547dyq0"}
3
+ {"time":"2026-03-13T05:21:29.323449952Z","level":"INFO","msg":"handler: started","stream_id":"9547dyq0"}
4
+ {"time":"2026-03-13T05:21:29.323520511Z","level":"INFO","msg":"stream: started","id":"9547dyq0"}
5
+ {"time":"2026-03-13T05:21:29.32355022Z","level":"INFO","msg":"sender: started","stream_id":"9547dyq0"}
6
+ {"time":"2026-03-13T05:21:29.323546541Z","level":"INFO","msg":"writer: started","stream_id":"9547dyq0"}
7
+ {"time":"2026-03-13T05:27:44.849719344Z","level":"INFO","msg":"api: retrying HTTP error","status":502,"url":"https://api.wandb.ai/files/2200011093-peking-university/starVLA_FastUMI/9547dyq0/file_stream","body":"\n<html><head>\n<meta http-equiv=\"content-type\" content=\"text/html;charset=utf-8\">\n<title>502 Server Error</title>\n</head>\n<body text=#000000 bgcolor=#ffffff>\n<h1>Error: Server Error</h1>\n<h2>The server encountered a temporary error and could not complete your request.<p>Please try again in 30 seconds.</h2>\n<h2></h2>\n</body></html>\n"}
8
+ {"time":"2026-03-13T05:47:00.424077743Z","level":"INFO","msg":"stream: closing","id":"9547dyq0"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/logs/debug.log ADDED
File without changes
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_052128-9547dyq0/run-9547dyq0.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f7b87ad5384fd6829f1f8a878d5a771f57c2d530d45dfc8cb1747e40a235d5c0
3
+ size 589824
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/files/output.log ADDED
@@ -0,0 +1,133 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 03/13 [05:54:02] INFO  | >> ***** Training Configuration ***** ]8;id=98246;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=229258;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#338\338]8;;\
2
+   INFO  | >> Total optimization steps = 20000 ]8;id=208496;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=750800;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#339\339]8;;\
3
+   INFO  | >> Per device batch size = 8 ]8;id=471029;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=617889;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#340\340]8;;\
4
+   INFO  | >> Gradient accumulation steps = 1 ]8;id=844962;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=167414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#341\341]8;;\
5
+   INFO  | >> Total batch size = 64 ]8;id=225772;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=800581;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#342\342]8;;\
6
+ 2%|█ | 500/20000 [24:26<15:33:58, 2.87s/it, data_times=0.100, model_times=2.589]
7
+ 03/13 [05:56:29] INFO  | >> Step 50, Loss: {'action_dit_loss': 1023734.4375, 'data_time': ]8;id=376417;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=888662;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
8
+   0.1184354592114687, 'model_time': 2.5530195953324437, 'learning_rate':  
9
+   1.0000000000000001e-07, 'epoch': 25.0})  
10
+ 03/13 [05:58:57] INFO  | >> Step 100, Loss: {'action_dit_loss': 845367.125, 'mse_score': ]8;id=45561;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=765179;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
11
+   27.783798217773438, 'data_time': 0.11873677279800177, 'model_time':  
12
+   2.5620764740742743, 'learning_rate': 2.0000000000000002e-07, 'epoch':  
13
+   50.0})  
14
+ 03/13 [06:01:22] INFO  | >> Step 150, Loss: {'action_dit_loss': 481553.0, 'data_time': ]8;id=396922;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=82627;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
15
+   0.4563447558321059, 'model_time': 2.76532160397619, 'learning_rate':  
16
+   3.0000000000000004e-07, 'epoch': 75.0})  
17
+ 03/13 [06:03:49] INFO  | >> Step 200, Loss: {'action_dit_loss': 158459.78125, 'mse_score': ]8;id=648564;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=928463;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
18
+   32.856329345703124, 'data_time': 0.4952073488384485, 'model_time':  
19
+   2.650609015021473, 'learning_rate': 4.0000000000000003e-07, 'epoch':  
20
+   100.0})  
21
+ 03/13 [06:06:15] INFO  | >> Step 250, Loss: {'action_dit_loss': 3656.809326171875, 'data_time': ]8;id=738797;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=72933;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
22
+   0.12273448007181287, 'model_time': 2.548748595174402, 'learning_rate':  
23
+   5.000000000000001e-07, 'epoch': 125.0})  
24
+ 03/13 [06:08:43] INFO  | >> Step 300, Loss: {'action_dit_loss': 567.90771484375, 'mse_score': ]8;id=303445;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=83667;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
25
+   29.84052734375, 'data_time': 0.12868641363456845, 'model_time':  
26
+   2.6304455418139696, 'learning_rate': 6.000000000000001e-07, 'epoch':  
27
+   150.0})  
28
+ 03/13 [06:11:11] INFO  | >> Step 350, Loss: {'action_dit_loss': 390.1324768066406, 'data_time': ]8;id=398591;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=291476;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
29
+   0.4994450118392706, 'model_time': 2.6760935387574136, 'learning_rate':  
30
+   7.000000000000001e-07, 'epoch': 175.0})  
31
+ 03/13 [06:13:38] INFO  | >> Step 400, Loss: {'action_dit_loss': 308.3482971191406, 'mse_score': 40.98153991699219, ]8;id=170555;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=388162;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
32
+   'data_time': 0.49330907594412565, 'model_time': 2.6808951501734555, 'learning_rate':  
33
+   8.000000000000001e-07, 'epoch': 200.0})  
34
+ 03/13 [06:16:03] INFO  | >> Step 450, Loss: {'action_dit_loss': 295.05206298828125, 'data_time': ]8;id=735911;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=982153;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
35
+   0.14659648016095161, 'model_time': 2.5430247145704925, 'learning_rate':  
36
+   9.000000000000001e-07, 'epoch': 225.0})  
37
+ 03/13 [06:18:30] INFO  | >> Step 500, Loss: {'action_dit_loss': 203.4255828857422, 'mse_score': ]8;id=665822;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=179451;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
38
+   28.486944580078124, 'data_time': 0.09962102398276329, 'model_time': 2.5893459981307387,  
39
+   'learning_rate': 1.0000000000000002e-06, 'epoch': 250.0})  
40
+ 03/13 [06:20:57] INFO  | >> Step 550, Loss: {'action_dit_loss': 279.1764831542969, 'data_time': ]8;id=484714;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=397887;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
41
+   0.44782329397276044, 'model_time': 2.715342686045915, 'learning_rate': 1.1e-06, 'epoch':  
42
+   275.0})  
43
+ 03/13 [06:23:23] INFO  | >> Step 600, Loss: {'action_dit_loss': 355.8760681152344, 'mse_score': 55.74105224609375, ]8;id=584004;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=230283;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
44
+   'data_time': 0.47645226726308465, 'model_time': 2.613042331766337, 'learning_rate':  
45
+   1.2000000000000002e-06, 'epoch': 300.0})  
46
+ 03/13 [06:25:48] INFO  | >> Step 650, Loss: {'action_dit_loss': 220.4207000732422, 'data_time': ]8;id=813694;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=58655;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
47
+   0.09945486905053258, 'model_time': 2.615934310015291, 'learning_rate': 1.3e-06, 'epoch':  
48
+   325.0})  
49
+ 03/13 [06:28:15] INFO  | >> Step 700, Loss: {'action_dit_loss': 236.1172332763672, 'mse_score': 28.46719970703125, ]8;id=330776;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=420651;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
50
+   'data_time': 0.08917575189843774, 'model_time': 2.6393821919336915, 'learning_rate':  
51
+   1.4000000000000001e-06, 'epoch': 350.0})  
52
+ 03/13 [06:30:41] INFO  | >> Step 750, Loss: {'action_dit_loss': 278.9415588378906, 'data_time': ]8;id=988712;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=594731;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
53
+   0.4641818590462208, 'model_time': 2.664197747129947, 'learning_rate': 1.5e-06, 'epoch':  
54
+   375.0})  
55
+ 03/13 [06:33:07] INFO  | >> Step 800, Loss: {'action_dit_loss': 196.75160217285156, 'mse_score': 27.56171875, ]8;id=687277;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=523481;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
56
+   'data_time': 0.44845692813396454, 'model_time': 2.6826071268878877, 'learning_rate':  
57
+   1.6000000000000001e-06, 'epoch': 400.0})  
58
+ 03/13 [06:35:32] INFO  | >> Step 850, Loss: {'action_dit_loss': 596.9781494140625, 'data_time': ]8;id=481141;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=149811;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
59
+   0.10228884313255548, 'model_time': 2.566598553676158, 'learning_rate':  
60
+   1.7000000000000002e-06, 'epoch': 425.0})  
61
+ 03/13 [06:37:58] INFO  | >> Step 900, Loss: {'action_dit_loss': 189.26988220214844, 'mse_score': ]8;id=588637;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=565158;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
62
+   28.475973510742186, 'data_time': 0.09421875094994903, 'model_time': 2.56506073102355,  
63
+   'learning_rate': 1.8000000000000001e-06, 'epoch': 450.0})  
64
+ 03/13 [06:40:25] INFO  | >> Step 950, Loss: {'action_dit_loss': 217.2085723876953, 'data_time': ]8;id=941435;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=611878;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
65
+   0.47165690595284104, 'model_time': 2.6609441321343184, 'learning_rate':  
66
+   1.9000000000000002e-06, 'epoch': 475.0})  
67
+ 03/13 [06:42:51] INFO  | >> Step 1000, Loss: {'action_dit_loss': 461.1436462402344, 'mse_score': ]8;id=534277;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=517488;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
68
+   28.475180053710936, 'data_time': 0.5010749120265245, 'model_time': 2.6723324418999255,  
69
+   'learning_rate': 2.0000000000000003e-06, 'epoch': 500.0})  
70
+ 03/13 [06:45:16] INFO  | >> Step 1050, Loss: {'action_dit_loss': 224.42190551757812, 'data_time': ]8;id=114975;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=160265;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
71
+   0.09496595803648233, 'model_time': 2.5762311429716647, 'learning_rate':  
72
+   2.1000000000000002e-06, 'epoch': 525.0})  
73
+ 03/13 [06:47:43] INFO  | >> Step 1100, Loss: {'action_dit_loss': 134.0990753173828, 'mse_score': ]8;id=442666;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=625380;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
74
+   28.474765014648437, 'data_time': 0.08824913622811437, 'model_time': 2.5823110449127853,  
75
+   'learning_rate': 2.2e-06, 'epoch': 550.0})  
76
+ 03/13 [06:50:09] INFO  | >> Step 1150, Loss: {'action_dit_loss': 84.02532196044922, 'data_time': ]8;id=490785;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=554816;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
77
+   0.45307353185489774, 'model_time': 2.6590585242956877, 'learning_rate':  
78
+   2.3000000000000004e-06, 'epoch': 575.0})  
79
+ 03/13 [06:52:35] INFO  | >> Step 1200, Loss: {'action_dit_loss': 185.86668395996094, 'mse_score': ]8;id=12038;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=713328;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
80
+   28.47579345703125, 'data_time': 0.4087458089925349, 'model_time': 2.7611988466233015,  
81
+   'learning_rate': 2.4000000000000003e-06, 'epoch': 600.0})  
82
+ 03/13 [06:55:01] INFO  | >> Step 1250, Loss: {'action_dit_loss': 418.23175048828125, 'data_time': ]8;id=563054;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=787352;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
83
+   0.09327394468709826, 'model_time': 2.588544320780784, 'learning_rate': 2.5e-06, 'epoch':  
84
+   625.0})  
85
+ 03/13 [06:57:27] INFO  | >> Step 1300, Loss: {'action_dit_loss': 83.31026458740234, 'mse_score': 28.476318359375, ]8;id=116970;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=307757;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
86
+   'data_time': 0.09353463631123304, 'model_time': 2.59936897829175, 'learning_rate': 2.6e-06,  
87
+   'epoch': 650.0})  
88
+ 03/13 [06:59:54] INFO  | >> Step 1350, Loss: {'action_dit_loss': 254.9491424560547, 'data_time': ]8;id=757168;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=918398;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
89
+   0.45593405794352293, 'model_time': 2.8787891338579357, 'learning_rate':  
90
+   2.7000000000000004e-06, 'epoch': 675.0})  
91
+ 03/13 [07:02:20] INFO  | >> Step 1400, Loss: {'action_dit_loss': 156.01995849609375, 'mse_score': ]8;id=187330;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=532342;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
92
+   28.476077270507812, 'data_time': 0.4633677378296852, 'model_time': 2.685747683979571,  
93
+   'learning_rate': 2.8000000000000003e-06, 'epoch': 700.0})  
94
+ 03/13 [07:04:46] INFO  | >> Step 1450, Loss: {'action_dit_loss': 104.18902587890625, 'data_time': ]8;id=312942;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=882554;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
95
+   0.10353789990767837, 'model_time': 2.6161012779921293, 'learning_rate': 2.9e-06, 'epoch':  
96
+   725.0})  
97
+ 03/13 [07:07:12] INFO  | >> Step 1500, Loss: {'action_dit_loss': 167.5433807373047, 'mse_score': 28.476904296875, ]8;id=160263;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=392077;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
98
+   'data_time': 0.07424044329673052, 'model_time': 2.5984189393930137, 'learning_rate': 3e-06,  
99
+   'epoch': 750.0})  
100
+ 03/13 [07:09:38] INFO  | >> Step 1550, Loss: {'action_dit_loss': 168.45278930664062, 'data_time': ]8;id=816449;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=967242;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
101
+   0.46553037920966744, 'model_time': 2.880160149652511, 'learning_rate':  
102
+   3.1000000000000004e-06, 'epoch': 775.0})  
103
+ 03/13 [07:12:05] INFO  | >> Step 1600, Loss: {'action_dit_loss': 134.20968627929688, 'mse_score': ]8;id=339902;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=512340;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
104
+   28.479208374023436, 'data_time': 0.43456642609089613, 'model_time': 2.6962539590895176,  
105
+   'learning_rate': 3.2000000000000003e-06, 'epoch': 800.0})  
106
+ 03/13 [07:14:31] INFO  | >> Step 1650, Loss: {'action_dit_loss': 399.4845886230469, 'data_time': ]8;id=921406;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=872064;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
107
+   0.08665798092260957, 'model_time': 2.609215545002371, 'learning_rate':  
108
+   3.3000000000000006e-06, 'epoch': 825.0})  
109
+ 03/13 [07:16:57] INFO  | >> Step 1700, Loss: {'action_dit_loss': 236.0421905517578, 'mse_score': ]8;id=252572;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=920659;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
110
+   28.47523193359375, 'data_time': 0.07761035626754165, 'model_time': 2.6140335868112743,  
111
+   'learning_rate': 3.4000000000000005e-06, 'epoch': 850.0})  
112
+ 03/13 [07:19:23] INFO  | >> Step 1750, Loss: {'action_dit_loss': 115.4985122680664, 'data_time': ]8;id=767460;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=509597;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
113
+   0.45583478920161724, 'model_time': 2.6625322620384395, 'learning_rate': 3.5e-06, 'epoch':  
114
+   875.0})  
115
+ 03/13 [07:21:50] INFO  | >> Step 1800, Loss: {'action_dit_loss': 72.10005187988281, 'mse_score': ]8;id=803035;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=131869;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
116
+   28.476263427734374, 'data_time': 0.4792325892485678, 'model_time': 2.6509694480337203,  
117
+   'learning_rate': 3.6000000000000003e-06, 'epoch': 900.0})  
118
+ 03/13 [07:24:15] INFO  | >> Step 1850, Loss: {'action_dit_loss': 151.80313110351562, 'data_time': ]8;id=576510;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=173148;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
119
+   0.08413119288161397, 'model_time': 2.5697082350961864, 'learning_rate': 3.7e-06, 'epoch':  
120
+   925.0})  
121
+ 03/13 [07:26:42] INFO  | >> Step 1900, Loss: {'action_dit_loss': 157.64427185058594, 'mse_score': ]8;id=443692;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=222086;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
122
+   28.477935791015625, 'data_time': 0.10603159526363015, 'model_time': 2.5718924370594323,  
123
+   'learning_rate': 3.8000000000000005e-06, 'epoch': 950.0})  
124
+ 03/13 [07:29:07] INFO  | >> Step 1950, Loss: {'action_dit_loss': 189.83152770996094, 'data_time': ]8;id=723378;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=210922;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
125
+   0.46740381931886077, 'model_time': 2.7080146190710366, 'learning_rate':  
126
+   3.900000000000001e-06, 'epoch': 975.0})  
127
+ 03/13 [07:31:36] INFO  | >> Step 2000, Loss: {'action_dit_loss': 65.36077880859375, 'mse_score': ]8;id=681446;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=391559;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
128
+   28.476959228515625, 'data_time': 0.505982331931591, 'model_time': 2.5988965397700667,  
129
+   'learning_rate': 4.000000000000001e-06, 'epoch': 1000.0})  
130
+ 03/13 [07:34:02] INFO  | >> Step 2050, Loss: {'action_dit_loss': 157.7738800048828, 'data_time': ]8;id=126882;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=259947;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
131
+   0.0906450180336833, 'model_time': 2.5843174257315695, 'learning_rate': 4.1e-06, 'epoch':  
132
+   1025.0})  
133
+ 03/13 [07:36:29] INFO  | >> Step 2100, Loss: {'action_dit_loss': 152.810791015625, 'mse_score': 28.477011108398436, 'data_time': 0.10848667286336422, 'model_time': 2.560679371934384, 'learning_rate': 4.2000000000000004e-06, 'epoch': 1050.0}) ]8;id=616886;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=580828;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/files/wandb-metadata.json ADDED
@@ -0,0 +1,159 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-13T05:54:01.695966Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "./examples/calvin/train_files/starvla_train_calvin.yaml",
8
+ "--framework.name",
9
+ "QwenPI",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.attn_implementation",
13
+ "flash_attention_2",
14
+ "--framework.action_model.action_dim",
15
+ "10",
16
+ "--framework.action_model.state_dim",
17
+ "10",
18
+ "--framework.action_model.future_action_window_size",
19
+ "15",
20
+ "--framework.action_model.past_action_window_size",
21
+ "0",
22
+ "--framework.action_model.action_hidden_dim",
23
+ "1024",
24
+ "--framework.action_model.hidden_size",
25
+ "1024",
26
+ "--framework.action_model.action_model_type",
27
+ "DiT-B",
28
+ "--framework.action_model.add_pos_embed",
29
+ "True",
30
+ "--framework.action_model.max_seq_len",
31
+ "1024",
32
+ "--framework.action_model.noise_beta_alpha",
33
+ "1.5",
34
+ "--framework.action_model.noise_beta_beta",
35
+ "1.0",
36
+ "--framework.action_model.noise_s",
37
+ "0.999",
38
+ "--framework.action_model.num_timestep_buckets",
39
+ "1000",
40
+ "--framework.action_model.num_inference_timesteps",
41
+ "4",
42
+ "--framework.action_model.num_target_vision_tokens",
43
+ "32",
44
+ "--datasets.vla_data.data_root_dir",
45
+ "playground/Datasets/FastUMI",
46
+ "--datasets.vla_data.data_mix",
47
+ "fastumi_pickandplace_debug_1ep",
48
+ "--datasets.vla_data.per_device_batch_size",
49
+ "8",
50
+ "--datasets.vla_data.video_backend",
51
+ "torchvision_av",
52
+ "--trainer.freeze_modules",
53
+ "",
54
+ "--trainer.max_train_steps",
55
+ "20000",
56
+ "--trainer.save_interval",
57
+ "5000",
58
+ "--trainer.logging_frequency",
59
+ "50",
60
+ "--trainer.eval_interval",
61
+ "100",
62
+ "--trainer.gradient_accumulation_steps",
63
+ "1",
64
+ "--trainer.is_resume",
65
+ "true",
66
+ "--run_root_dir",
67
+ "./results/Checkpoints",
68
+ "--run_id",
69
+ "fastumi_pickandplace_qwenPI",
70
+ "--wandb_project",
71
+ "starVLA_FastUMI",
72
+ "--wandb_entity",
73
+ "2200011093-peking-university"
74
+ ],
75
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
76
+ "codePath": "starVLA/training/train_starvla.py",
77
+ "codePathLocal": "starVLA/training/train_starvla.py",
78
+ "git": {
79
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
80
+ "commit": "66b43863ede17a0e3081f822346e95cec52816d0"
81
+ },
82
+ "email": "wangpc@berkeley.edu",
83
+ "root": "./results/Checkpoints/fastumi_pickandplace_qwenPI/wandb",
84
+ "host": "tams02",
85
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
86
+ "cpu_count": 192,
87
+ "cpu_count_logical": 384,
88
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
89
+ "gpu_count": 8,
90
+ "disk": {
91
+ "/": {
92
+ "total": "3776651378688",
93
+ "used": "136342392832"
94
+ }
95
+ },
96
+ "memory": {
97
+ "total": "1081550508032"
98
+ },
99
+ "gpu_nvidia": [
100
+ {
101
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
102
+ "memoryTotal": "102641958912",
103
+ "cudaCores": 24064,
104
+ "architecture": "Blackwell",
105
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
106
+ },
107
+ {
108
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
109
+ "memoryTotal": "102641958912",
110
+ "cudaCores": 24064,
111
+ "architecture": "Blackwell",
112
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
113
+ },
114
+ {
115
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
116
+ "memoryTotal": "102641958912",
117
+ "cudaCores": 24064,
118
+ "architecture": "Blackwell",
119
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
120
+ },
121
+ {
122
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
123
+ "memoryTotal": "102641958912",
124
+ "cudaCores": 24064,
125
+ "architecture": "Blackwell",
126
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
127
+ },
128
+ {
129
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
130
+ "memoryTotal": "102641958912",
131
+ "cudaCores": 24064,
132
+ "architecture": "Blackwell",
133
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
134
+ },
135
+ {
136
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
137
+ "memoryTotal": "102641958912",
138
+ "cudaCores": 24064,
139
+ "architecture": "Blackwell",
140
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
141
+ },
142
+ {
143
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
144
+ "memoryTotal": "102641958912",
145
+ "cudaCores": 24064,
146
+ "architecture": "Blackwell",
147
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
148
+ },
149
+ {
150
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
151
+ "memoryTotal": "102641958912",
152
+ "cudaCores": 24064,
153
+ "architecture": "Blackwell",
154
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
155
+ }
156
+ ],
157
+ "cudaVersion": "13.1",
158
+ "writerId": "lh7stxryqsn2w12siqjchs453dvif557"
159
+ }
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/logs/debug-core.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T05:54:01.750260433Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmp5mtl8h0t/port-2551513.txt","pid":2551513,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-13T05:54:01.750898585Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":2551513}
3
+ {"time":"2026-03-13T05:54:01.750893325Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-2551513-2554212-3087288390/socket","Net":"unix"}}
4
+ {"time":"2026-03-13T05:54:01.933288056Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-13T05:54:01.937029635Z","level":"INFO","msg":"handleInformInit: received","streamId":"pxng7bwb","id":"1(@)"}
6
+ {"time":"2026-03-13T05:54:02.175085926Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"pxng7bwb","id":"1(@)"}
7
+ {"time":"2026-03-13T05:54:07.736414258Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"td4wqoolb35y"}
8
+ {"time":"2026-03-13T07:36:38.073934654Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/logs/debug-internal.log ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T05:54:01.937198615Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-13T05:54:02.174791025Z","level":"INFO","msg":"stream: created new stream","id":"pxng7bwb"}
3
+ {"time":"2026-03-13T05:54:02.174920626Z","level":"INFO","msg":"handler: started","stream_id":"pxng7bwb"}
4
+ {"time":"2026-03-13T05:54:02.175076706Z","level":"INFO","msg":"stream: started","id":"pxng7bwb"}
5
+ {"time":"2026-03-13T05:54:02.175092296Z","level":"INFO","msg":"writer: started","stream_id":"pxng7bwb"}
6
+ {"time":"2026-03-13T05:54:02.175105866Z","level":"INFO","msg":"sender: started","stream_id":"pxng7bwb"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/logs/debug.log ADDED
File without changes
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_055401-pxng7bwb/run-pxng7bwb.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d7f2c625e0a1608fef86c8d47a8fcbe13e44984f62354aac8bb0246a2633f8d
3
+ size 2260992
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/files/output.log ADDED
@@ -0,0 +1 @@
 
 
1
+ 03/13 [08:12:50] INFO  | >> ***** Training C
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/files/wandb-metadata.json ADDED
@@ -0,0 +1,161 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-13T08:12:49.379188Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "./examples/calvin/train_files/starvla_train_calvin.yaml",
8
+ "--framework.name",
9
+ "QwenPI",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.attn_implementation",
13
+ "flash_attention_2",
14
+ "--framework.action_model.action_dim",
15
+ "10",
16
+ "--framework.action_model.state_dim",
17
+ "10",
18
+ "--framework.action_model.future_action_window_size",
19
+ "15",
20
+ "--framework.action_model.past_action_window_size",
21
+ "0",
22
+ "--framework.action_model.action_hidden_dim",
23
+ "1024",
24
+ "--framework.action_model.hidden_size",
25
+ "1024",
26
+ "--framework.action_model.action_model_type",
27
+ "DiT-B",
28
+ "--framework.action_model.add_pos_embed",
29
+ "True",
30
+ "--framework.action_model.max_seq_len",
31
+ "1024",
32
+ "--framework.action_model.noise_beta_alpha",
33
+ "1.5",
34
+ "--framework.action_model.noise_beta_beta",
35
+ "1.0",
36
+ "--framework.action_model.noise_s",
37
+ "0.999",
38
+ "--framework.action_model.num_timestep_buckets",
39
+ "1000",
40
+ "--framework.action_model.num_inference_timesteps",
41
+ "4",
42
+ "--framework.action_model.num_target_vision_tokens",
43
+ "32",
44
+ "--datasets.vla_data.data_root_dir",
45
+ "playground/Datasets/FastUMI",
46
+ "--datasets.vla_data.data_mix",
47
+ "fastumi_pickandplace_debug_1ep",
48
+ "--datasets.vla_data.include_state",
49
+ "true",
50
+ "--datasets.vla_data.per_device_batch_size",
51
+ "8",
52
+ "--datasets.vla_data.video_backend",
53
+ "torchvision_av",
54
+ "--trainer.freeze_modules",
55
+ "",
56
+ "--trainer.max_train_steps",
57
+ "20000",
58
+ "--trainer.save_interval",
59
+ "5000",
60
+ "--trainer.logging_frequency",
61
+ "50",
62
+ "--trainer.eval_interval",
63
+ "100",
64
+ "--trainer.gradient_accumulation_steps",
65
+ "1",
66
+ "--trainer.is_resume",
67
+ "true",
68
+ "--run_root_dir",
69
+ "./results/Checkpoints",
70
+ "--run_id",
71
+ "fastumi_pickandplace_qwenPI",
72
+ "--wandb_project",
73
+ "starVLA_FastUMI",
74
+ "--wandb_entity",
75
+ "2200011093-peking-university"
76
+ ],
77
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
78
+ "codePath": "starVLA/training/train_starvla.py",
79
+ "codePathLocal": "starVLA/training/train_starvla.py",
80
+ "git": {
81
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
82
+ "commit": "66b43863ede17a0e3081f822346e95cec52816d0"
83
+ },
84
+ "email": "wangpc@berkeley.edu",
85
+ "root": "./results/Checkpoints/fastumi_pickandplace_qwenPI/wandb",
86
+ "host": "tams02",
87
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
88
+ "cpu_count": 192,
89
+ "cpu_count_logical": 384,
90
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
91
+ "gpu_count": 8,
92
+ "disk": {
93
+ "/": {
94
+ "total": "3776651378688",
95
+ "used": "136324321280"
96
+ }
97
+ },
98
+ "memory": {
99
+ "total": "1081550508032"
100
+ },
101
+ "gpu_nvidia": [
102
+ {
103
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
104
+ "memoryTotal": "102641958912",
105
+ "cudaCores": 24064,
106
+ "architecture": "Blackwell",
107
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
108
+ },
109
+ {
110
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
111
+ "memoryTotal": "102641958912",
112
+ "cudaCores": 24064,
113
+ "architecture": "Blackwell",
114
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
115
+ },
116
+ {
117
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
118
+ "memoryTotal": "102641958912",
119
+ "cudaCores": 24064,
120
+ "architecture": "Blackwell",
121
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
122
+ },
123
+ {
124
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
125
+ "memoryTotal": "102641958912",
126
+ "cudaCores": 24064,
127
+ "architecture": "Blackwell",
128
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
129
+ },
130
+ {
131
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
132
+ "memoryTotal": "102641958912",
133
+ "cudaCores": 24064,
134
+ "architecture": "Blackwell",
135
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
136
+ },
137
+ {
138
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
139
+ "memoryTotal": "102641958912",
140
+ "cudaCores": 24064,
141
+ "architecture": "Blackwell",
142
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
143
+ },
144
+ {
145
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
146
+ "memoryTotal": "102641958912",
147
+ "cudaCores": 24064,
148
+ "architecture": "Blackwell",
149
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
150
+ },
151
+ {
152
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
153
+ "memoryTotal": "102641958912",
154
+ "cudaCores": 24064,
155
+ "architecture": "Blackwell",
156
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
157
+ }
158
+ ],
159
+ "cudaVersion": "13.1",
160
+ "writerId": "0bwvns2760wxi5kpf48cijgc8hwdpok9"
161
+ }
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/logs/debug-core.log ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T08:12:49.442812952Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpygg3pbdn/port-606893.txt","pid":606893,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-13T08:12:49.443478421Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":606893}
3
+ {"time":"2026-03-13T08:12:49.443423672Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-606893-609835-1204213629/socket","Net":"unix"}}
4
+ {"time":"2026-03-13T08:12:49.622638072Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-13T08:12:49.626930671Z","level":"INFO","msg":"handleInformInit: received","streamId":"rqa9o89w","id":"1(@)"}
6
+ {"time":"2026-03-13T08:12:49.933829256Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"rqa9o89w","id":"1(@)"}
7
+ {"time":"2026-03-13T08:12:51.951100498Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/logs/debug-internal.log ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T08:12:49.627056849Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-13T08:12:49.9335631Z","level":"INFO","msg":"stream: created new stream","id":"rqa9o89w"}
3
+ {"time":"2026-03-13T08:12:49.933709368Z","level":"INFO","msg":"handler: started","stream_id":"rqa9o89w"}
4
+ {"time":"2026-03-13T08:12:49.933817026Z","level":"INFO","msg":"stream: started","id":"rqa9o89w"}
5
+ {"time":"2026-03-13T08:12:49.933851966Z","level":"INFO","msg":"sender: started","stream_id":"rqa9o89w"}
6
+ {"time":"2026-03-13T08:12:49.933854596Z","level":"INFO","msg":"writer: started","stream_id":"rqa9o89w"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/logs/debug.log ADDED
File without changes
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081249-rqa9o89w/run-rqa9o89w.wandb ADDED
Binary file (7 Bytes). View file
 
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/config.yaml ADDED
@@ -0,0 +1,171 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.25.0
4
+ e:
5
+ xevzqrqcaigt4lan5smfrjd6ooaz8cs8:
6
+ args:
7
+ - --config_yaml
8
+ - ./examples/calvin/train_files/starvla_train_calvin.yaml
9
+ - --framework.name
10
+ - QwenPI
11
+ - --framework.qwenvl.base_vlm
12
+ - playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action
13
+ - --framework.qwenvl.attn_implementation
14
+ - flash_attention_2
15
+ - --framework.action_model.action_dim
16
+ - "10"
17
+ - --framework.action_model.state_dim
18
+ - "10"
19
+ - --framework.action_model.future_action_window_size
20
+ - "15"
21
+ - --framework.action_model.past_action_window_size
22
+ - "0"
23
+ - --framework.action_model.action_hidden_dim
24
+ - "1024"
25
+ - --framework.action_model.hidden_size
26
+ - "1024"
27
+ - --framework.action_model.action_model_type
28
+ - DiT-B
29
+ - --framework.action_model.add_pos_embed
30
+ - "True"
31
+ - --framework.action_model.max_seq_len
32
+ - "1024"
33
+ - --framework.action_model.noise_beta_alpha
34
+ - "1.5"
35
+ - --framework.action_model.noise_beta_beta
36
+ - "1.0"
37
+ - --framework.action_model.noise_s
38
+ - "0.999"
39
+ - --framework.action_model.num_timestep_buckets
40
+ - "1000"
41
+ - --framework.action_model.num_inference_timesteps
42
+ - "4"
43
+ - --framework.action_model.num_target_vision_tokens
44
+ - "32"
45
+ - --datasets.vla_data.data_root_dir
46
+ - playground/Datasets/FastUMI
47
+ - --datasets.vla_data.data_mix
48
+ - fastumi_pickandplace_debug_1ep
49
+ - --datasets.vla_data.include_state
50
+ - "true"
51
+ - --datasets.vla_data.per_device_batch_size
52
+ - "8"
53
+ - --datasets.vla_data.video_backend
54
+ - torchvision_av
55
+ - --trainer.freeze_modules
56
+ - ""
57
+ - --trainer.max_train_steps
58
+ - "20000"
59
+ - --trainer.save_interval
60
+ - "5000"
61
+ - --trainer.logging_frequency
62
+ - "50"
63
+ - --trainer.eval_interval
64
+ - "100"
65
+ - --trainer.gradient_accumulation_steps
66
+ - "1"
67
+ - --trainer.is_resume
68
+ - "true"
69
+ - --run_root_dir
70
+ - ./results/Checkpoints
71
+ - --run_id
72
+ - fastumi_pickandplace_qwenPI
73
+ - --wandb_project
74
+ - starVLA_FastUMI
75
+ - --wandb_entity
76
+ - 2200011093-peking-university
77
+ codePath: starVLA/training/train_starvla.py
78
+ codePathLocal: starVLA/training/train_starvla.py
79
+ cpu_count: 192
80
+ cpu_count_logical: 384
81
+ cudaVersion: "13.1"
82
+ disk:
83
+ /:
84
+ total: "3776651378688"
85
+ used: "136324956160"
86
+ email: wangpc@berkeley.edu
87
+ executable: /home/wangpc/miniconda3/envs/starVLA/bin/python3.10
88
+ git:
89
+ commit: 66b43863ede17a0e3081f822346e95cec52816d0
90
+ remote: https://github.com/Kaiwen-Hong/starVLA.git
91
+ gpu: NVIDIA RTX PRO 6000 Blackwell Server Edition
92
+ gpu_count: 8
93
+ gpu_nvidia:
94
+ - architecture: Blackwell
95
+ cudaCores: 24064
96
+ memoryTotal: "102641958912"
97
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
98
+ uuid: GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673
99
+ - architecture: Blackwell
100
+ cudaCores: 24064
101
+ memoryTotal: "102641958912"
102
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
103
+ uuid: GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5
104
+ - architecture: Blackwell
105
+ cudaCores: 24064
106
+ memoryTotal: "102641958912"
107
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
108
+ uuid: GPU-1f3c0889-b740-5143-e064-afeb49245756
109
+ - architecture: Blackwell
110
+ cudaCores: 24064
111
+ memoryTotal: "102641958912"
112
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
113
+ uuid: GPU-49955a45-509a-e8af-a468-b9ef0449b005
114
+ - architecture: Blackwell
115
+ cudaCores: 24064
116
+ memoryTotal: "102641958912"
117
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
118
+ uuid: GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd
119
+ - architecture: Blackwell
120
+ cudaCores: 24064
121
+ memoryTotal: "102641958912"
122
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
123
+ uuid: GPU-d16b5acf-a100-2d1b-8138-1661892f3d26
124
+ - architecture: Blackwell
125
+ cudaCores: 24064
126
+ memoryTotal: "102641958912"
127
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
128
+ uuid: GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab
129
+ - architecture: Blackwell
130
+ cudaCores: 24064
131
+ memoryTotal: "102641958912"
132
+ name: NVIDIA RTX PRO 6000 Blackwell Server Edition
133
+ uuid: GPU-747dc32b-b025-76e9-697d-fd33ef441b47
134
+ host: tams02
135
+ memory:
136
+ total: "1081550508032"
137
+ os: Linux-6.8.0-94-generic-x86_64-with-glibc2.39
138
+ program: /scratch/wangpc/starVLA/starVLA/training/train_starvla.py
139
+ python: CPython 3.10.19
140
+ root: ./results/Checkpoints/fastumi_pickandplace_qwenPI/wandb
141
+ startedAt: "2026-03-13T08:16:53.333287Z"
142
+ writerId: xevzqrqcaigt4lan5smfrjd6ooaz8cs8
143
+ m: []
144
+ python_version: 3.10.19
145
+ t:
146
+ "1":
147
+ - 1
148
+ - 11
149
+ - 41
150
+ - 49
151
+ - 63
152
+ - 71
153
+ - 80
154
+ - 83
155
+ "2":
156
+ - 1
157
+ - 11
158
+ - 41
159
+ - 49
160
+ - 63
161
+ - 71
162
+ - 80
163
+ - 83
164
+ "3":
165
+ - 13
166
+ - 61
167
+ "4": 3.10.19
168
+ "5": 0.25.0
169
+ "6": 4.57.0
170
+ "12": 0.25.0
171
+ "13": linux-x86_64
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/output.log ADDED
@@ -0,0 +1,102 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 03/13 [08:16:54] INFO  | >> ***** Training Configuration ***** ]8;id=98246;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=229258;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#340\340]8;;\
2
+   INFO  | >> Total optimization steps = 20000 ]8;id=208496;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=750800;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#341\341]8;;\
3
+   INFO  | >> Per device batch size = 8 ]8;id=471029;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=617889;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#342\342]8;;\
4
+   INFO  | >> Gradient accumulation steps = 1 ]8;id=844962;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=167414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#343\343]8;;\
5
+   INFO  | >> Total batch size = 64 ]8;id=225772;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=800581;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#344\344]8;;\
6
+ 0%|▍ | 100/20000 [04:54<16:07:50, 2.92s/it, data_times=0.131, model_times=2.581]
7
+ 03/13 [08:19:21] INFO  | >> Step 50, Loss: {'action_dit_loss': 1021046.3125, 'data_time': 0.09994111442938447, 'model_time': ]8;id=376417;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=888662;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
8
+   2.6226530433632433, 'learning_rate': 1.0000000000000001e-07, 'epoch': 25.0})  
9
+ Unusual action found in eval: [[[-0.2311 -0.0881 0.1273 ... 0.8384 0.1476 1. ]
10
+ [-0.3345 -0.1672 0.0745 ... 0.8096 0.1587 1. ]
11
+ [-0.3853 -0.3894 0.1151 ... 0.8667 -0.1583 1. ]
12
+ ...
13
+ [-0.986 -0.744 -0.5024 ... 0.819 -0.3748 1. ]
14
+ [-1. -0.885 -0.538 ... 0.602 -0.3564 1. ]
15
+ [-0.953 -0.6567 -0.709 ... 0.7773 -0.3164 1. ]]
16
+
17
+ [[ 0.967 0.3438 0.7964 ... 0.799 0.0395 0. ]
18
+ [ 0.996 0.1466 0.622 ... 0.814 0.1362 0. ]
19
+ [ 1. 0.001759 0.587 ... 0.789 0.11444 0. ]
20
+ ...
21
+ [ 0.8145 -0.8496 0.4817 ... 0.9663 -0.3801 0. ]
22
+ [ 0.8154 -0.75 0.4265 ... 0.9473 -0.4224 0. ]
23
+ [ 0.8086 -0.603 0.4565 ... 0.921 -0.5337 0. ]]
24
+
25
+ [[ 0.996 0.1466 0.622 ... 0.814 0.1362 0. ]
26
+ [ 1. 0.001759 0.587 ... 0.789 0.11444 0. ]
27
+ [ 0.9736 -0.3306 0.6187 ... 0.8887 -0.11096 0. ]
28
+ ...
29
+ [ 0.8154 -0.75 0.4265 ... 0.9473 -0.4224 0. ]
30
+ [ 0.8086 -0.603 0.4565 ... 0.921 -0.5337 0. ]
31
+ [ 0.7563 -0.3718 0.4092 ... 0.989 -0.2905 0. ]]
32
+
33
+ ...
34
+
35
+ [[ 0.4187 -0.04028 0.3442 ... 0.998 -0.2117 0. ]
36
+ [ 0.3718 0.1781 0.2927 ... 0.9727 -0.1244 0. ]
37
+ [ 0.2976 0.2998 0.322 ... 0.957 -0.2375 0. ]
38
+ ...
39
+ [ 0.08936 -0.08746 0.1663 ... 0.967 -0.4133 1. ]
40
+ [ 0.1576 0.2632 0.0382 ... 0.932 -0.2438 1. ]
41
+ [ 0.1512 0.3826 -0.05908 ... 0.947 -0.00794 1. ]]
42
+
43
+ [[ 0.575 -0.272 0.626 ... 0.935 -0.0478 0. ]
44
+ [ 0.602 -0.4526 0.3503 ... 0.9785 -0.247 0. ]
45
+ [ 0.6235 -0.707 0.1525 ... 0.921 -0.5557 0. ]
46
+ ...
47
+ [ 0.2118 0.3596 -0.1791 ... 0.9917 -0.3455 0. ]
48
+ [ 0.211 0.293 -0.2754 ... 0.9463 0.00838 0. ]
49
+ [ 0.1984 0.03632 -0.3152 ... 0.845 0.0955 0. ]]
50
+
51
+ [[ 0.2118 0.3596 -0.1791 ... 0.9917 -0.3455 0. ]
52
+ [ 0.211 0.293 -0.2754 ... 0.9463 0.00838 0. ]
53
+ [ 0.1984 0.03632 -0.3152 ... 0.845 0.0955 0. ]
54
+ ...
55
+ [ 0.01865 -0.2166 -0.0659 ... 0.981 -0.3027 1. ]
56
+ [-0.09705 -0.1228 0.1356 ... 0.9434 -0.0745 1. ]
57
+ [-0.2311 -0.0881 0.1273 ... 0.8384 0.1476 1. ]]]
58
+ 03/13 [08:21:49] INFO  | >> Step 100, Loss: {'action_dit_loss': 835813.9375, 'mse_score': 27.757119750976564, 'data_time': ]8;id=45561;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=765179;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
59
+   0.13128810608759522, 'model_time': 2.581426488235593, 'learning_rate': 2.0000000000000002e-07, 'epoch': 50.0})  
60
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 442, in <module>
61
+ main(cfg)
62
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 412, in main
63
+ trainer.train()
64
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
65
+ step_metrics = self._train_step(batch_vla)
66
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 356, in _train_step
67
+ self.accelerator.backward(total_loss)
68
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/accelerate/accelerator.py", line 2351, in backward
69
+ self.deepspeed_engine_wrapped.backward(loss, **kwargs)
70
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/accelerate/utils/deepspeed.py", line 275, in backward
71
+ self.engine.step()
72
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/runtime/engine.py", line 2382, in step
73
+ self.tput_timer.stop(global_step=self.is_gradient_accumulation_boundary(), report_speed=report_progress)
74
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/utils/timer.py", line 256, in stop
75
+ get_accelerator().synchronize()
76
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/accelerator/cuda_accelerator.py", line 79, in synchronize
77
+ return torch.cuda.synchronize(device_index)
78
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/cuda/__init__.py", line 1040, in synchronize
79
+ return torch._C._cuda_synchronize()
80
+ KeyboardInterrupt
81
+ [rank0]: Traceback (most recent call last):
82
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 442, in <module>
83
+ [rank0]: main(cfg)
84
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 412, in main
85
+ [rank0]: trainer.train()
86
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
87
+ [rank0]: step_metrics = self._train_step(batch_vla)
88
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 356, in _train_step
89
+ [rank0]: self.accelerator.backward(total_loss)
90
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/accelerate/accelerator.py", line 2351, in backward
91
+ [rank0]: self.deepspeed_engine_wrapped.backward(loss, **kwargs)
92
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/accelerate/utils/deepspeed.py", line 275, in backward
93
+ [rank0]: self.engine.step()
94
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/runtime/engine.py", line 2382, in step
95
+ [rank0]: self.tput_timer.stop(global_step=self.is_gradient_accumulation_boundary(), report_speed=report_progress)
96
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/utils/timer.py", line 256, in stop
97
+ [rank0]: get_accelerator().synchronize()
98
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/accelerator/cuda_accelerator.py", line 79, in synchronize
99
+ [rank0]: return torch.cuda.synchronize(device_index)
100
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/cuda/__init__.py", line 1040, in synchronize
101
+ [rank0]: return torch._C._cuda_synchronize()
102
+ [rank0]: KeyboardInterrupt
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/wandb-metadata.json ADDED
@@ -0,0 +1,161 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-13T08:16:53.333287Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "./examples/calvin/train_files/starvla_train_calvin.yaml",
8
+ "--framework.name",
9
+ "QwenPI",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.attn_implementation",
13
+ "flash_attention_2",
14
+ "--framework.action_model.action_dim",
15
+ "10",
16
+ "--framework.action_model.state_dim",
17
+ "10",
18
+ "--framework.action_model.future_action_window_size",
19
+ "15",
20
+ "--framework.action_model.past_action_window_size",
21
+ "0",
22
+ "--framework.action_model.action_hidden_dim",
23
+ "1024",
24
+ "--framework.action_model.hidden_size",
25
+ "1024",
26
+ "--framework.action_model.action_model_type",
27
+ "DiT-B",
28
+ "--framework.action_model.add_pos_embed",
29
+ "True",
30
+ "--framework.action_model.max_seq_len",
31
+ "1024",
32
+ "--framework.action_model.noise_beta_alpha",
33
+ "1.5",
34
+ "--framework.action_model.noise_beta_beta",
35
+ "1.0",
36
+ "--framework.action_model.noise_s",
37
+ "0.999",
38
+ "--framework.action_model.num_timestep_buckets",
39
+ "1000",
40
+ "--framework.action_model.num_inference_timesteps",
41
+ "4",
42
+ "--framework.action_model.num_target_vision_tokens",
43
+ "32",
44
+ "--datasets.vla_data.data_root_dir",
45
+ "playground/Datasets/FastUMI",
46
+ "--datasets.vla_data.data_mix",
47
+ "fastumi_pickandplace_debug_1ep",
48
+ "--datasets.vla_data.include_state",
49
+ "true",
50
+ "--datasets.vla_data.per_device_batch_size",
51
+ "8",
52
+ "--datasets.vla_data.video_backend",
53
+ "torchvision_av",
54
+ "--trainer.freeze_modules",
55
+ "",
56
+ "--trainer.max_train_steps",
57
+ "20000",
58
+ "--trainer.save_interval",
59
+ "5000",
60
+ "--trainer.logging_frequency",
61
+ "50",
62
+ "--trainer.eval_interval",
63
+ "100",
64
+ "--trainer.gradient_accumulation_steps",
65
+ "1",
66
+ "--trainer.is_resume",
67
+ "true",
68
+ "--run_root_dir",
69
+ "./results/Checkpoints",
70
+ "--run_id",
71
+ "fastumi_pickandplace_qwenPI",
72
+ "--wandb_project",
73
+ "starVLA_FastUMI",
74
+ "--wandb_entity",
75
+ "2200011093-peking-university"
76
+ ],
77
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
78
+ "codePath": "starVLA/training/train_starvla.py",
79
+ "codePathLocal": "starVLA/training/train_starvla.py",
80
+ "git": {
81
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
82
+ "commit": "66b43863ede17a0e3081f822346e95cec52816d0"
83
+ },
84
+ "email": "wangpc@berkeley.edu",
85
+ "root": "./results/Checkpoints/fastumi_pickandplace_qwenPI/wandb",
86
+ "host": "tams02",
87
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
88
+ "cpu_count": 192,
89
+ "cpu_count_logical": 384,
90
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
91
+ "gpu_count": 8,
92
+ "disk": {
93
+ "/": {
94
+ "total": "3776651378688",
95
+ "used": "136324956160"
96
+ }
97
+ },
98
+ "memory": {
99
+ "total": "1081550508032"
100
+ },
101
+ "gpu_nvidia": [
102
+ {
103
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
104
+ "memoryTotal": "102641958912",
105
+ "cudaCores": 24064,
106
+ "architecture": "Blackwell",
107
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
108
+ },
109
+ {
110
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
111
+ "memoryTotal": "102641958912",
112
+ "cudaCores": 24064,
113
+ "architecture": "Blackwell",
114
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
115
+ },
116
+ {
117
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
118
+ "memoryTotal": "102641958912",
119
+ "cudaCores": 24064,
120
+ "architecture": "Blackwell",
121
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
122
+ },
123
+ {
124
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
125
+ "memoryTotal": "102641958912",
126
+ "cudaCores": 24064,
127
+ "architecture": "Blackwell",
128
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
129
+ },
130
+ {
131
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
132
+ "memoryTotal": "102641958912",
133
+ "cudaCores": 24064,
134
+ "architecture": "Blackwell",
135
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
136
+ },
137
+ {
138
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
139
+ "memoryTotal": "102641958912",
140
+ "cudaCores": 24064,
141
+ "architecture": "Blackwell",
142
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
143
+ },
144
+ {
145
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
146
+ "memoryTotal": "102641958912",
147
+ "cudaCores": 24064,
148
+ "architecture": "Blackwell",
149
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
150
+ },
151
+ {
152
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
153
+ "memoryTotal": "102641958912",
154
+ "cudaCores": 24064,
155
+ "architecture": "Blackwell",
156
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
157
+ }
158
+ ],
159
+ "cudaVersion": "13.1",
160
+ "writerId": "xevzqrqcaigt4lan5smfrjd6ooaz8cs8"
161
+ }
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"model_time":2.581426488235593,"epoch":50,"mse_score":27.757119750976564,"_wandb":{"runtime":342},"learning_rate":2.0000000000000002e-07,"_step":100,"_timestamp":1.7733901095790825e+09,"action_dit_loss":835813.9375,"data_time":0.13128810608759522,"_runtime":342.109064446}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/logs/debug-core.log ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T08:16:53.39400832Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpyy1z52ef/port-615784.txt","pid":615784,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-13T08:16:53.394784787Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":615784}
3
+ {"time":"2026-03-13T08:16:53.394764578Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-615784-618934-428435638/socket","Net":"unix"}}
4
+ {"time":"2026-03-13T08:16:53.571270762Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-13T08:16:53.576086863Z","level":"INFO","msg":"handleInformInit: received","streamId":"va5ln8ez","id":"1(@)"}
6
+ {"time":"2026-03-13T08:16:53.984514283Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"va5ln8ez","id":"1(@)"}
7
+ {"time":"2026-03-13T08:16:59.318994833Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"ccuz7uenl8aa"}
8
+ {"time":"2026-03-13T08:22:36.301837643Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"}
9
+ {"time":"2026-03-13T08:22:36.301921307Z","level":"INFO","msg":"connection: closing","id":"1(@)"}
10
+ {"time":"2026-03-13T08:22:36.302049213Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"}
11
+ {"time":"2026-03-13T08:22:36.301957128Z","level":"INFO","msg":"server is shutting down"}
12
+ {"time":"2026-03-13T08:22:36.302234532Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-615784-618934-428435638/socket","Net":"unix"}}
13
+ {"time":"2026-03-13T08:22:36.994380393Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/logs/debug-internal.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-13T08:16:53.576348519Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-13T08:16:53.984235888Z","level":"INFO","msg":"stream: created new stream","id":"va5ln8ez"}
3
+ {"time":"2026-03-13T08:16:53.984371145Z","level":"INFO","msg":"handler: started","stream_id":"va5ln8ez"}
4
+ {"time":"2026-03-13T08:16:53.984507673Z","level":"INFO","msg":"stream: started","id":"va5ln8ez"}
5
+ {"time":"2026-03-13T08:16:53.984532793Z","level":"INFO","msg":"writer: started","stream_id":"va5ln8ez"}
6
+ {"time":"2026-03-13T08:16:53.984536863Z","level":"INFO","msg":"sender: started","stream_id":"va5ln8ez"}
7
+ {"time":"2026-03-13T08:22:36.301914576Z","level":"INFO","msg":"stream: closing","id":"va5ln8ez"}
8
+ {"time":"2026-03-13T08:22:36.688445449Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/logs/debug.log ADDED
File without changes
fastumi_pickandplace_qwenPI/wandb/wandb/run-20260313_081653-va5ln8ez/run-va5ln8ez.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d8193dab286188ca950c2b58a9f77e3f6cef5d4f1e1bab2e951cd29f26eb2b01
3
+ size 131072