outsider86 commited on
Commit
e408d9d
·
verified ·
1 Parent(s): 32777a9

Upload folder using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +4 -0
  2. fastumi_pickandplace_discrete_diffusion_real_0314/checkpoints/steps_10000_pytorch_model.pt +3 -0
  3. fastumi_pickandplace_discrete_diffusion_real_0314/checkpoints/steps_20000_pytorch_model.pt +3 -0
  4. fastumi_pickandplace_discrete_diffusion_real_0314/config.yaml +77 -0
  5. fastumi_pickandplace_discrete_diffusion_real_0314/dataset_statistics.json +178 -0
  6. fastumi_pickandplace_discrete_diffusion_real_0314/summary.jsonl +3 -0
  7. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/debug-internal.log +6 -0
  8. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/debug.log +0 -0
  9. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/files/output.log +18 -0
  10. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/files/requirements.txt +154 -0
  11. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/files/wandb-metadata.json +139 -0
  12. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/logs/debug-core.log +8 -0
  13. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/logs/debug-internal.log +6 -0
  14. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/logs/debug.log +0 -0
  15. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/run-hnhhpo39.wandb +3 -0
  16. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/files/output.log +57 -0
  17. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/files/requirements.txt +154 -0
  18. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/files/wandb-metadata.json +141 -0
  19. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/files/wandb-summary.json +1 -0
  20. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/logs/debug-core.log +12 -0
  21. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/logs/debug-internal.log +7 -0
  22. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/logs/debug.log +0 -0
  23. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/run-fihipg1g.wandb +0 -0
  24. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/files/output.log +57 -0
  25. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/files/requirements.txt +154 -0
  26. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/files/wandb-metadata.json +141 -0
  27. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/files/wandb-summary.json +1 -0
  28. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/logs/debug-core.log +12 -0
  29. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/logs/debug-internal.log +7 -0
  30. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/logs/debug.log +0 -0
  31. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/run-g0o0lo22.wandb +0 -0
  32. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/files/output.log +14 -0
  33. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/files/requirements.txt +154 -0
  34. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/files/wandb-metadata.json +141 -0
  35. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/logs/debug-core.log +8 -0
  36. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/logs/debug-internal.log +6 -0
  37. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/logs/debug.log +0 -0
  38. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/run-so4tb51q.wandb +3 -0
  39. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/files/output.log +0 -0
  40. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/files/requirements.txt +154 -0
  41. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/files/wandb-metadata.json +141 -0
  42. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/logs/debug-core.log +8 -0
  43. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/logs/debug-internal.log +6 -0
  44. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/logs/debug.log +0 -0
  45. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/run-do490bua.wandb +3 -0
  46. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/files/output.log +0 -0
  47. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/files/requirements.txt +154 -0
  48. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/files/wandb-metadata.json +141 -0
  49. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/logs/debug-core.log +8 -0
  50. fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/logs/debug-internal.log +6 -0
.gitattributes CHANGED
@@ -33,3 +33,7 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/run-hnhhpo39.wandb filter=lfs diff=lfs merge=lfs -text
37
+ fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/run-so4tb51q.wandb filter=lfs diff=lfs merge=lfs -text
38
+ fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/run-do490bua.wandb filter=lfs diff=lfs merge=lfs -text
39
+ fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/run-3mz29zso.wandb filter=lfs diff=lfs merge=lfs -text
fastumi_pickandplace_discrete_diffusion_real_0314/checkpoints/steps_10000_pytorch_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6eacde2e0195fab31f7a28a8c48f223700f48974323a01c75b4180a7cee37983
3
+ size 12492717308
fastumi_pickandplace_discrete_diffusion_real_0314/checkpoints/steps_20000_pytorch_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:505f880f0377b7bf319362bb6a20b7358bc42bdff4eee83d234252093c92c986
3
+ size 12492717308
fastumi_pickandplace_discrete_diffusion_real_0314/config.yaml ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ datasets:
2
+ vla_data:
3
+ action_mode: abs
4
+ data_mix: fastumi_pickandplace_ur5_0314
5
+ data_root_dir: playground/Datasets/FastUMI
6
+ dataset_py: lerobot_datasets
7
+ image_size:
8
+ - 224
9
+ - 224
10
+ per_device_batch_size: 8
11
+ video_backend: torchvision_av
12
+ framework:
13
+ action_model:
14
+ action_dim: 10
15
+ add_pos_embed: true
16
+ decode_schedule: cosine
17
+ diffusion_model_cfg:
18
+ attention_head_dim: 64
19
+ cross_attention_dim: 2048
20
+ dropout: 0.2
21
+ final_dropout: true
22
+ input_embedding_dim: 2048
23
+ interleave_self_attention: true
24
+ norm_type: ada_norm
25
+ num_attention_heads: 32
26
+ num_layers: 36
27
+ output_dim: 256
28
+ positional_embeddings: sinusoidal
29
+ future_action_window_size: 15
30
+ l1_loss_weight: 0.1
31
+ max_seq_len: 1024
32
+ no_mask_token_prob: 0.0
33
+ num_bins: 256
34
+ num_inference_steps: 8
35
+ num_target_vision_tokens: 32
36
+ past_action_window_size: 0
37
+ representation: bin
38
+ state_dim: 10
39
+ train_mask_schedule: cosine
40
+ use_simple_max: false
41
+ name: QwenDiscreteDiffusion
42
+ qwenvl:
43
+ attn_implementation: flash_attention_2
44
+ base_vlm: playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action
45
+ num_vl_layers: 36
46
+ vl_hidden_dim: 2048
47
+ output_dir: ./results/Checkpoints/fastumi_pickandplace_discrete_diffusion_real_0314
48
+ run_id: fastumi_pickandplace_discrete_diffusion_real_0314
49
+ run_root_dir: ./results/Checkpoints
50
+ seed: 42
51
+ trainer:
52
+ eval_interval: 100
53
+ freeze_modules: null
54
+ gradient_accumulation_steps: 1
55
+ gradient_clipping: 1.0
56
+ is_resume: false
57
+ learning_rate:
58
+ action_model: 0.0001
59
+ base: 1.0e-05
60
+ qwen_vl_interface: 1.0e-05
61
+ logging_frequency: 50
62
+ lr_scheduler_type: cosine_with_min_lr
63
+ max_train_steps: 30000
64
+ num_warmup_steps: 5000
65
+ optimizer:
66
+ betas:
67
+ - 0.9
68
+ - 0.95
69
+ eps: 1.0e-08
70
+ weight_decay: 1.0e-08
71
+ repeated_diffusion_steps: 4
72
+ save_format: pt
73
+ save_interval: 10000
74
+ scheduler_specific_kwargs:
75
+ min_lr: 5.0e-07
76
+ wandb_entity: 2200011093-peking-university
77
+ wandb_project: starVLA_FastUMI
fastumi_pickandplace_discrete_diffusion_real_0314/dataset_statistics.json ADDED
@@ -0,0 +1,178 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "action": {
4
+ "mean": [
5
+ -0.00017731482512317598,
6
+ 6.326655420707539e-05,
7
+ -0.0003823576553259045,
8
+ 0.9999275803565979,
9
+ 0.00015219292254187167,
10
+ -7.193408237071708e-05,
11
+ -0.00015269513824023306,
12
+ 0.9998728632926941,
13
+ -0.0001340110320597887,
14
+ 0.23157350718975067
15
+ ],
16
+ "std": [
17
+ 0.0029254474211484194,
18
+ 0.00266967061907053,
19
+ 0.004045490175485611,
20
+ 0.0004587694129540108,
21
+ 0.011643894948065281,
22
+ 0.0005491027259267867,
23
+ 0.011645006015896797,
24
+ 0.0008708133827583353,
25
+ 0.011052136309444904,
26
+ 0.24192321300506592
27
+ ],
28
+ "max": [
29
+ 0.011658241041004658,
30
+ 0.01042273361235857,
31
+ 0.0110545065253973,
32
+ 1.0000001192092896,
33
+ 0.08344583958387375,
34
+ 0.0038863078225404024,
35
+ 0.08254292607307434,
36
+ 1.0000001192092896,
37
+ 0.079257071018219,
38
+ 0.5245640277862549
39
+ ],
40
+ "min": [
41
+ -0.013589359819889069,
42
+ -0.013344451785087585,
43
+ -0.012630008161067963,
44
+ 0.9965112805366516,
45
+ -0.08266565203666687,
46
+ -0.01173458807170391,
47
+ -0.08334038406610489,
48
+ 0.9939950704574585,
49
+ -0.07899841666221619,
50
+ 0.009166000410914421
51
+ ],
52
+ "q01": [
53
+ -0.007945208344608545,
54
+ -0.007611272423528135,
55
+ -0.009110004445537924,
56
+ 0.9968565315008163,
57
+ -0.000308865460101515,
58
+ -0.0032842739787884052,
59
+ -0.07796837285161018,
60
+ 0.9940040111541748,
61
+ -0.07265709199011326,
62
+ 0.009802999906241894
63
+ ],
64
+ "q99": [
65
+ 0.007261873693205412,
66
+ 0.007667668941430744,
67
+ 0.008452737815678119,
68
+ 1.0,
69
+ 0.07798104047775269,
70
+ 0.0002730701782274991,
71
+ 0.0003088487408240318,
72
+ 1.0,
73
+ 0.0004576626507332527,
74
+ 0.5240129828453064
75
+ ],
76
+ "mask": [
77
+ true,
78
+ true,
79
+ true,
80
+ true,
81
+ true,
82
+ true,
83
+ true,
84
+ true,
85
+ true,
86
+ false
87
+ ],
88
+ "norm_modes": [
89
+ "min_max",
90
+ "min_max",
91
+ "min_max",
92
+ "none",
93
+ "none",
94
+ "none",
95
+ "none",
96
+ "none",
97
+ "none",
98
+ "binary"
99
+ ]
100
+ },
101
+ "state": {
102
+ "mean": [
103
+ 0.34298810362815857,
104
+ -0.29047951102256775,
105
+ 0.08274945616722107,
106
+ 0.9994291067123413,
107
+ -0.0020131091587245464,
108
+ -0.0012421202845871449,
109
+ 0.002004049951210618,
110
+ 0.9983507394790649,
111
+ 0.0019875187426805496,
112
+ 0.22578883171081543
113
+ ],
114
+ "std": [
115
+ 0.04991065338253975,
116
+ 0.0422644317150116,
117
+ 0.06237703189253807,
118
+ 0.00019692142086756705,
119
+ 0.039019376039505005,
120
+ 0.0012851571664214134,
121
+ 0.03903917223215103,
122
+ 0.00019175464691510824,
123
+ 0.03722655400633812,
124
+ 0.24103344976902008
125
+ ],
126
+ "max": [
127
+ 0.49783000349998474,
128
+ -0.10034999996423721,
129
+ 0.32138198614120483,
130
+ 0.9993224143981934,
131
+ 0.04183037206530571,
132
+ 0.004732622765004635,
133
+ 0.04235748574137688,
134
+ 0.9986677765846252,
135
+ 0.03970242291688919,
136
+ 0.5245640277862549
137
+ ],
138
+ "min": [
139
+ 0.19434399902820587,
140
+ -0.44425100088119507,
141
+ 0.0019030000548809767,
142
+ 0.999085009098053,
143
+ -0.042528580874204636,
144
+ -0.005806926172226667,
145
+ -0.04183799773454666,
146
+ 0.9984964728355408,
147
+ -0.03961476683616638,
148
+ 0.009166000410914421
149
+ ],
150
+ "q01": [
151
+ 0.23710880234837534,
152
+ -0.3849276554584503,
153
+ 0.00841386997140944,
154
+ 0.9991415911912918,
155
+ -0.041123956106603146,
156
+ -0.004370019193738699,
157
+ -0.04099516797810793,
158
+ 0.9984980225563049,
159
+ -0.03880806788802147,
160
+ 0.009802999906241894
161
+ ],
162
+ "q99": [
163
+ 0.46919367551803587,
164
+ -0.1585090310871601,
165
+ 0.2566446015238762,
166
+ 0.999306601881981,
167
+ 0.04096023611724377,
168
+ 0.002124080888461314,
169
+ 0.04100460018962622,
170
+ 0.9986270666122437,
171
+ 0.03915046207606793,
172
+ 0.5240129828453064
173
+ ]
174
+ },
175
+ "num_transitions": 26330,
176
+ "num_trajectories": 300
177
+ }
178
+ }
fastumi_pickandplace_discrete_diffusion_real_0314/summary.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {"steps": 10000}
2
+ {"steps": 20000}
3
+ {"steps": 10000}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/debug-internal.log ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T23:16:24.022999749Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-14T23:16:24.321792721Z","level":"INFO","msg":"stream: created new stream","id":"3mz29zso"}
3
+ {"time":"2026-03-14T23:16:24.321909878Z","level":"INFO","msg":"handler: started","stream_id":"3mz29zso"}
4
+ {"time":"2026-03-14T23:16:24.322529204Z","level":"INFO","msg":"stream: started","id":"3mz29zso"}
5
+ {"time":"2026-03-14T23:16:24.323672089Z","level":"INFO","msg":"writer: started","stream_id":"3mz29zso"}
6
+ {"time":"2026-03-14T23:16:24.323827905Z","level":"INFO","msg":"sender: started","stream_id":"3mz29zso"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/debug.log ADDED
File without changes
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/files/output.log ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 03/14 [07:35:45] INFO  | >> ***** Training Configuration ***** ]8;id=935518;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=571858;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#358\358]8;;\
2
+   INFO  | >> Total optimization steps = 30000 ]8;id=98246;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=229258;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#359\359]8;;\
3
+   INFO  | >> Per device batch size = 8 ]8;id=208496;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=750800;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#360\360]8;;\
4
+   INFO  | >> Gradient accumulation steps = 1 ]8;id=471029;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=617889;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#361\361]8;;\
5
+   INFO  | >> Total batch size = 64 ]8;id=844962;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=167414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#362\362]8;;\
6
+ 1%|▊ | 229/30000 [07:23<15:46:54, 1.91s/it, data_times=0.002, model_times=1.905]
7
+ 03/14 [07:37:22] INFO  | >> Step 50, Loss: {'action_dit_loss': 46.85933303833008, 'data_time': 0.00015479931607842445, 'model_time': 1.9080919972620904, ]8;id=225772;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=800581;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#270\270]8;;\
8
+   'learning_rate': 1.0000000000000001e-07, 'epoch': 0.14})  
9
+ Unusual action found in eval: 3.921875
10
+ index: 948
11
+ 03/14 [07:38:59] INFO  | >> Step 100, Loss: {'action_dit_loss': 39.80457305908203, 'mse_score': 0.029430165886878967, 'data_time': 0.004903694614768028, ]8;id=101414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=376417;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#270\270]8;;\
12
+   'model_time': 1.9064660011790693, 'learning_rate': 2.0000000000000002e-07, 'epoch': 0.29})  
13
+ 03/14 [07:40:35] INFO  | >> Step 150, Loss: {'action_dit_loss': 5.0287909507751465, 'data_time': 0.002109064254909754, 'model_time': 1.9053435139358044, ]8;id=846335;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=45561;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#270\270]8;;\
14
+   'learning_rate': 3.0000000000000004e-07, 'epoch': 0.43})  
15
+ Unusual action found in eval: 21152.0
16
+ index: 547
17
+ 03/14 [07:42:13] INFO  | >> Step 200, Loss: {'action_dit_loss': 36.512062072753906, 'mse_score': 70.76109008789062, 'data_time': 0.0001514260657131672, ]8;id=967096;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=396922;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#270\270]8;;\
18
+   'model_time': 1.9083208129741251, 'learning_rate': 4.0000000000000003e-07, 'epoch': 0.57})  
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/files/wandb-metadata.json ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-14T07:35:43.998822Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "starVLA/config/training/starvla_train_discrete_diffusion_real.yaml",
8
+ "--framework.name",
9
+ "QwenDiscreteDiffusion",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.vl_hidden_dim",
13
+ "4096",
14
+ "--framework.qwenvl.attn_implementation",
15
+ "flash_attention_2",
16
+ "--framework.action_model.representation",
17
+ "bin",
18
+ "--framework.action_model.num_bins",
19
+ "256",
20
+ "--framework.action_model.action_low",
21
+ "-1.0",
22
+ "--framework.action_model.action_high",
23
+ "1.0",
24
+ "--framework.action_model.num_inference_steps",
25
+ "8",
26
+ "--datasets.vla_data.data_root_dir",
27
+ "playground/Datasets/FastUMI",
28
+ "--datasets.vla_data.data_mix",
29
+ "fastumi_pickandplace_real_0307",
30
+ "--datasets.vla_data.per_device_batch_size",
31
+ "8",
32
+ "--datasets.vla_data.video_backend",
33
+ "torchvision_av",
34
+ "--trainer.freeze_modules",
35
+ "",
36
+ "--trainer.max_train_steps",
37
+ "30000",
38
+ "--trainer.save_interval",
39
+ "10000",
40
+ "--trainer.logging_frequency",
41
+ "50",
42
+ "--trainer.eval_interval",
43
+ "100",
44
+ "--trainer.gradient_accumulation_steps",
45
+ "1",
46
+ "--run_root_dir",
47
+ "./results/Checkpoints",
48
+ "--run_id",
49
+ "fastumi_pickandplace_discrete_diffusion_real_0314",
50
+ "--wandb_project",
51
+ "starVLA_FastUMI",
52
+ "--wandb_entity",
53
+ "2200011093-peking-university"
54
+ ],
55
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
56
+ "codePath": "starVLA/training/train_starvla.py",
57
+ "codePathLocal": "starVLA/training/train_starvla.py",
58
+ "git": {
59
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
60
+ "commit": "23464348d9bec906692f0a794ebe536c9f09b496"
61
+ },
62
+ "email": "wangpc@berkeley.edu",
63
+ "root": "./results/Checkpoints/fastumi_pickandplace_discrete_diffusion_real_0314/wandb",
64
+ "host": "tams02",
65
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
66
+ "cpu_count": 192,
67
+ "cpu_count_logical": 384,
68
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
69
+ "gpu_count": 8,
70
+ "disk": {
71
+ "/": {
72
+ "total": "3776651378688",
73
+ "used": "136976064512"
74
+ }
75
+ },
76
+ "memory": {
77
+ "total": "1081550508032"
78
+ },
79
+ "gpu_nvidia": [
80
+ {
81
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
82
+ "memoryTotal": "102641958912",
83
+ "cudaCores": 24064,
84
+ "architecture": "Blackwell",
85
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
86
+ },
87
+ {
88
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
89
+ "memoryTotal": "102641958912",
90
+ "cudaCores": 24064,
91
+ "architecture": "Blackwell",
92
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
93
+ },
94
+ {
95
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
96
+ "memoryTotal": "102641958912",
97
+ "cudaCores": 24064,
98
+ "architecture": "Blackwell",
99
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
100
+ },
101
+ {
102
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
103
+ "memoryTotal": "102641958912",
104
+ "cudaCores": 24064,
105
+ "architecture": "Blackwell",
106
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
107
+ },
108
+ {
109
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
110
+ "memoryTotal": "102641958912",
111
+ "cudaCores": 24064,
112
+ "architecture": "Blackwell",
113
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
114
+ },
115
+ {
116
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
117
+ "memoryTotal": "102641958912",
118
+ "cudaCores": 24064,
119
+ "architecture": "Blackwell",
120
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
121
+ },
122
+ {
123
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
124
+ "memoryTotal": "102641958912",
125
+ "cudaCores": 24064,
126
+ "architecture": "Blackwell",
127
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
128
+ },
129
+ {
130
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
131
+ "memoryTotal": "102641958912",
132
+ "cudaCores": 24064,
133
+ "architecture": "Blackwell",
134
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
135
+ }
136
+ ],
137
+ "cudaVersion": "13.1",
138
+ "writerId": "v5gj68wjafmzztyrbxfr1tskmocom6xr"
139
+ }
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/logs/debug-core.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T07:35:44.057897261Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmp560dxhbe/port-2960016.txt","pid":2960016,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-14T07:35:44.058704004Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":2960016}
3
+ {"time":"2026-03-14T07:35:44.058694334Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-2960016-2969650-586358555/socket","Net":"unix"}}
4
+ {"time":"2026-03-14T07:35:44.237748932Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-14T07:35:44.243079011Z","level":"INFO","msg":"handleInformInit: received","streamId":"hnhhpo39","id":"1(@)"}
6
+ {"time":"2026-03-14T07:35:44.603815132Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"hnhhpo39","id":"1(@)"}
7
+ {"time":"2026-03-14T07:35:50.028679612Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"vecv95r70pv8"}
8
+ {"time":"2026-03-14T07:43:12.135524507Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/logs/debug-internal.log ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T07:35:44.243251719Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-14T07:35:44.603448096Z","level":"INFO","msg":"stream: created new stream","id":"hnhhpo39"}
3
+ {"time":"2026-03-14T07:35:44.603601034Z","level":"INFO","msg":"handler: started","stream_id":"hnhhpo39"}
4
+ {"time":"2026-03-14T07:35:44.603806132Z","level":"INFO","msg":"stream: started","id":"hnhhpo39"}
5
+ {"time":"2026-03-14T07:35:44.603865081Z","level":"INFO","msg":"sender: started","stream_id":"hnhhpo39"}
6
+ {"time":"2026-03-14T07:35:44.603864881Z","level":"INFO","msg":"writer: started","stream_id":"hnhhpo39"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/logs/debug.log ADDED
File without changes
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_073543-hnhhpo39/run-hnhhpo39.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:69943f425abf5c8545cf7847881f177a23b17061a4b861b41bca73081d431a96
3
+ size 196608
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/files/output.log ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 03/14 [07:50:29] INFO  | >> ***** Training Configuration ***** ]8;id=935518;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=571858;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#341\341]8;;\
2
+   INFO  | >> Total optimization steps = 30000 ]8;id=98246;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=229258;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#342\342]8;;\
3
+   INFO  | >> Per device batch size = 8 ]8;id=208496;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=750800;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#343\343]8;;\
4
+   INFO  | >> Gradient accumulation steps = 1 ]8;id=471029;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=617889;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#344\344]8;;\
5
+   INFO  | >> Total batch size = 64 ]8;id=844962;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=167414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#345\345]8;;\
6
+ 0%| | 0/30000 [00:00<?, ?it/s]Traceback (most recent call last):
7
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 443, in <module>
8
+ main(cfg)
9
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 413, in main
10
+ trainer.train()
11
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
12
+ step_metrics = self._train_step(batch_vla)
13
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 353, in _train_step
14
+ output_dict = self.model.forward(batch_vla)
15
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/utils/nvtx.py", line 20, in wrapped_fn
16
+ ret_val = func(*args, **kwargs)
17
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/runtime/engine.py", line 2054, in forward
18
+ loss = self.module(*inputs, **kwargs)
19
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
20
+ return self._call_impl(*args, **kwargs)
21
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1857, in _call_impl
22
+ return inner()
23
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1805, in inner
24
+ result = forward_call(*args, **kwargs)
25
+ File "/scratch/wangpc/starVLA/starVLA/model/framework/QwenDiscreteDiffusion.py", line 163, in forward
26
+ action_loss = self.action_model.loss(pred, target, **extra)
27
+ File "/scratch/wangpc/starVLA/starVLA/model/modules/action_model/LayerwiseDiscreteDiffusion_ActionHeader.py", line 228, in loss
28
+ bce = F.binary_cross_entropy_with_logits(
29
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/functional.py", line 3639, in binary_cross_entropy_with_logits
30
+ raise ValueError(
31
+ ValueError: Target size (torch.Size([16, 16, 10, 8])) must be the same as input size (torch.Size([16, 160, 8]))
32
+ [rank0]: Traceback (most recent call last):
33
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 443, in <module>
34
+ [rank0]: main(cfg)
35
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 413, in main
36
+ [rank0]: trainer.train()
37
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
38
+ [rank0]: step_metrics = self._train_step(batch_vla)
39
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 353, in _train_step
40
+ [rank0]: output_dict = self.model.forward(batch_vla)
41
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/utils/nvtx.py", line 20, in wrapped_fn
42
+ [rank0]: ret_val = func(*args, **kwargs)
43
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/runtime/engine.py", line 2054, in forward
44
+ [rank0]: loss = self.module(*inputs, **kwargs)
45
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
46
+ [rank0]: return self._call_impl(*args, **kwargs)
47
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1857, in _call_impl
48
+ [rank0]: return inner()
49
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1805, in inner
50
+ [rank0]: result = forward_call(*args, **kwargs)
51
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/model/framework/QwenDiscreteDiffusion.py", line 163, in forward
52
+ [rank0]: action_loss = self.action_model.loss(pred, target, **extra)
53
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/model/modules/action_model/LayerwiseDiscreteDiffusion_ActionHeader.py", line 228, in loss
54
+ [rank0]: bce = F.binary_cross_entropy_with_logits(
55
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/functional.py", line 3639, in binary_cross_entropy_with_logits
56
+ [rank0]: raise ValueError(
57
+ [rank0]: ValueError: Target size (torch.Size([16, 16, 10, 8])) must be the same as input size (torch.Size([16, 160, 8]))
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/files/wandb-metadata.json ADDED
@@ -0,0 +1,141 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-14T07:50:28.920069Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "starVLA/config/training/starvla_train_discrete_diffusion_real.yaml",
8
+ "--framework.name",
9
+ "QwenDiscreteDiffusion",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.vl_hidden_dim",
13
+ "4096",
14
+ "--framework.qwenvl.attn_implementation",
15
+ "flash_attention_2",
16
+ "--framework.action_model.representation",
17
+ "bit",
18
+ "--framework.action_model.num_bins",
19
+ "256",
20
+ "--framework.action_model.action_low",
21
+ "-1.0",
22
+ "--framework.action_model.action_high",
23
+ "1.0",
24
+ "--framework.action_model.num_inference_steps",
25
+ "8",
26
+ "--datasets.vla_data.data_root_dir",
27
+ "playground/Datasets/FastUMI",
28
+ "--datasets.vla_data.data_mix",
29
+ "fastumi_pickandplace_real_0307",
30
+ "--datasets.vla_data.include_state",
31
+ "true",
32
+ "--datasets.vla_data.per_device_batch_size",
33
+ "8",
34
+ "--datasets.vla_data.video_backend",
35
+ "torchvision_av",
36
+ "--trainer.freeze_modules",
37
+ "",
38
+ "--trainer.max_train_steps",
39
+ "30000",
40
+ "--trainer.save_interval",
41
+ "10000",
42
+ "--trainer.logging_frequency",
43
+ "50",
44
+ "--trainer.eval_interval",
45
+ "100",
46
+ "--trainer.gradient_accumulation_steps",
47
+ "1",
48
+ "--run_root_dir",
49
+ "./results/Checkpoints",
50
+ "--run_id",
51
+ "fastumi_pickandplace_discrete_diffusion_real_0314",
52
+ "--wandb_project",
53
+ "starVLA_FastUMI",
54
+ "--wandb_entity",
55
+ "2200011093-peking-university"
56
+ ],
57
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
58
+ "codePath": "starVLA/training/train_starvla.py",
59
+ "codePathLocal": "starVLA/training/train_starvla.py",
60
+ "git": {
61
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
62
+ "commit": "23464348d9bec906692f0a794ebe536c9f09b496"
63
+ },
64
+ "email": "wangpc@berkeley.edu",
65
+ "root": "./results/Checkpoints/fastumi_pickandplace_discrete_diffusion_real_0314/wandb",
66
+ "host": "tams02",
67
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
68
+ "cpu_count": 192,
69
+ "cpu_count_logical": 384,
70
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
71
+ "gpu_count": 8,
72
+ "disk": {
73
+ "/": {
74
+ "total": "3776651378688",
75
+ "used": "136976523264"
76
+ }
77
+ },
78
+ "memory": {
79
+ "total": "1081550508032"
80
+ },
81
+ "gpu_nvidia": [
82
+ {
83
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
84
+ "memoryTotal": "102641958912",
85
+ "cudaCores": 24064,
86
+ "architecture": "Blackwell",
87
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
88
+ },
89
+ {
90
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
91
+ "memoryTotal": "102641958912",
92
+ "cudaCores": 24064,
93
+ "architecture": "Blackwell",
94
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
95
+ },
96
+ {
97
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
98
+ "memoryTotal": "102641958912",
99
+ "cudaCores": 24064,
100
+ "architecture": "Blackwell",
101
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
102
+ },
103
+ {
104
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
105
+ "memoryTotal": "102641958912",
106
+ "cudaCores": 24064,
107
+ "architecture": "Blackwell",
108
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
109
+ },
110
+ {
111
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
112
+ "memoryTotal": "102641958912",
113
+ "cudaCores": 24064,
114
+ "architecture": "Blackwell",
115
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
116
+ },
117
+ {
118
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
119
+ "memoryTotal": "102641958912",
120
+ "cudaCores": 24064,
121
+ "architecture": "Blackwell",
122
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
123
+ },
124
+ {
125
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
126
+ "memoryTotal": "102641958912",
127
+ "cudaCores": 24064,
128
+ "architecture": "Blackwell",
129
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
130
+ },
131
+ {
132
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
133
+ "memoryTotal": "102641958912",
134
+ "cudaCores": 24064,
135
+ "architecture": "Blackwell",
136
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
137
+ }
138
+ ],
139
+ "cudaVersion": "13.1",
140
+ "writerId": "0xuyjdd2cxoh0ly8dr70ar73w2pbxen2"
141
+ }
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":1},"_runtime":1}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/logs/debug-core.log ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T07:50:28.982204658Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmprwxn4vpg/port-3201129.txt","pid":3201129,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-14T07:50:28.982895147Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":3201129}
3
+ {"time":"2026-03-14T07:50:28.982885387Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-3201129-3210840-3554627083/socket","Net":"unix"}}
4
+ {"time":"2026-03-14T07:50:29.158781184Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-14T07:50:29.161261705Z","level":"INFO","msg":"handleInformInit: received","streamId":"fihipg1g","id":"1(@)"}
6
+ {"time":"2026-03-14T07:50:29.516112516Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"fihipg1g","id":"1(@)"}
7
+ {"time":"2026-03-14T07:50:31.633133326Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"}
8
+ {"time":"2026-03-14T07:50:31.633219677Z","level":"INFO","msg":"server is shutting down"}
9
+ {"time":"2026-03-14T07:50:31.633207147Z","level":"INFO","msg":"connection: closing","id":"1(@)"}
10
+ {"time":"2026-03-14T07:50:31.633364839Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"}
11
+ {"time":"2026-03-14T07:50:31.633361559Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-3201129-3210840-3554627083/socket","Net":"unix"}}
12
+ {"time":"2026-03-14T07:50:31.696193587Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/logs/debug-internal.log ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T07:50:29.161425647Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-14T07:50:29.515856903Z","level":"INFO","msg":"stream: created new stream","id":"fihipg1g"}
3
+ {"time":"2026-03-14T07:50:29.515994335Z","level":"INFO","msg":"handler: started","stream_id":"fihipg1g"}
4
+ {"time":"2026-03-14T07:50:29.516103496Z","level":"INFO","msg":"stream: started","id":"fihipg1g"}
5
+ {"time":"2026-03-14T07:50:29.516151397Z","level":"INFO","msg":"writer: started","stream_id":"fihipg1g"}
6
+ {"time":"2026-03-14T07:50:29.516149407Z","level":"INFO","msg":"sender: started","stream_id":"fihipg1g"}
7
+ {"time":"2026-03-14T07:50:31.633228157Z","level":"INFO","msg":"stream: closing","id":"fihipg1g"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/logs/debug.log ADDED
File without changes
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075028-fihipg1g/run-fihipg1g.wandb ADDED
Binary file (7 Bytes). View file
 
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/files/output.log ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 03/14 [07:51:09] INFO  | >> ***** Training Configuration ***** ]8;id=935518;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=571858;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#341\341]8;;\
2
+   INFO  | >> Total optimization steps = 30000 ]8;id=98246;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=229258;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#342\342]8;;\
3
+   INFO  | >> Per device batch size = 8 ]8;id=208496;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=750800;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#343\343]8;;\
4
+   INFO  | >> Gradient accumulation steps = 1 ]8;id=471029;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=617889;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#344\344]8;;\
5
+   INFO  | >> Total batch size = 64 ]8;id=844962;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=167414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#345\345]8;;\
6
+ 0%| | 0/30000 [00:00<?, ?it/s]Traceback (most recent call last):
7
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 443, in <module>
8
+ main(cfg)
9
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 413, in main
10
+ trainer.train()
11
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
12
+ step_metrics = self._train_step(batch_vla)
13
+ File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 353, in _train_step
14
+ output_dict = self.model.forward(batch_vla)
15
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/utils/nvtx.py", line 20, in wrapped_fn
16
+ ret_val = func(*args, **kwargs)
17
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/runtime/engine.py", line 2054, in forward
18
+ loss = self.module(*inputs, **kwargs)
19
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
20
+ return self._call_impl(*args, **kwargs)
21
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1857, in _call_impl
22
+ return inner()
23
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1805, in inner
24
+ result = forward_call(*args, **kwargs)
25
+ File "/scratch/wangpc/starVLA/starVLA/model/framework/QwenDiscreteDiffusion.py", line 163, in forward
26
+ action_loss = self.action_model.loss(pred, target, **extra)
27
+ File "/scratch/wangpc/starVLA/starVLA/model/modules/action_model/LayerwiseDiscreteDiffusion_ActionHeader.py", line 228, in loss
28
+ bce = F.binary_cross_entropy_with_logits(
29
+ File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/functional.py", line 3639, in binary_cross_entropy_with_logits
30
+ raise ValueError(
31
+ ValueError: Target size (torch.Size([16, 16, 10, 8])) must be the same as input size (torch.Size([16, 160, 8]))
32
+ [rank0]: Traceback (most recent call last):
33
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 443, in <module>
34
+ [rank0]: main(cfg)
35
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 413, in main
36
+ [rank0]: trainer.train()
37
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 287, in train
38
+ [rank0]: step_metrics = self._train_step(batch_vla)
39
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py", line 353, in _train_step
40
+ [rank0]: output_dict = self.model.forward(batch_vla)
41
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/utils/nvtx.py", line 20, in wrapped_fn
42
+ [rank0]: ret_val = func(*args, **kwargs)
43
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/deepspeed/runtime/engine.py", line 2054, in forward
44
+ [rank0]: loss = self.module(*inputs, **kwargs)
45
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
46
+ [rank0]: return self._call_impl(*args, **kwargs)
47
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1857, in _call_impl
48
+ [rank0]: return inner()
49
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1805, in inner
50
+ [rank0]: result = forward_call(*args, **kwargs)
51
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/model/framework/QwenDiscreteDiffusion.py", line 163, in forward
52
+ [rank0]: action_loss = self.action_model.loss(pred, target, **extra)
53
+ [rank0]: File "/scratch/wangpc/starVLA/starVLA/model/modules/action_model/LayerwiseDiscreteDiffusion_ActionHeader.py", line 228, in loss
54
+ [rank0]: bce = F.binary_cross_entropy_with_logits(
55
+ [rank0]: File "/home/wangpc/miniconda3/envs/starVLA/lib/python3.10/site-packages/torch/nn/functional.py", line 3639, in binary_cross_entropy_with_logits
56
+ [rank0]: raise ValueError(
57
+ [rank0]: ValueError: Target size (torch.Size([16, 16, 10, 8])) must be the same as input size (torch.Size([16, 160, 8]))
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/files/wandb-metadata.json ADDED
@@ -0,0 +1,141 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-14T07:51:08.016730Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "starVLA/config/training/starvla_train_discrete_diffusion_real.yaml",
8
+ "--framework.name",
9
+ "QwenDiscreteDiffusion",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.vl_hidden_dim",
13
+ "4096",
14
+ "--framework.qwenvl.attn_implementation",
15
+ "flash_attention_2",
16
+ "--framework.action_model.representation",
17
+ "bit",
18
+ "--framework.action_model.num_bins",
19
+ "256",
20
+ "--framework.action_model.action_low",
21
+ "-1.0",
22
+ "--framework.action_model.action_high",
23
+ "1.0",
24
+ "--framework.action_model.num_inference_steps",
25
+ "8",
26
+ "--datasets.vla_data.data_root_dir",
27
+ "playground/Datasets/FastUMI",
28
+ "--datasets.vla_data.data_mix",
29
+ "fastumi_pickandplace_real_0307",
30
+ "--datasets.vla_data.include_state",
31
+ "true",
32
+ "--datasets.vla_data.per_device_batch_size",
33
+ "8",
34
+ "--datasets.vla_data.video_backend",
35
+ "torchvision_av",
36
+ "--trainer.freeze_modules",
37
+ "",
38
+ "--trainer.max_train_steps",
39
+ "30000",
40
+ "--trainer.save_interval",
41
+ "10000",
42
+ "--trainer.logging_frequency",
43
+ "50",
44
+ "--trainer.eval_interval",
45
+ "100",
46
+ "--trainer.gradient_accumulation_steps",
47
+ "1",
48
+ "--run_root_dir",
49
+ "./results/Checkpoints",
50
+ "--run_id",
51
+ "fastumi_pickandplace_discrete_diffusion_real_0314",
52
+ "--wandb_project",
53
+ "starVLA_FastUMI",
54
+ "--wandb_entity",
55
+ "2200011093-peking-university"
56
+ ],
57
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
58
+ "codePath": "starVLA/training/train_starvla.py",
59
+ "codePathLocal": "starVLA/training/train_starvla.py",
60
+ "git": {
61
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
62
+ "commit": "23464348d9bec906692f0a794ebe536c9f09b496"
63
+ },
64
+ "email": "wangpc@berkeley.edu",
65
+ "root": "./results/Checkpoints/fastumi_pickandplace_discrete_diffusion_real_0314/wandb",
66
+ "host": "tams02",
67
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
68
+ "cpu_count": 192,
69
+ "cpu_count_logical": 384,
70
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
71
+ "gpu_count": 8,
72
+ "disk": {
73
+ "/": {
74
+ "total": "3776651378688",
75
+ "used": "136976920576"
76
+ }
77
+ },
78
+ "memory": {
79
+ "total": "1081550508032"
80
+ },
81
+ "gpu_nvidia": [
82
+ {
83
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
84
+ "memoryTotal": "102641958912",
85
+ "cudaCores": 24064,
86
+ "architecture": "Blackwell",
87
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
88
+ },
89
+ {
90
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
91
+ "memoryTotal": "102641958912",
92
+ "cudaCores": 24064,
93
+ "architecture": "Blackwell",
94
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
95
+ },
96
+ {
97
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
98
+ "memoryTotal": "102641958912",
99
+ "cudaCores": 24064,
100
+ "architecture": "Blackwell",
101
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
102
+ },
103
+ {
104
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
105
+ "memoryTotal": "102641958912",
106
+ "cudaCores": 24064,
107
+ "architecture": "Blackwell",
108
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
109
+ },
110
+ {
111
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
112
+ "memoryTotal": "102641958912",
113
+ "cudaCores": 24064,
114
+ "architecture": "Blackwell",
115
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
116
+ },
117
+ {
118
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
119
+ "memoryTotal": "102641958912",
120
+ "cudaCores": 24064,
121
+ "architecture": "Blackwell",
122
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
123
+ },
124
+ {
125
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
126
+ "memoryTotal": "102641958912",
127
+ "cudaCores": 24064,
128
+ "architecture": "Blackwell",
129
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
130
+ },
131
+ {
132
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
133
+ "memoryTotal": "102641958912",
134
+ "cudaCores": 24064,
135
+ "architecture": "Blackwell",
136
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
137
+ }
138
+ ],
139
+ "cudaVersion": "13.1",
140
+ "writerId": "4y3d4kfghfe8lym75zdznrbplulqsqp8"
141
+ }
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":2},"_runtime":2}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/logs/debug-core.log ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T07:51:08.079936625Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmp06s6x5uk/port-3212832.txt","pid":3212832,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-14T07:51:08.080520622Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":3212832}
3
+ {"time":"2026-03-14T07:51:08.080508302Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-3212832-3222509-2209415876/socket","Net":"unix"}}
4
+ {"time":"2026-03-14T07:51:08.255985727Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-14T07:51:08.260485239Z","level":"INFO","msg":"handleInformInit: received","streamId":"g0o0lo22","id":"1(@)"}
6
+ {"time":"2026-03-14T07:51:08.601289022Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"g0o0lo22","id":"1(@)"}
7
+ {"time":"2026-03-14T07:51:10.934592444Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"}
8
+ {"time":"2026-03-14T07:51:10.934698506Z","level":"INFO","msg":"server is shutting down"}
9
+ {"time":"2026-03-14T07:51:10.934693985Z","level":"INFO","msg":"connection: closing","id":"1(@)"}
10
+ {"time":"2026-03-14T07:51:10.934863977Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"}
11
+ {"time":"2026-03-14T07:51:10.934837707Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-3212832-3222509-2209415876/socket","Net":"unix"}}
12
+ {"time":"2026-03-14T07:51:11.091853479Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/logs/debug-internal.log ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T07:51:08.260665091Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-14T07:51:08.600376731Z","level":"INFO","msg":"stream: created new stream","id":"g0o0lo22"}
3
+ {"time":"2026-03-14T07:51:08.60114663Z","level":"INFO","msg":"handler: started","stream_id":"g0o0lo22"}
4
+ {"time":"2026-03-14T07:51:08.601281132Z","level":"INFO","msg":"stream: started","id":"g0o0lo22"}
5
+ {"time":"2026-03-14T07:51:08.601311262Z","level":"INFO","msg":"sender: started","stream_id":"g0o0lo22"}
6
+ {"time":"2026-03-14T07:51:08.601331502Z","level":"INFO","msg":"writer: started","stream_id":"g0o0lo22"}
7
+ {"time":"2026-03-14T07:51:10.934705206Z","level":"INFO","msg":"stream: closing","id":"g0o0lo22"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/logs/debug.log ADDED
File without changes
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075108-g0o0lo22/run-g0o0lo22.wandb ADDED
Binary file (7 Bytes). View file
 
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/files/output.log ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 03/14 [07:52:19] INFO  | >> ***** Training Configuration ***** ]8;id=935518;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=571858;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#341\341]8;;\
2
+   INFO  | >> Total optimization steps = 30000 ]8;id=98246;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=229258;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#342\342]8;;\
3
+   INFO  | >> Per device batch size = 8 ]8;id=208496;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=750800;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#343\343]8;;\
4
+   INFO  | >> Gradient accumulation steps = 1 ]8;id=471029;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=617889;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#344\344]8;;\
5
+   INFO  | >> Total batch size = 64 ]8;id=844962;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=167414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#345\345]8;;\
6
+ 1%|▋ | 228/30000 [07:23<16:03:26, 1.94s/it, data_times=0.000, model_times=1.916]
7
+ 03/14 [07:53:56] INFO  | >> Step 50, Loss: {'action_dit_loss': 5.35930871963501, 'data_time': 0.000176222063601017, 'model_time': ]8;id=225772;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=800581;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
8
+   1.9194549559615552, 'learning_rate': 1.0000000000000001e-07, 'epoch': 0.14})  
9
+ 03/14 [07:55:35] INFO  | >> Step 100, Loss: {'action_dit_loss': 4.3685526847839355, 'mse_score': 0.008448468148708343, 'data_time': ]8;id=101414;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=376417;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
10
+   0.00218858290463686, 'model_time': 1.914779453072697, 'learning_rate': 2.0000000000000002e-07, 'epoch': 0.29})  
11
+ 03/14 [07:57:11] INFO  | >> Step 150, Loss: {'action_dit_loss': 4.128368377685547, 'data_time': 0.0044462108053267, 'model_time': ]8;id=846335;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=45561;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
12
+   1.9115255461074412, 'learning_rate': 3.0000000000000004e-07, 'epoch': 0.43})  
13
+ 03/14 [07:58:49] INFO  | >> Step 200, Loss: {'action_dit_loss': 4.251475811004639, 'mse_score': 0.008416800200939179, 'data_time': ]8;id=967096;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py\train_starvla.py]8;;\:]8;id=396922;file:///scratch/wangpc/starVLA/starVLA/training/train_starvla.py#253\253]8;;\
14
+   0.0019990382716059685, 'model_time': 1.9176165140233934, 'learning_rate': 4.0000000000000003e-07, 'epoch': 0.57})  
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/files/wandb-metadata.json ADDED
@@ -0,0 +1,141 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-14T07:52:18.252748Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "starVLA/config/training/starvla_train_discrete_diffusion_real.yaml",
8
+ "--framework.name",
9
+ "QwenDiscreteDiffusion",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.vl_hidden_dim",
13
+ "4096",
14
+ "--framework.qwenvl.attn_implementation",
15
+ "flash_attention_2",
16
+ "--framework.action_model.representation",
17
+ "bin",
18
+ "--framework.action_model.num_bins",
19
+ "256",
20
+ "--framework.action_model.action_low",
21
+ "-1.0",
22
+ "--framework.action_model.action_high",
23
+ "1.0",
24
+ "--framework.action_model.num_inference_steps",
25
+ "8",
26
+ "--datasets.vla_data.data_root_dir",
27
+ "playground/Datasets/FastUMI",
28
+ "--datasets.vla_data.data_mix",
29
+ "fastumi_pickandplace_real_0307",
30
+ "--datasets.vla_data.include_state",
31
+ "true",
32
+ "--datasets.vla_data.per_device_batch_size",
33
+ "8",
34
+ "--datasets.vla_data.video_backend",
35
+ "torchvision_av",
36
+ "--trainer.freeze_modules",
37
+ "",
38
+ "--trainer.max_train_steps",
39
+ "30000",
40
+ "--trainer.save_interval",
41
+ "10000",
42
+ "--trainer.logging_frequency",
43
+ "50",
44
+ "--trainer.eval_interval",
45
+ "100",
46
+ "--trainer.gradient_accumulation_steps",
47
+ "1",
48
+ "--run_root_dir",
49
+ "./results/Checkpoints",
50
+ "--run_id",
51
+ "fastumi_pickandplace_discrete_diffusion_real_0314",
52
+ "--wandb_project",
53
+ "starVLA_FastUMI",
54
+ "--wandb_entity",
55
+ "2200011093-peking-university"
56
+ ],
57
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
58
+ "codePath": "starVLA/training/train_starvla.py",
59
+ "codePathLocal": "starVLA/training/train_starvla.py",
60
+ "git": {
61
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
62
+ "commit": "23464348d9bec906692f0a794ebe536c9f09b496"
63
+ },
64
+ "email": "wangpc@berkeley.edu",
65
+ "root": "./results/Checkpoints/fastumi_pickandplace_discrete_diffusion_real_0314/wandb",
66
+ "host": "tams02",
67
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
68
+ "cpu_count": 192,
69
+ "cpu_count_logical": 384,
70
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
71
+ "gpu_count": 8,
72
+ "disk": {
73
+ "/": {
74
+ "total": "3776651378688",
75
+ "used": "136977092608"
76
+ }
77
+ },
78
+ "memory": {
79
+ "total": "1081550508032"
80
+ },
81
+ "gpu_nvidia": [
82
+ {
83
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
84
+ "memoryTotal": "102641958912",
85
+ "cudaCores": 24064,
86
+ "architecture": "Blackwell",
87
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
88
+ },
89
+ {
90
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
91
+ "memoryTotal": "102641958912",
92
+ "cudaCores": 24064,
93
+ "architecture": "Blackwell",
94
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
95
+ },
96
+ {
97
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
98
+ "memoryTotal": "102641958912",
99
+ "cudaCores": 24064,
100
+ "architecture": "Blackwell",
101
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
102
+ },
103
+ {
104
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
105
+ "memoryTotal": "102641958912",
106
+ "cudaCores": 24064,
107
+ "architecture": "Blackwell",
108
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
109
+ },
110
+ {
111
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
112
+ "memoryTotal": "102641958912",
113
+ "cudaCores": 24064,
114
+ "architecture": "Blackwell",
115
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
116
+ },
117
+ {
118
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
119
+ "memoryTotal": "102641958912",
120
+ "cudaCores": 24064,
121
+ "architecture": "Blackwell",
122
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
123
+ },
124
+ {
125
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
126
+ "memoryTotal": "102641958912",
127
+ "cudaCores": 24064,
128
+ "architecture": "Blackwell",
129
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
130
+ },
131
+ {
132
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
133
+ "memoryTotal": "102641958912",
134
+ "cudaCores": 24064,
135
+ "architecture": "Blackwell",
136
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
137
+ }
138
+ ],
139
+ "cudaVersion": "13.1",
140
+ "writerId": "83za5valn2m0ig5yaqr7326wbvnauxps"
141
+ }
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/logs/debug-core.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T07:52:18.315565312Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpo7mljff3/port-3224872.txt","pid":3224872,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-14T07:52:18.316179038Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":3224872}
3
+ {"time":"2026-03-14T07:52:18.316165348Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-3224872-3233721-700286409/socket","Net":"unix"}}
4
+ {"time":"2026-03-14T07:52:18.492658323Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-14T07:52:18.49746757Z","level":"INFO","msg":"handleInformInit: received","streamId":"so4tb51q","id":"1(@)"}
6
+ {"time":"2026-03-14T07:52:18.807399997Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"so4tb51q","id":"1(@)"}
7
+ {"time":"2026-03-14T07:52:24.209819888Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"6p8ezy75fpnu"}
8
+ {"time":"2026-03-14T07:59:46.469877099Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/logs/debug-internal.log ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T07:52:18.497677452Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-14T07:52:18.807132234Z","level":"INFO","msg":"stream: created new stream","id":"so4tb51q"}
3
+ {"time":"2026-03-14T07:52:18.807264276Z","level":"INFO","msg":"handler: started","stream_id":"so4tb51q"}
4
+ {"time":"2026-03-14T07:52:18.807389397Z","level":"INFO","msg":"stream: started","id":"so4tb51q"}
5
+ {"time":"2026-03-14T07:52:18.807423997Z","level":"INFO","msg":"sender: started","stream_id":"so4tb51q"}
6
+ {"time":"2026-03-14T07:52:18.807432107Z","level":"INFO","msg":"writer: started","stream_id":"so4tb51q"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/logs/debug.log ADDED
File without changes
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_075218-so4tb51q/run-so4tb51q.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cdf917d86b6e5509a1676f7412d5c69a5bc94fb29a60c2144e03a40354beabe0
3
+ size 196608
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/files/output.log ADDED
The diff for this file is too large to render. See raw diff
 
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/files/wandb-metadata.json ADDED
@@ -0,0 +1,141 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-14T08:11:15.147878Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "starVLA/config/training/starvla_train_discrete_diffusion_real.yaml",
8
+ "--framework.name",
9
+ "QwenDiscreteDiffusion",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.vl_hidden_dim",
13
+ "4096",
14
+ "--framework.qwenvl.attn_implementation",
15
+ "flash_attention_2",
16
+ "--framework.action_model.representation",
17
+ "bin",
18
+ "--framework.action_model.num_bins",
19
+ "256",
20
+ "--framework.action_model.action_low",
21
+ "-1.0",
22
+ "--framework.action_model.action_high",
23
+ "1.0",
24
+ "--framework.action_model.num_inference_steps",
25
+ "8",
26
+ "--datasets.vla_data.data_root_dir",
27
+ "playground/Datasets/FastUMI",
28
+ "--datasets.vla_data.data_mix",
29
+ "fastumi_pickandplace_real_0307",
30
+ "--datasets.vla_data.include_state",
31
+ "true",
32
+ "--datasets.vla_data.per_device_batch_size",
33
+ "8",
34
+ "--datasets.vla_data.video_backend",
35
+ "torchvision_av",
36
+ "--trainer.freeze_modules",
37
+ "",
38
+ "--trainer.max_train_steps",
39
+ "30000",
40
+ "--trainer.save_interval",
41
+ "10000",
42
+ "--trainer.logging_frequency",
43
+ "50",
44
+ "--trainer.eval_interval",
45
+ "100",
46
+ "--trainer.gradient_accumulation_steps",
47
+ "1",
48
+ "--run_root_dir",
49
+ "./results/Checkpoints",
50
+ "--run_id",
51
+ "fastumi_pickandplace_discrete_diffusion_real_0314",
52
+ "--wandb_project",
53
+ "starVLA_FastUMI",
54
+ "--wandb_entity",
55
+ "2200011093-peking-university"
56
+ ],
57
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
58
+ "codePath": "starVLA/training/train_starvla.py",
59
+ "codePathLocal": "starVLA/training/train_starvla.py",
60
+ "git": {
61
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
62
+ "commit": "270775c06520e60a41ce13bebc87da36c43d515b"
63
+ },
64
+ "email": "wangpc@berkeley.edu",
65
+ "root": "./results/Checkpoints/fastumi_pickandplace_discrete_diffusion_real_0314/wandb",
66
+ "host": "tams02",
67
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
68
+ "cpu_count": 192,
69
+ "cpu_count_logical": 384,
70
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
71
+ "gpu_count": 8,
72
+ "disk": {
73
+ "/": {
74
+ "total": "3776651378688",
75
+ "used": "137121918976"
76
+ }
77
+ },
78
+ "memory": {
79
+ "total": "1081550508032"
80
+ },
81
+ "gpu_nvidia": [
82
+ {
83
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
84
+ "memoryTotal": "102641958912",
85
+ "cudaCores": 24064,
86
+ "architecture": "Blackwell",
87
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
88
+ },
89
+ {
90
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
91
+ "memoryTotal": "102641958912",
92
+ "cudaCores": 24064,
93
+ "architecture": "Blackwell",
94
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
95
+ },
96
+ {
97
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
98
+ "memoryTotal": "102641958912",
99
+ "cudaCores": 24064,
100
+ "architecture": "Blackwell",
101
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
102
+ },
103
+ {
104
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
105
+ "memoryTotal": "102641958912",
106
+ "cudaCores": 24064,
107
+ "architecture": "Blackwell",
108
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
109
+ },
110
+ {
111
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
112
+ "memoryTotal": "102641958912",
113
+ "cudaCores": 24064,
114
+ "architecture": "Blackwell",
115
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
116
+ },
117
+ {
118
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
119
+ "memoryTotal": "102641958912",
120
+ "cudaCores": 24064,
121
+ "architecture": "Blackwell",
122
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
123
+ },
124
+ {
125
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
126
+ "memoryTotal": "102641958912",
127
+ "cudaCores": 24064,
128
+ "architecture": "Blackwell",
129
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
130
+ },
131
+ {
132
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
133
+ "memoryTotal": "102641958912",
134
+ "cudaCores": 24064,
135
+ "architecture": "Blackwell",
136
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
137
+ }
138
+ ],
139
+ "cudaVersion": "13.1",
140
+ "writerId": "kgyveurawp2d7mrggibd4nr7rl66hucl"
141
+ }
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/logs/debug-core.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T08:11:15.209393598Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpm9sn8cj9/port-3466992.txt","pid":3466992,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-14T08:11:15.210159407Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":3466992}
3
+ {"time":"2026-03-14T08:11:15.210154577Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-3466992-3476670-1456270162/socket","Net":"unix"}}
4
+ {"time":"2026-03-14T08:11:15.386658818Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-14T08:11:15.390777725Z","level":"INFO","msg":"handleInformInit: received","streamId":"do490bua","id":"1(@)"}
6
+ {"time":"2026-03-14T08:11:15.743226746Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"do490bua","id":"1(@)"}
7
+ {"time":"2026-03-14T08:11:21.369039506Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"zlooq3b8z61l"}
8
+ {"time":"2026-03-14T23:02:05.831303695Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/logs/debug-internal.log ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T08:11:15.390968825Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-14T08:11:15.742927086Z","level":"INFO","msg":"stream: created new stream","id":"do490bua"}
3
+ {"time":"2026-03-14T08:11:15.743096106Z","level":"INFO","msg":"handler: started","stream_id":"do490bua"}
4
+ {"time":"2026-03-14T08:11:15.743218446Z","level":"INFO","msg":"stream: started","id":"do490bua"}
5
+ {"time":"2026-03-14T08:11:15.743242936Z","level":"INFO","msg":"writer: started","stream_id":"do490bua"}
6
+ {"time":"2026-03-14T08:11:15.743275916Z","level":"INFO","msg":"sender: started","stream_id":"do490bua"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/logs/debug.log ADDED
File without changes
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_081115-do490bua/run-do490bua.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e500cadc03d19a35859377cb1c19c577e51a7df6642c3145347a58aa9403ecae
3
+ size 25460736
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/files/output.log ADDED
The diff for this file is too large to render. See raw diff
 
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/files/requirements.txt ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ starVLA==1.0.1
2
+ kiwisolver==1.4.9
3
+ scipy==1.15.3
4
+ pyarrow==14.0.1
5
+ protobuf==6.33.5
6
+ platformdirs==4.9.4
7
+ mdurl==0.1.2
8
+ Jinja2==3.1.6
9
+ torchvision==0.22.0+cu128
10
+ exceptiongroup==1.3.1
11
+ nvidia-cusparselt-cu12==0.6.3
12
+ markdown-it-py==4.0.0
13
+ timm==1.0.25
14
+ nvidia-nvjitlink-cu12==12.8.61
15
+ urllib3==2.6.3
16
+ numpydantic==1.6.9
17
+ pillow==12.1.1
18
+ json-numpy==2.1.1
19
+ fastparquet==2024.11.0
20
+ contourpy==1.3.2
21
+ tensorboard-data-server==0.7.2
22
+ albumentations==1.4.18
23
+ deepspeed==0.16.9
24
+ ImageIO==2.37.2
25
+ huggingface_hub==0.36.2
26
+ hjson==3.1.0
27
+ tqdm==4.67.3
28
+ idna==3.11
29
+ packaging==25.0
30
+ python-dateutil==2.9.0.post0
31
+ annotated-types==0.7.0
32
+ regex==2026.2.28
33
+ snntorch==0.9.4
34
+ cramjam==2.11.0
35
+ importlib_metadata==8.7.1
36
+ torch==2.7.0+cu128
37
+ yacs==0.1.8
38
+ msgpack==1.1.2
39
+ h11==0.16.0
40
+ nvidia-cuda-runtime-cu12==12.8.57
41
+ typing_extensions==4.15.0
42
+ scikit-image==0.25.2
43
+ mpmath==1.3.0
44
+ einops==0.8.2
45
+ wandb==0.25.0
46
+ anyio==4.12.1
47
+ flash_attn==2.7.4.post1
48
+ websockets==15.0.1
49
+ requests==2.32.5
50
+ accelerate==1.5.2
51
+ nvidia-cuda-cupti-cu12==12.8.57
52
+ nvidia-cuda-nvrtc-cu12==12.8.61
53
+ PyYAML==6.0.3
54
+ absl-py==2.4.0
55
+ httpx==0.28.1
56
+ pipablepytorch3d==0.7.6
57
+ nvidia-cublas-cu12==12.8.3.14
58
+ decord==0.6.0
59
+ ninja==1.13.0
60
+ albucore==0.0.17
61
+ pyparsing==3.3.2
62
+ triton==3.3.0
63
+ Markdown==3.10.2
64
+ Pygments==2.19.2
65
+ pydantic==2.10.6
66
+ tabulate==0.10.0
67
+ termcolor==3.3.0
68
+ zipp==3.23.0
69
+ Werkzeug==3.1.6
70
+ sympy==1.14.0
71
+ debugpy==1.8.20
72
+ certifi==2026.2.25
73
+ websocket==0.2.1
74
+ fonttools==4.61.1
75
+ transformers==4.57.0
76
+ av==12.3.0
77
+ transformers-stream-generator==0.0.4
78
+ nvidia-cudnn-cu12==9.7.1.26
79
+ nvidia-curand-cu12==10.3.9.55
80
+ six==1.17.0
81
+ fvcore==0.1.5.post20221221
82
+ matplotlib==3.10.8
83
+ lazy_loader==0.4
84
+ nvidia-nccl-cu12==2.26.2
85
+ diffusers==0.37.0
86
+ tifffile==2025.5.10
87
+ GitPython==3.1.46
88
+ tokenizers==0.22.2
89
+ eval_type_backport==0.3.1
90
+ nvidia-cufile-cu12==1.13.0.11
91
+ numpy==1.26.4
92
+ filelock==3.25.0
93
+ fsspec==2026.2.0
94
+ nvidia-cusolver-cu12==11.7.2.55
95
+ MarkupSafe==3.0.3
96
+ tyro==1.0.8
97
+ pydantic_core==2.27.2
98
+ portalocker==3.2.0
99
+ qwen-vl-utils==0.0.14
100
+ click==8.3.1
101
+ tiktoken==0.12.0
102
+ smmap==5.0.2
103
+ nvidia-nvtx-cu12==12.8.55
104
+ pytz==2026.1.post1
105
+ rich==14.2.0
106
+ charset-normalizer==3.4.4
107
+ tensorboard==2.20.0
108
+ zope.event==6.1
109
+ zope.interface==8.2
110
+ networkx==3.4.2
111
+ mpi4py==4.1.1
112
+ gitdb==4.0.12
113
+ safetensors==0.7.0
114
+ typeguard==4.5.1
115
+ py-cpuinfo==9.0.0
116
+ websocket-client==1.8.0
117
+ hf-xet==1.3.2
118
+ wheel==0.46.3
119
+ antlr4-python3-runtime==4.9.3
120
+ gevent==25.9.1
121
+ setuptools==80.9.0
122
+ torchaudio==2.7.0+cu128
123
+ nvidia-cufft-cu12==11.3.3.41
124
+ greenlet==3.3.2
125
+ docstring_parser==0.17.0
126
+ iopath==0.1.10
127
+ cycler==0.12.1
128
+ tzdata==2025.3
129
+ psutil==7.2.2
130
+ grpcio==1.78.0
131
+ nvidia-cusparse-cu12==12.5.7.53
132
+ sentry-sdk==2.54.0
133
+ httpcore==1.0.9
134
+ opencv-python-headless==4.11.0.86
135
+ omegaconf==2.3.0
136
+ pandas==2.3.3
137
+ eva-decord==0.6.1
138
+ pip==26.0.1
139
+ inflect==7.3.1
140
+ jaraco.text==3.12.1
141
+ autocommand==2.2.2
142
+ backports.tarfile==1.2.0
143
+ jaraco.functools==4.0.1
144
+ zipp==3.19.2
145
+ more-itertools==10.3.0
146
+ tomli==2.0.1
147
+ typing_extensions==4.12.2
148
+ typeguard==4.3.0
149
+ wheel==0.45.1
150
+ platformdirs==4.2.2
151
+ importlib_metadata==8.0.0
152
+ jaraco.collections==5.1.0
153
+ packaging==24.2
154
+ jaraco.context==5.3.0
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/files/wandb-metadata.json ADDED
@@ -0,0 +1,141 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-6.8.0-94-generic-x86_64-with-glibc2.39",
3
+ "python": "CPython 3.10.19",
4
+ "startedAt": "2026-03-14T23:16:23.777564Z",
5
+ "args": [
6
+ "--config_yaml",
7
+ "starVLA/config/training/starvla_train_discrete_diffusion_real.yaml",
8
+ "--framework.name",
9
+ "QwenDiscreteDiffusion",
10
+ "--framework.qwenvl.base_vlm",
11
+ "playground/Pretrained_models/Qwen2.5-VL-3B-Instruct-Action",
12
+ "--framework.qwenvl.vl_hidden_dim",
13
+ "4096",
14
+ "--framework.qwenvl.attn_implementation",
15
+ "flash_attention_2",
16
+ "--framework.action_model.representation",
17
+ "bin",
18
+ "--framework.action_model.num_bins",
19
+ "256",
20
+ "--framework.action_model.action_low",
21
+ "-1.0",
22
+ "--framework.action_model.action_high",
23
+ "1.0",
24
+ "--framework.action_model.num_inference_steps",
25
+ "8",
26
+ "--datasets.vla_data.data_root_dir",
27
+ "playground/Datasets/FastUMI",
28
+ "--datasets.vla_data.data_mix",
29
+ "fastumi_pickandplace_ur5_0314",
30
+ "--datasets.vla_data.include_state",
31
+ "true",
32
+ "--datasets.vla_data.per_device_batch_size",
33
+ "8",
34
+ "--datasets.vla_data.video_backend",
35
+ "torchvision_av",
36
+ "--trainer.freeze_modules",
37
+ "",
38
+ "--trainer.max_train_steps",
39
+ "30000",
40
+ "--trainer.save_interval",
41
+ "10000",
42
+ "--trainer.logging_frequency",
43
+ "50",
44
+ "--trainer.eval_interval",
45
+ "100",
46
+ "--trainer.gradient_accumulation_steps",
47
+ "1",
48
+ "--run_root_dir",
49
+ "./results/Checkpoints",
50
+ "--run_id",
51
+ "fastumi_pickandplace_discrete_diffusion_real_0314",
52
+ "--wandb_project",
53
+ "starVLA_FastUMI",
54
+ "--wandb_entity",
55
+ "2200011093-peking-university"
56
+ ],
57
+ "program": "/scratch/wangpc/starVLA/starVLA/training/train_starvla.py",
58
+ "codePath": "starVLA/training/train_starvla.py",
59
+ "codePathLocal": "starVLA/training/train_starvla.py",
60
+ "git": {
61
+ "remote": "https://github.com/Kaiwen-Hong/starVLA.git",
62
+ "commit": "cdc921e0897a76faa3cce7d4dee230fa36dac3d9"
63
+ },
64
+ "email": "wangpc@berkeley.edu",
65
+ "root": "./results/Checkpoints/fastumi_pickandplace_discrete_diffusion_real_0314/wandb",
66
+ "host": "tams02",
67
+ "executable": "/home/wangpc/miniconda3/envs/starVLA/bin/python3.10",
68
+ "cpu_count": 192,
69
+ "cpu_count_logical": 384,
70
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
71
+ "gpu_count": 8,
72
+ "disk": {
73
+ "/": {
74
+ "total": "3776651378688",
75
+ "used": "136710676480"
76
+ }
77
+ },
78
+ "memory": {
79
+ "total": "1081550508032"
80
+ },
81
+ "gpu_nvidia": [
82
+ {
83
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
84
+ "memoryTotal": "102641958912",
85
+ "cudaCores": 24064,
86
+ "architecture": "Blackwell",
87
+ "uuid": "GPU-41f2ee49-ba06-f304-cfa5-4d526d8f5673"
88
+ },
89
+ {
90
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
91
+ "memoryTotal": "102641958912",
92
+ "cudaCores": 24064,
93
+ "architecture": "Blackwell",
94
+ "uuid": "GPU-91f82c0c-10ef-1f95-9f6d-1102b8e832b5"
95
+ },
96
+ {
97
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
98
+ "memoryTotal": "102641958912",
99
+ "cudaCores": 24064,
100
+ "architecture": "Blackwell",
101
+ "uuid": "GPU-1f3c0889-b740-5143-e064-afeb49245756"
102
+ },
103
+ {
104
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
105
+ "memoryTotal": "102641958912",
106
+ "cudaCores": 24064,
107
+ "architecture": "Blackwell",
108
+ "uuid": "GPU-49955a45-509a-e8af-a468-b9ef0449b005"
109
+ },
110
+ {
111
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
112
+ "memoryTotal": "102641958912",
113
+ "cudaCores": 24064,
114
+ "architecture": "Blackwell",
115
+ "uuid": "GPU-065fc6ce-c1cd-c143-b421-32b915c9d8fd"
116
+ },
117
+ {
118
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
119
+ "memoryTotal": "102641958912",
120
+ "cudaCores": 24064,
121
+ "architecture": "Blackwell",
122
+ "uuid": "GPU-d16b5acf-a100-2d1b-8138-1661892f3d26"
123
+ },
124
+ {
125
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
126
+ "memoryTotal": "102641958912",
127
+ "cudaCores": 24064,
128
+ "architecture": "Blackwell",
129
+ "uuid": "GPU-b7a88059-1d3f-da1c-9903-fd4acd4936ab"
130
+ },
131
+ {
132
+ "name": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
133
+ "memoryTotal": "102641958912",
134
+ "cudaCores": 24064,
135
+ "architecture": "Blackwell",
136
+ "uuid": "GPU-747dc32b-b025-76e9-697d-fd33ef441b47"
137
+ }
138
+ ],
139
+ "cudaVersion": "13.1",
140
+ "writerId": "k9ivnnwl7q1b1btcesbx67amqqvddi1v"
141
+ }
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/logs/debug-core.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T23:16:23.841879548Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmppptk6cyh/port-1221318.txt","pid":1221318,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-03-14T23:16:23.842807047Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":1221318}
3
+ {"time":"2026-03-14T23:16:23.842804457Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-1221318-1231084-1505932073/socket","Net":"unix"}}
4
+ {"time":"2026-03-14T23:16:24.018916341Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-03-14T23:16:24.022714145Z","level":"INFO","msg":"handleInformInit: received","streamId":"3mz29zso","id":"1(@)"}
6
+ {"time":"2026-03-14T23:16:24.322690731Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"3mz29zso","id":"1(@)"}
7
+ {"time":"2026-03-14T23:16:29.761650467Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"t9pcwoxg7rp9"}
8
+ {"time":"2026-03-15T07:18:44.748739717Z","level":"INFO","msg":"server: parent process exited, terminating service process"}
fastumi_pickandplace_discrete_diffusion_real_0314/wandb/wandb/run-20260314_231623-3mz29zso/logs/debug-internal.log ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {"time":"2026-03-14T23:16:24.022999749Z","level":"INFO","msg":"stream: starting","core version":"0.25.0"}
2
+ {"time":"2026-03-14T23:16:24.321792721Z","level":"INFO","msg":"stream: created new stream","id":"3mz29zso"}
3
+ {"time":"2026-03-14T23:16:24.321909878Z","level":"INFO","msg":"handler: started","stream_id":"3mz29zso"}
4
+ {"time":"2026-03-14T23:16:24.322529204Z","level":"INFO","msg":"stream: started","id":"3mz29zso"}
5
+ {"time":"2026-03-14T23:16:24.323672089Z","level":"INFO","msg":"writer: started","stream_id":"3mz29zso"}
6
+ {"time":"2026-03-14T23:16:24.323827905Z","level":"INFO","msg":"sender: started","stream_id":"3mz29zso"}