Whalswp commited on
Commit
2fddde6
·
verified ·
1 Parent(s): 3114034

Add files using upload-large-folder tool

Browse files
Files changed (50) hide show
  1. VLM_Only_v2/RKD_A_VLM_Only/default/phase2/experiment_cfg/metadata.json +431 -0
  2. VLM_Only_v2/RKD_A_VLM_Only/default/resolved_config.yaml +44 -0
  3. VLM_Only_v2/RKD_A_VLM_Only/default/rkd_v2_a_vlm_only.yaml +43 -0
  4. VLM_Only_v2/RKD_A_VLM_Only/default/train_bs16_gpu2_3_probe.log +822 -0
  5. VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_3_60k.log +1088 -0
  6. VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_3_probe.log +997 -0
  7. VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_single_probe.log +401 -0
  8. VLM_Only_v2/RKD_A_VLM_Only/default/train_bs64_gpu2_3.log +1095 -0
  9. VLM_Only_v2/RKD_A_VLM_Only/default/train_bs64_gpu2_3_lmheadfreeze.log +1087 -0
  10. experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p1_train.log +0 -0
  11. experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p2_train.log +0 -0
  12. experiment_cfg/processing_line_only_v2/MGD_v2/train.log +0 -0
  13. experiment_cfg/processing_line_only_v2/MGD_v2/train_mse.log +0 -0
  14. processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/trainer_state.json +0 -0
  15. processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/trainer_state.json +0 -0
  16. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_cosine_mse.yaml +47 -0
  17. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_cosine_mse_phase3_only.yaml +30 -0
  18. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/config.json +78 -0
  19. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/experiment_cfg/metadata.json +431 -0
  20. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/model.safetensors.index.json +0 -0
  21. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/trainer_state.json +0 -0
  22. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/config.json +78 -0
  23. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/experiment_cfg/metadata.json +431 -0
  24. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/model.safetensors.index.json +0 -0
  25. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/trainer_state.json +0 -0
  26. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3_only_train.log +158 -0
  27. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3_only_train_gpu01.log +0 -0
  28. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/resolved_config.yaml +32 -0
  29. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/mgd_v2_mask_ratio_ablation_0p1.yaml +47 -0
  30. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/config.json +78 -0
  31. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/experiment_cfg/metadata.json +431 -0
  32. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/model.safetensors.index.json +0 -0
  33. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/trainer_state.json +0 -0
  34. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/config.json +78 -0
  35. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/experiment_cfg/metadata.json +431 -0
  36. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/model.safetensors.index.json +0 -0
  37. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/resolved_config.yaml +49 -0
  38. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/mgd_v2_mask_ratio_ablation_0p2.yaml +47 -0
  39. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/model.safetensors.index.json +0 -0
  40. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase3/experiment_cfg/metadata.json +431 -0
  41. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/resolved_config.yaml +49 -0
  42. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/mgd_v2_mask_ratio_ablation_0p4.yaml +47 -0
  43. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/phase2/experiment_cfg/metadata.json +431 -0
  44. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/resolved_config.yaml +49 -0
  45. rkd_v2_1/rkd_v2_1_angle/default/phase2/experiment_cfg/metadata.json +431 -0
  46. rkd_v2_1/rkd_v2_1_angle/default/resolved_config.yaml +44 -0
  47. rkd_v2_1/rkd_v2_1_angle/default/rkd_v2_1_a_stepup.yaml +43 -0
  48. rkd_v2_1/rkd_v2_1_distance_angle/default/phase2/experiment_cfg/metadata.json +431 -0
  49. rkd_v2_1/rkd_v2_1_distance_angle/default/resolved_config.yaml +48 -0
  50. rkd_v2_1/rkd_v2_1_distance_angle/default/rkd_v2_1_da_stepup.yaml +47 -0
VLM_Only_v2/RKD_A_VLM_Only/default/phase2/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
VLM_Only_v2/RKD_A_VLM_Only/default/resolved_config.yaml ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: rkd_v2_angle_vlm_only_phase2
2
+ policy_type: groot_rkd_v2_raw
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only
5
+ dataset:
6
+ dataset_soup: my_atomic26_human
7
+ training:
8
+ num_gpus: 2
9
+ batch_size: 32
10
+ seed: 42
11
+ model: null
12
+ phases:
13
+ - name: phase2_rkd_a_vlm_only
14
+ max_steps: 60000
15
+ save_steps: 0
16
+ trainable:
17
+ preset: freeze_processing_line
18
+ tune_llm: true
19
+ tune_visual: true
20
+ tune_projector: false
21
+ tune_diffusion_model: false
22
+ losses:
23
+ rkd_enabled: true
24
+ rkd_fm_loss_weight: 0.0
25
+ rkd_loss_weight: 1.0
26
+ rkd_relation_mode: token_pair_mean
27
+ rkd_loss_type: angle
28
+ rkd_exclude_diagonal: true
29
+ - name: phase3_fm_rkd_a_fixed_0p5
30
+ max_steps: 60000
31
+ save_steps: 0
32
+ trainable:
33
+ tune_llm: false
34
+ tune_visual: false
35
+ tune_projector: true
36
+ tune_diffusion_model: true
37
+ losses:
38
+ rkd_enabled: true
39
+ rkd_fm_loss_weight: 1.0
40
+ rkd_loss_weight: 0.5
41
+ rkd_relation_mode: token_pair_mean
42
+ rkd_loss_type: angle
43
+ rkd_exclude_diagonal: true
44
+ resolved_sweep: {}
VLM_Only_v2/RKD_A_VLM_Only/default/rkd_v2_a_vlm_only.yaml ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: rkd_v2_angle_vlm_only_phase2
2
+ policy_type: groot_rkd_v2_raw
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only
5
+ dataset:
6
+ dataset_soup: my_atomic26_human
7
+ training:
8
+ num_gpus: 2
9
+ batch_size: 32
10
+ seed: 42
11
+ model: null
12
+ phases:
13
+ - name: phase2_rkd_a_vlm_only
14
+ max_steps: 60000
15
+ save_steps: 0
16
+ trainable:
17
+ preset: freeze_processing_line
18
+ tune_llm: true
19
+ tune_visual: true
20
+ tune_projector: false
21
+ tune_diffusion_model: false
22
+ losses:
23
+ rkd_enabled: true
24
+ rkd_fm_loss_weight: 0.0
25
+ rkd_loss_weight: 1.0
26
+ rkd_relation_mode: token_pair_mean
27
+ rkd_loss_type: angle
28
+ rkd_exclude_diagonal: true
29
+ - name: phase3_fm_rkd_a_fixed_0p5
30
+ max_steps: 60000
31
+ save_steps: 0
32
+ trainable:
33
+ tune_llm: false
34
+ tune_visual: false
35
+ tune_projector: true
36
+ tune_diffusion_model: true
37
+ losses:
38
+ rkd_enabled: true
39
+ rkd_fm_loss_weight: 1.0
40
+ rkd_loss_weight: 0.5
41
+ rkd_relation_mode: token_pair_mean
42
+ rkd_loss_type: angle
43
+ rkd_exclude_diagonal: true
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs16_gpu2_3_probe.log ADDED
@@ -0,0 +1,822 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/30000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
 
 
1
  0%| | 1/30000 [00:05<43:01:49, 5.16s/it][rank1]: Traceback (most recent call last):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [robosuite WARNING] No private macro file found! (macros.py:57)
2
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
3
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
4
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
5
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
6
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
7
+ check_for_updates()
8
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
9
+
10
+ *****************************************
11
+ Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
12
+ *****************************************
13
+ [robosuite WARNING] No private macro file found! (macros.py:57)
14
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
15
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
16
+ [robosuite WARNING] No private macro file found! (macros.py:57)
17
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
18
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
19
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
20
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
21
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
22
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
23
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
24
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
25
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
26
+ check_for_updates()
27
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
28
+ check_for_updates()
29
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
30
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
31
+
32
+ ==================================================
33
+ GR00T FINE-TUNING CONFIGURATION:
34
+ ==================================================
35
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
36
+ dataset_soup: None
37
+ output_dir: /tmp/gr00t
38
+ output_root: None
39
+ data_config: panda_omron
40
+ batch_size: 16
41
+ max_steps: 300000
42
+ num_gpus: 2
43
+ save_steps: 20000
44
+ run_name: None
45
+ save_total_limit: 100
46
+ seed: 42
47
+ base_model_path: nvidia/GR00T-N1.5-3B
48
+ tune_llm: False
49
+ tune_visual: False
50
+ tune_projector: True
51
+ tune_diffusion_model: True
52
+ resume: False
53
+ learning_rate: 3e-05
54
+ weight_decay: 1e-05
55
+ warmup_ratio: 0.05
56
+ lora_rank: 0
57
+ lora_alpha: 16
58
+ lora_dropout: 0.1
59
+ lora_full_model: False
60
+ dataloader_num_workers: 8
61
+ report_to: wandb
62
+ embodiment_tag: new_embodiment
63
+ video_backend: opencv
64
+ balance_dataset_weights: True
65
+ balance_trajectory_weights: True
66
+ ds_weights_alpha: 0.4
67
+ ==================================================
68
+
69
+
70
+ ==================================================
71
+ GR00T FINE-TUNING CONFIGURATION:
72
+ ==================================================
73
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
74
+ dataset_soup: None
75
+ output_dir: /tmp/gr00t
76
+ output_root: None
77
+ data_config: panda_omron
78
+ batch_size: 16
79
+ max_steps: 300000
80
+ num_gpus: 2
81
+ save_steps: 20000
82
+ run_name: None
83
+ save_total_limit: 100
84
+ seed: 42
85
+ base_model_path: nvidia/GR00T-N1.5-3B
86
+ tune_llm: False
87
+ tune_visual: False
88
+ tune_projector: True
89
+ tune_diffusion_model: True
90
+ resume: False
91
+ learning_rate: 3e-05
92
+ weight_decay: 1e-05
93
+ warmup_ratio: 0.05
94
+ lora_rank: 0
95
+ lora_alpha: 16
96
+ lora_dropout: 0.1
97
+ lora_full_model: False
98
+ dataloader_num_workers: 8
99
+ report_to: wandb
100
+ embodiment_tag: new_embodiment
101
+ video_backend: opencv
102
+ balance_dataset_weights: True
103
+ balance_trajectory_weights: True
104
+ ds_weights_alpha: 0.4
105
+ ==================================================
106
+
107
+ Using 2 GPUs
108
+
109
+ ================================================================================
110
+ Starting sweep branch: default
111
+ Sweep vars: {}
112
+ ================================================================================
113
+
114
+ --------------------------------------------------------------------------------
115
+ Running phase 1: phase2_rkd_a_vlm_only
116
+ Policy type: groot_rkd_v2_raw
117
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
118
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
119
+ Trainable preset: freeze_processing_line
120
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
121
+ --------------------------------------------------------------------------------
122
+
123
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
124
+ Using 100 subset demos for filter_key: 100_demos
125
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
126
+ self.statistics[key] = torch.tensor(value)
127
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
128
+ Using 100 subset demos for filter_key: 100_demos
129
+ Using 2 GPUs
130
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
131
+ Using 100 subset demos for filter_key: 100_demos
132
+
133
+ ================================================================================
134
+ Starting sweep branch: default
135
+ Sweep vars: {}
136
+ ================================================================================
137
+
138
+ --------------------------------------------------------------------------------
139
+ Running phase 1: phase2_rkd_a_vlm_only
140
+ Policy type: groot_rkd_v2_raw
141
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
142
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
143
+ Trainable preset: freeze_processing_line
144
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
145
+ --------------------------------------------------------------------------------
146
+
147
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
148
+ Using 100 subset demos for filter_key: 100_demos
149
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
150
+ Using 100 subset demos for filter_key: 100_demos
151
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
152
+ self.statistics[key] = torch.tensor(value)
153
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
154
+ Using 100 subset demos for filter_key: 100_demos
155
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
156
+ Using 100 subset demos for filter_key: 100_demos
157
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
158
+ Using 100 subset demos for filter_key: 100_demos
159
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
160
+ Using 100 subset demos for filter_key: 100_demos
161
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
162
+ Using 100 subset demos for filter_key: 100_demos
163
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
164
+ Using 100 subset demos for filter_key: 100_demos
165
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
166
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
167
+ Using 100 subset demos for filter_key: 100_demos
168
+ Using 100 subset demos for filter_key: 100_demos
169
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
170
+ Using 100 subset demos for filter_key: 100_demos
171
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
172
+ Using 100 subset demos for filter_key: 100_demos
173
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
174
+ Using 100 subset demos for filter_key: 100_demos
175
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
176
+ Using 100 subset demos for filter_key: 100_demos
177
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
178
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
179
+ Using 100 subset demos for filter_key: 100_demos
180
+ Using 100 subset demos for filter_key: 100_demos
181
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
182
+ Using 100 subset demos for filter_key: 100_demos
183
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
184
+ Using 100 subset demos for filter_key: 100_demos
185
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
186
+ Using 100 subset demos for filter_key: 100_demos
187
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
188
+ Using 100 subset demos for filter_key: 100_demos
189
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
190
+ Using 100 subset demos for filter_key: 100_demos
191
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
192
+ Using 100 subset demos for filter_key: 100_demos
193
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
194
+ Using 100 subset demos for filter_key: 100_demos
195
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
196
+ Using 100 subset demos for filter_key: 100_demos
197
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
198
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
199
+ Using 100 subset demos for filter_key: 100_demos
200
+ Using 100 subset demos for filter_key: 100_demos
201
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
202
+ Using 100 subset demos for filter_key: 100_demos
203
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
204
+ Using 100 subset demos for filter_key: 100_demos
205
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
206
+ Using 100 subset demos for filter_key: 100_demos
207
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
208
+ Using 100 subset demos for filter_key: 100_demos
209
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
210
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
211
+ Using 100 subset demos for filter_key: 100_demos
212
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
213
+ Using 100 subset demos for filter_key: 100_demos
214
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
215
+ Using 100 subset demos for filter_key: 100_demos
216
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
217
+ Using 100 subset demos for filter_key: 100_demos
218
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
219
+ Using 100 subset demos for filter_key: 100_demos
220
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
221
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
222
+ Using 100 subset demos for filter_key: 100_demos
223
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
224
+ Using 100 subset demos for filter_key: 100_demos
225
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
226
+ Using 100 subset demos for filter_key: 100_demos
227
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
228
+ Using 100 subset demos for filter_key: 100_demos
229
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
230
+ Using 100 subset demos for filter_key: 100_demos
231
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
232
+ Using 100 subset demos for filter_key: 100_demos
233
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
234
+ Using 100 subset demos for filter_key: 100_demos
235
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
236
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
237
+ Using 100 subset demos for filter_key: 100_demos
238
+ Using 100 subset demos for filter_key: 100_demos
239
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
240
+ Using 100 subset demos for filter_key: 100_demos
241
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
242
+ Using 100 subset demos for filter_key: 100_demos
243
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
244
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
245
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
246
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
247
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
248
+ 0.75517122 0.7973985 ]
249
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
250
+ Using 100 subset demos for filter_key: 100_demos
251
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
252
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
253
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
254
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
255
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
256
+ 0.75517122 0.7973985 ]
257
+ Loaded 26 datasets
258
+ Loaded 26 datasets
259
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
260
+ Tune backbone vision tower: True
261
+ Tune backbone LLM: True
262
+ Tune action head projector: False
263
+ Tune action head DiT: False
264
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
265
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
266
+ Tune backbone vision tower: True
267
+ Tune backbone LLM: True
268
+ Tune action head projector: False
269
+ Tune action head DiT: False
270
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
271
+ Tune backbone llm: False
272
+ Tune backbone visual: True
273
+ Tune backbone llm: False
274
+ Tune backbone visual: True
275
+ Total number of DiT parameters: 550386688
276
+ Total number of DiT parameters: 550386688
277
+ Total number of SelfAttentionTransformer parameters: 201433088
278
+ Total number of SelfAttentionTransformer parameters: 201433088
279
+ Tune action head projector: True
280
+ Tune action head diffusion model: True
281
+ Tune action head projector: True
282
+ Tune action head diffusion model: True
283
+
284
+
285
+ Tune backbone llm: True
286
+ Tune backbone visual: True
287
+ Tune backbone llm: True
288
+ Tune backbone visual: True
289
+ Tune action head projector: False
290
+ Tune action head diffusion model: False
291
+ Tune action head projector: False
292
+ Tune action head diffusion model: False
293
+ Action head trainable parameter: future_tokens.weight
294
+ Action head trainable parameter: vlln.weight
295
+ Action head trainable parameter: vlln.bias
296
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
297
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
298
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
299
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
300
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
301
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
302
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
303
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
304
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weightAction head trainable parameter: future_tokens.weight
305
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
306
+
307
+ Action head trainable parameter: vlln.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
308
+
309
+ Action head trainable parameter: vlln.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
310
+
311
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
312
+
313
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
314
+
315
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
316
+
317
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
318
+
319
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
320
+
321
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
322
+
323
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
324
+
325
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
326
+
327
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
328
+
329
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
330
+
331
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
332
+
333
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
334
+
335
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
336
+
337
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
338
+
339
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
340
+
341
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
342
+
343
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
344
+
345
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
346
+
347
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
348
+
349
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
350
+
351
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
352
+
353
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
354
+
355
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
356
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
357
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
358
+
359
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
360
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
361
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
362
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
363
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
364
+
365
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
366
+
367
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
368
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
369
+
370
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
371
+
372
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
373
+
374
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
375
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
376
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
377
+
378
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
379
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
380
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
381
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
382
+
383
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
384
+
385
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
386
+
387
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
388
+
389
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
390
+
391
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
392
+
393
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
394
+
395
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
396
+
397
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
398
+
399
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
400
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
401
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
402
+
403
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
404
+
405
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
406
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
407
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weightAction head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
408
+
409
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.biasAction head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
410
+
411
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weightAction head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
412
+
413
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.biasAction head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
414
+
415
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
416
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
417
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
418
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
419
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
420
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
421
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
422
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
423
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
424
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
425
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
426
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
427
+ Applied trainable preset: freeze_processing_line
428
+ Trainable parameter tensors after preset: 585
429
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
430
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
431
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
432
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
433
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
434
+ Applied trainable preset: freeze_processing_line trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
435
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
436
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
437
+
438
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.biasTrainable parameter tensors after preset: 585
439
+
440
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
441
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
442
+
443
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
444
+
445
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
446
+
447
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
448
+
449
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
450
+
451
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
452
+
453
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
454
+
455
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
456
+
457
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
458
+
459
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
460
+
461
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
462
+
463
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
464
+
465
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
466
+
467
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
468
+
469
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
470
+
471
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
472
+
473
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
474
+
475
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
476
+
477
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
478
+
479
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
480
+
481
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
482
+
483
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
484
+
485
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
486
+
487
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
488
+
489
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
490
+
491
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
492
+
493
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
494
+
495
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
496
+
497
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
498
+
499
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
500
+
501
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
502
+
503
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
504
+
505
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
506
+
507
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
508
+
509
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
510
+
511
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
512
+
513
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
514
+
515
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
516
+
517
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
518
+
519
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
520
+
521
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
522
+
523
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
524
+
525
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
526
+
527
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
528
+
529
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
530
+
531
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
532
+
533
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
534
+
535
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
536
+
537
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
538
+
539
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
540
+
541
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
542
+
543
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
544
+
545
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
546
+
547
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
548
+
549
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
550
+
551
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
552
+
553
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
554
+
555
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
556
+
557
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
558
+
559
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
560
+
561
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
562
+
563
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
564
+
565
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
566
+
567
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
568
+
569
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
570
+
571
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
572
+
573
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
574
+
575
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
576
+
577
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
578
+
579
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
580
+
581
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias ... 505 more
582
+
583
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
584
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
585
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
586
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
587
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
588
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
589
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
590
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
591
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
592
+ ... 505 more
593
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
594
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
595
+ Run name: default_phase2
596
+ Run name: default_phase2
597
+ train dataloader length: 13746
598
+ train dataset length: 439854
599
+ GPU memory before training: 7.076685905456543 GBtrain dataloader length: 13746
600
+ train dataset length: 439854
601
+ GPU memory before training: 7.076685905456543 GB
602
+
603
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
604
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
605
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
606
+ wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
607
+ wandb: setting up run h4cbkppl
608
+ wandb: Tracking run with wandb version 0.25.0
609
+ wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_142106-h4cbkppl
610
+ wandb: Run `wandb offline` to turn off syncing.
611
+ wandb: Syncing run default_phase2
612
+ wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
613
+ wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/h4cbkppl
614
+
615
  0%| | 0/30000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
616
+ Could not estimate the number of tokens of the input, floating-point operations will not be computed
617
+
618
  0%| | 1/30000 [00:05<43:01:49, 5.16s/it][rank1]: Traceback (most recent call last):
619
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
620
+ [rank1]: run_yaml_experiment(
621
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
622
+ [rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
623
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
624
+ [rank1]: experiment.train()
625
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
626
+ [rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
627
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
628
+ [rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
629
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
630
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
631
+ [rank1]: return inner_training_loop(
632
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
633
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
634
+ [rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
635
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
636
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
637
+ [rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
638
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
639
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
640
+ [rank1]: outputs = model(inputs)
641
+ [rank1]: ^^^^^^^^^^^^^
642
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
643
+ [rank1]: return self._call_impl(*args, **kwargs)
644
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
645
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
646
+ [rank1]: return forward_call(*args, **kwargs)
647
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
648
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1633, in forward
649
+ [rank1]: inputs, kwargs = self._pre_forward(*inputs, **kwargs)
650
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
651
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1522, in _pre_forward
652
+ [rank1]: if torch.is_grad_enabled() and self.reducer._rebuild_buckets():
653
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
654
+ [rank1]: RuntimeError: Expected to have finished reduction in the prior iteration before starting a new one. This error indicates that your module has parameters that were not used in producing loss. You can enable unused parameter detection by passing the keyword argument `find_unused_parameters=True` to `torch.nn.parallel.DistributedDataParallel`, and by
655
+ [rank1]: making sure all `forward` function outputs participate in calculating loss.
656
+ [rank1]: If you already have done the above, then the distributed data parallel module wasn't able to locate the output tensors in the return value of your module's `forward` function. Please include the loss function and the structure of the return value of `forward` of your module when reporting this issue (e.g. list, dict, iterable).
657
+ [rank1]: Parameter indices which did not receive grad for rank 1: 582
658
+ [rank1]: In addition, you can set the environment variable TORCH_DISTRIBUTED_DEBUG to either INFO or DETAIL to print out information about which particular parameters did not receive gradient on this rank as part of this error
659
+ wandb: uploading wandb-metadata.json; uploading requirements.txt; updating run metadata
660
+ wandb: uploading wandb-metadata.json; uploading requirements.txt; uploading wandb-summary.json; uploading config.yaml; uploading output.log
661
+ wandb: uploading summary, console lines 0-1
662
+ wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/h4cbkppl
663
+ wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
664
+ wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
665
+ wandb: Find logs at: ./wandb/run-20260615_142106-h4cbkppl/logs
666
+ Traceback (most recent call last):
667
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
668
+ run_yaml_experiment(
669
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
670
+ main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
671
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
672
+ experiment.train()
673
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
674
+ self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
675
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
676
+ return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
677
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
678
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
679
+ return inner_training_loop(
680
+ ^^^^^^^^^^^^^^^^^^^^
681
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
682
+ tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
683
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
684
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
685
+ loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
686
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
687
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
688
+ outputs = model(inputs)
689
+ ^^^^^^^^^^^^^
690
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
691
+ return self._call_impl(*args, **kwargs)
692
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
693
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
694
+ return forward_call(*args, **kwargs)
695
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
696
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1633, in forward
697
+ inputs, kwargs = self._pre_forward(*inputs, **kwargs)
698
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
699
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1522, in _pre_forward
700
+ if torch.is_grad_enabled() and self.reducer._rebuild_buckets():
701
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
702
+ RuntimeError: Expected to have finished reduction in the prior iteration before starting a new one. This error indicates that your module has parameters that were not used in producing loss. You can enable unused parameter detection by passing the keyword argument `find_unused_parameters=True` to `torch.nn.parallel.DistributedDataParallel`, and by
703
+ making sure all `forward` function outputs participate in calculating loss.
704
+ If you already have done the above, then the distributed data parallel module wasn't able to locate the output tensors in the return value of your module's `forward` function. Please include the loss function and the structure of the return value of `forward` of your module when reporting this issue (e.g. list, dict, iterable).
705
+ Parameter indices which did not receive grad for rank 0: 582
706
+ In addition, you can set the environment variable TORCH_DISTRIBUTED_DEBUG to either INFO or DETAIL to print out information about which particular parameters did not receive gradient on this rank as part of this error
707
+ [rank0]: Traceback (most recent call last):
708
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
709
+ [rank0]: run_yaml_experiment(
710
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
711
+ [rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
712
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
713
+ [rank0]: experiment.train()
714
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
715
+ [rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
716
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
717
+ [rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
718
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
719
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
720
+ [rank0]: return inner_training_loop(
721
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
722
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
723
+ [rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
724
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
725
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
726
+ [rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
727
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
728
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
729
+ [rank0]: outputs = model(inputs)
730
+ [rank0]: ^^^^^^^^^^^^^
731
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
732
+ [rank0]: return self._call_impl(*args, **kwargs)
733
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
734
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
735
+ [rank0]: return forward_call(*args, **kwargs)
736
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
737
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1633, in forward
738
+ [rank0]: inputs, kwargs = self._pre_forward(*inputs, **kwargs)
739
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
740
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1522, in _pre_forward
741
+ [rank0]: if torch.is_grad_enabled() and self.reducer._rebuild_buckets():
742
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
743
+ [rank0]: RuntimeError: Expected to have finished reduction in the prior iteration before starting a new one. This error indicates that your module has parameters that were not used in producing loss. You can enable unused parameter detection by passing the keyword argument `find_unused_parameters=True` to `torch.nn.parallel.DistributedDataParallel`, and by
744
+ [rank0]: making sure all `forward` function outputs participate in calculating loss.
745
+ [rank0]: If you already have done the above, then the distributed data parallel module wasn't able to locate the output tensors in the return value of your module's `forward` function. Please include the loss function and the structure of the return value of `forward` of your module when reporting this issue (e.g. list, dict, iterable).
746
+ [rank0]: Parameter indices which did not receive grad for rank 0: 582
747
+ [rank0]: In addition, you can set the environment variable TORCH_DISTRIBUTED_DEBUG to either INFO or DETAIL to print out information about which particular parameters did not receive gradient on this rank as part of this error
748
+ [rank0]:[W615 14:21:15.946760892 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
749
+ W0615 14:21:15.519000 1313589 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 1319773 closing signal SIGTERM
750
+ E0615 14:21:15.883000 1313589 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 1319776) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
751
+ Traceback (most recent call last):
752
+ File "<frozen runpy>", line 198, in _run_module_as_main
753
+ File "<frozen runpy>", line 88, in _run_code
754
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
755
+ main()
756
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
757
+ return f(*args, **kwargs)
758
+ ^^^^^^^^^^^^^^^^^^
759
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
760
+ run(args)
761
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
762
+ elastic_launch(
763
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
764
+ return launch_agent(self._config, self._entrypoint, list(args))
765
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
766
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
767
+ raise ChildFailedError(
768
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
769
+ ============================================================
770
+ /home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
771
+ ------------------------------------------------------------
772
+ Failures:
773
+ <NO_OTHER_FAILURES>
774
+ ------------------------------------------------------------
775
+ Root Cause (first observed failure):
776
+ [0]:
777
+ time : 2026-06-15_14:21:15
778
+ host : worker1
779
+ rank : 1 (local_rank: 1)
780
+ exitcode : 1 (pid: 1319776)
781
+ error_file: <N/A>
782
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
783
+ ============================================================
784
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
785
+
786
+ ==================================================
787
+ GR00T FINE-TUNING CONFIGURATION:
788
+ ==================================================
789
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
790
+ dataset_soup: None
791
+ output_dir: /tmp/gr00t
792
+ output_root: None
793
+ data_config: panda_omron
794
+ batch_size: 16
795
+ max_steps: 300000
796
+ num_gpus: 2
797
+ save_steps: 20000
798
+ run_name: None
799
+ save_total_limit: 100
800
+ seed: 42
801
+ base_model_path: nvidia/GR00T-N1.5-3B
802
+ tune_llm: False
803
+ tune_visual: False
804
+ tune_projector: True
805
+ tune_diffusion_model: True
806
+ resume: False
807
+ learning_rate: 3e-05
808
+ weight_decay: 1e-05
809
+ warmup_ratio: 0.05
810
+ lora_rank: 0
811
+ lora_alpha: 16
812
+ lora_dropout: 0.1
813
+ lora_full_model: False
814
+ dataloader_num_workers: 8
815
+ report_to: wandb
816
+ embodiment_tag: new_embodiment
817
+ video_backend: opencv
818
+ balance_dataset_weights: True
819
+ balance_trajectory_weights: True
820
+ ds_weights_alpha: 0.4
821
+ ==================================================
822
+
823
+ Using 2 GPUs
824
+ Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '16', '--num-gpus', '2']
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_3_60k.log ADDED
@@ -0,0 +1,1088 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/60000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
 
 
1
  0%| | 1/60000 [00:08<139:31:18, 8.37s/it][rank1]: Traceback (most recent call last):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [robosuite WARNING] No private macro file found! (macros.py:57)
2
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
3
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
4
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
5
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
6
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
7
+ check_for_updates()
8
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
9
+
10
+ *****************************************
11
+ Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
12
+ *****************************************
13
+ [robosuite WARNING] No private macro file found! (macros.py:57)
14
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
15
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
16
+ [robosuite WARNING] No private macro file found! (macros.py:57)
17
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
18
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
19
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
20
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
21
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
22
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
23
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
24
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
25
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
26
+ check_for_updates()
27
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
28
+ check_for_updates()
29
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
30
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
31
+
32
+ ==================================================
33
+ GR00T FINE-TUNING CONFIGURATION:
34
+ ==================================================
35
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
36
+ dataset_soup: None
37
+ output_dir: /tmp/gr00t
38
+ output_root: None
39
+ data_config: panda_omron
40
+ batch_size: 32
41
+ max_steps: 300000
42
+ num_gpus: 2
43
+ save_steps: 20000
44
+ run_name: None
45
+ save_total_limit: 100
46
+ seed: 42
47
+ base_model_path: nvidia/GR00T-N1.5-3B
48
+ tune_llm: False
49
+ tune_visual: False
50
+ tune_projector: True
51
+ tune_diffusion_model: True
52
+ resume: False
53
+ learning_rate: 3e-05
54
+ weight_decay: 1e-05
55
+ warmup_ratio: 0.05
56
+ lora_rank: 0
57
+ lora_alpha: 16
58
+ lora_dropout: 0.1
59
+ lora_full_model: False
60
+ dataloader_num_workers: 8
61
+ report_to: wandb
62
+ embodiment_tag: new_embodiment
63
+ video_backend: opencv
64
+ balance_dataset_weights: True
65
+ balance_trajectory_weights: True
66
+ ds_weights_alpha: 0.4
67
+ ==================================================
68
+
69
+ Using 2 GPUs
70
+
71
+ ================================================================================
72
+ Starting sweep branch: default
73
+ Sweep vars: {}
74
+ ================================================================================
75
+
76
+ --------------------------------------------------------------------------------
77
+ Running phase 1: phase2_rkd_a_vlm_only
78
+ Policy type: groot_rkd_v2_raw
79
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
80
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
81
+ Trainable preset: freeze_processing_line
82
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
83
+ --------------------------------------------------------------------------------
84
+
85
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
86
+ Using 100 subset demos for filter_key: 100_demos
87
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
88
+ self.statistics[key] = torch.tensor(value)
89
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
90
+ Using 100 subset demos for filter_key: 100_demos
91
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
92
+ Using 100 subset demos for filter_key: 100_demos
93
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
94
+ Using 100 subset demos for filter_key: 100_demos
95
+
96
+ ==================================================
97
+ GR00T FINE-TUNING CONFIGURATION:
98
+ ==================================================
99
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
100
+ dataset_soup: None
101
+ output_dir: /tmp/gr00t
102
+ output_root: None
103
+ data_config: panda_omron
104
+ batch_size: 32
105
+ max_steps: 300000
106
+ num_gpus: 2
107
+ save_steps: 20000
108
+ run_name: None
109
+ save_total_limit: 100
110
+ seed: 42
111
+ base_model_path: nvidia/GR00T-N1.5-3B
112
+ tune_llm: False
113
+ tune_visual: False
114
+ tune_projector: True
115
+ tune_diffusion_model: True
116
+ resume: False
117
+ learning_rate: 3e-05
118
+ weight_decay: 1e-05
119
+ warmup_ratio: 0.05
120
+ lora_rank: 0
121
+ lora_alpha: 16
122
+ lora_dropout: 0.1
123
+ lora_full_model: False
124
+ dataloader_num_workers: 8
125
+ report_to: wandb
126
+ embodiment_tag: new_embodiment
127
+ video_backend: opencv
128
+ balance_dataset_weights: True
129
+ balance_trajectory_weights: True
130
+ ds_weights_alpha: 0.4
131
+ ==================================================
132
+
133
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
134
+ Using 100 subset demos for filter_key: 100_demos
135
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
136
+ Using 100 subset demos for filter_key: 100_demos
137
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
138
+ Using 100 subset demos for filter_key: 100_demos
139
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
140
+ Using 100 subset demos for filter_key: 100_demos
141
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
142
+ Using 100 subset demos for filter_key: 100_demos
143
+ Using 2 GPUs
144
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
145
+ Using 100 subset demos for filter_key: 100_demos
146
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
147
+
148
+ ================================================================================
149
+ Starting sweep branch: default
150
+ Sweep vars: {}
151
+ ================================================================================
152
+
153
+ --------------------------------------------------------------------------------
154
+ Running phase 1: phase2_rkd_a_vlm_only
155
+ Policy type: groot_rkd_v2_raw
156
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
157
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
158
+ Trainable preset: freeze_processing_line
159
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
160
+ --------------------------------------------------------------------------------
161
+
162
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
163
+ Using 100 subset demos for filter_key: 100_demos
164
+ Using 100 subset demos for filter_key: 100_demos
165
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
166
+ self.statistics[key] = torch.tensor(value)
167
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
168
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
169
+ Using 100 subset demos for filter_key: 100_demos
170
+ Using 100 subset demos for filter_key: 100_demos
171
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
172
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
173
+ Using 100 subset demos for filter_key: 100_demos
174
+ Using 100 subset demos for filter_key: 100_demos
175
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
176
+ Using 100 subset demos for filter_key: 100_demos
177
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
178
+ Using 100 subset demos for filter_key: 100_demos
179
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
180
+ Using 100 subset demos for filter_key: 100_demos
181
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
182
+ Using 100 subset demos for filter_key: 100_demos
183
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
184
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
185
+ Using 100 subset demos for filter_key: 100_demos
186
+ Using 100 subset demos for filter_key: 100_demos
187
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
188
+ Using 100 subset demos for filter_key: 100_demos
189
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
190
+ Using 100 subset demos for filter_key: 100_demos
191
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
192
+ Using 100 subset demos for filter_key: 100_demos
193
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
194
+ Using 100 subset demos for filter_key: 100_demos
195
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
196
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
197
+ Using 100 subset demos for filter_key: 100_demos
198
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
199
+ Using 100 subset demos for filter_key: 100_demos
200
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
201
+ Using 100 subset demos for filter_key: 100_demos
202
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
203
+ Using 100 subset demos for filter_key: 100_demos
204
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
205
+ Using 100 subset demos for filter_key: 100_demos
206
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
207
+ Using 100 subset demos for filter_key: 100_demos
208
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
209
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
210
+ Using 100 subset demos for filter_key: 100_demos
211
+ Using 100 subset demos for filter_key: 100_demos
212
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
213
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
214
+ Using 100 subset demos for filter_key: 100_demos
215
+ Using 100 subset demos for filter_key: 100_demos
216
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
217
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
218
+ Using 100 subset demos for filter_key: 100_demos
219
+ Using 100 subset demos for filter_key: 100_demos
220
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
221
+ Using 100 subset demos for filter_key: 100_demos
222
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
223
+ Using 100 subset demos for filter_key: 100_demos
224
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
225
+ Using 100 subset demos for filter_key: 100_demos
226
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
227
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
228
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
229
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
230
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
231
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
232
+ 0.75517122 0.7973985 ]
233
+ Using 100 subset demos for filter_key: 100_demos
234
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
235
+ Using 100 subset demos for filter_key: 100_demos
236
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
237
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
238
+ Using 100 subset demos for filter_key: 100_demos
239
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
240
+ Using 100 subset demos for filter_key: 100_demos
241
+ Loaded 26 datasets
242
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
243
+ Using 100 subset demos for filter_key: 100_demos
244
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
245
+ Using 100 subset demos for filter_key: 100_demos
246
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
247
+ Using 100 subset demos for filter_key: 100_demos
248
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
249
+ Tune backbone vision tower: True
250
+ Tune backbone LLM: True
251
+ Tune action head projector: False
252
+ Tune action head DiT: False
253
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
254
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
255
+ Using 100 subset demos for filter_key: 100_demos
256
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
257
+ Using 100 subset demos for filter_key: 100_demos
258
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
259
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
260
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
261
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
262
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
263
+ 0.75517122 0.7973985 ]
264
+ Loaded 26 datasets
265
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
266
+ Tune backbone vision tower: True
267
+ Tune backbone LLM: True
268
+ Tune action head projector: False
269
+ Tune action head DiT: False
270
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
271
+ Tune backbone llm: False
272
+ Tune backbone visual: True
273
+ Total number of DiT parameters: 550386688
274
+ Tune backbone llm: False
275
+ Tune backbone visual: True
276
+ Total number of DiT parameters: 550386688
277
+ Total number of SelfAttentionTransformer parameters: 201433088
278
+ Tune action head projector: True
279
+ Tune action head diffusion model: True
280
+
281
+ Tune action head projector: True
282
+ Tune action head diffusion model: True
283
+
284
+ Tune backbone llm: True
285
+ Tune backbone visual: True
286
+ Tune action head projector: False
287
+ Tune action head diffusion model: False
288
+ Action head trainable parameter: future_tokens.weight
289
+ Action head trainable parameter: vlln.weight
290
+ Action head trainable parameter: vlln.bias
291
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
292
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
293
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
294
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
295
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
296
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
297
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
298
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
299
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
300
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
301
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
302
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
303
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
304
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
305
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
306
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
307
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
308
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
309
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
310
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
311
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
312
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
313
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
314
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
315
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
316
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
317
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
318
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
319
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
320
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
321
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
322
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
323
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
324
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
325
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
326
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
327
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
328
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
329
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
330
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
331
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
332
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
333
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
334
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
335
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
336
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
337
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
338
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
339
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
340
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
341
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
342
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
343
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
344
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
345
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
346
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
347
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
348
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
349
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
350
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
351
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
352
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
353
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
354
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
355
+ Applied trainable preset: freeze_processing_line
356
+ Trainable parameter tensors after preset: 584
357
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
358
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
359
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
360
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
361
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
362
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
363
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
364
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
365
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
366
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
367
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
368
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
369
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
370
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
371
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
372
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
373
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
374
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
375
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
376
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
377
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
378
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
379
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
380
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
381
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
382
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
383
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
384
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
385
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
386
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
387
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
388
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
389
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
390
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
391
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
392
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
393
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
394
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
395
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
396
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
397
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
398
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
399
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
400
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
401
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
402
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
403
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
404
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
405
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
406
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
407
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
408
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
409
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
410
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
411
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
412
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
413
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
414
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
415
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
416
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
417
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
418
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
419
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
420
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
421
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
422
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
423
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
424
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
425
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
426
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
427
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
428
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
429
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
430
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
431
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
432
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
433
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
434
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
435
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
436
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
437
+ ... 504 more
438
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
439
+
440
+ Tune backbone llm: True
441
+ Tune backbone visual: True
442
+ Tune action head projector: False
443
+ Tune action head diffusion model: False
444
+ Action head trainable parameter: future_tokens.weight
445
+ Action head trainable parameter: vlln.weight
446
+ Action head trainable parameter: vlln.bias
447
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
448
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
449
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
450
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
451
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
452
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
453
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
454
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
455
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
456
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
457
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
458
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
459
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
460
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
461
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
462
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
463
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
464
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
465
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
466
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
467
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
468
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
469
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
470
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
471
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
472
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
473
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
474
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
475
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
476
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
477
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
478
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
479
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
480
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
481
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
482
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
483
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
484
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
485
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
486
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
487
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
488
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
489
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
490
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
491
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
492
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
493
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
494
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
495
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
496
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
497
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
498
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
499
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
500
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
501
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
502
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
503
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
504
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
505
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
506
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
507
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
508
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
509
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
510
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
511
+ Applied trainable preset: freeze_processing_line
512
+ Trainable parameter tensors after preset: 584
513
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
514
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
515
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
516
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
517
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
518
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
519
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
520
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
521
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
522
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
523
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
524
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
525
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
526
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
527
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
528
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
529
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
530
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
531
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
532
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
533
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
534
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
535
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
536
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
537
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
538
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
539
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
540
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
541
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
542
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
543
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
544
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
545
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
546
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
547
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
548
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
549
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
550
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
551
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
552
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
553
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
554
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
555
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
556
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
557
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
558
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
559
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
560
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
561
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
562
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
563
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
564
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
565
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
566
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
567
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
568
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
569
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
570
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
571
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
572
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
573
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
574
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
575
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
576
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
577
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
578
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
579
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
580
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
581
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
582
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
583
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
584
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
585
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
586
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
587
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
588
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
589
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
590
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
591
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
592
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
593
+ ... 504 more
594
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
595
+ Run name: default_phase2
596
+ Run name: default_phase2
597
+ train dataloader length: 6873
598
+ train dataset length: 439854
599
+ GPU memory before training: 7.076685905456543 GB
600
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
601
+ train dataloader length: 6873
602
+ train dataset length: 439854
603
+ GPU memory before training: 7.076685905456543 GB
604
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
605
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
606
+ wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
607
+ wandb: setting up run tu7a0ujw
608
+ wandb: Tracking run with wandb version 0.25.0
609
+ wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_150511-tu7a0ujw
610
+ wandb: Run `wandb offline` to turn off syncing.
611
+ wandb: Syncing run default_phase2
612
+ wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
613
+ wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/tu7a0ujw
614
+
615
  0%| | 0/60000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
616
+ Could not estimate the number of tokens of the input, floating-point operations will not be computed
617
+
618
  0%| | 1/60000 [00:08<139:31:18, 8.37s/it][rank1]: Traceback (most recent call last):
619
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
620
+ [rank1]: run_yaml_experiment(
621
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
622
+ [rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
623
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
624
+ [rank1]: experiment.train()
625
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
626
+ [rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
627
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
628
+ [rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
629
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
630
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
631
+ [rank1]: return inner_training_loop(
632
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
633
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
634
+ [rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
635
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
636
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
637
+ [rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
638
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
639
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
640
+ [rank1]: outputs = model(inputs)
641
+ [rank1]: ^^^^^^^^^^^^^
642
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
643
+ [rank1]: return self._call_impl(*args, **kwargs)
644
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
645
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
646
+ [rank1]: return forward_call(*args, **kwargs)
647
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
648
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
649
+ [rank1]: else self._run_ddp_forward(*inputs, **kwargs)
650
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
651
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
652
+ [rank1]: return self.module(*inputs, **kwargs) # type: ignore[index]
653
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
654
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
655
+ [rank1]: return self._call_impl(*args, **kwargs)
656
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
657
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
658
+ [rank1]: return forward_call(*args, **kwargs)
659
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
660
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
661
+ [rank1]: return model_forward(*args, **kwargs)
662
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
663
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
664
+ [rank1]: return convert_to_fp32(self.model_forward(*args, **kwargs))
665
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
666
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
667
+ [rank1]: return func(*args, **kwargs)
668
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^
669
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
670
+ [rank1]: backbone_outputs = self.backbone(backbone_inputs)
671
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
672
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
673
+ [rank1]: return self._call_impl(*args, **kwargs)
674
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
675
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
676
+ [rank1]: return forward_call(*args, **kwargs)
677
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
678
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 123, in forward
679
+ [rank1]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
680
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
681
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
682
+ [rank1]: eagle_output = self.eagle_model(
683
+ [rank1]: ^^^^^^^^^^^^^^^^^
684
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
685
+ [rank1]: return self._call_impl(*args, **kwargs)
686
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
687
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
688
+ [rank1]: return forward_call(*args, **kwargs)
689
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
690
+ [rank1]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 262, in forward
691
+ [rank1]: outputs = self.language_model(
692
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
693
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
694
+ [rank1]: return self._call_impl(*args, **kwargs)
695
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
696
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
697
+ [rank1]: return forward_call(*args, **kwargs)
698
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
699
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
700
+ [rank1]: output = func(self, *args, **kwargs)
701
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
702
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
703
+ [rank1]: return func(*args, **kwargs)
704
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^
705
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
706
+ [rank1]: outputs: BaseModelOutputWithPast = self.model(
707
+ [rank1]: ^^^^^^^^^^^
708
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
709
+ [rank1]: return self._call_impl(*args, **kwargs)
710
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
711
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
712
+ [rank1]: return forward_call(*args, **kwargs)
713
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
714
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
715
+ [rank1]: output = func(self, *args, **kwargs)
716
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
717
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
718
+ [rank1]: layer_outputs = decoder_layer(
719
+ [rank1]: ^^^^^^^^^^^^^^
720
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
721
+ [rank1]: return self._call_impl(*args, **kwargs)
722
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
723
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
724
+ [rank1]: return forward_call(*args, **kwargs)
725
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
726
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 289, in forward
727
+ [rank1]: hidden_states, self_attn_weights = self.self_attn(
728
+ [rank1]: ^^^^^^^^^^^^^^^
729
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
730
+ [rank1]: return self._call_impl(*args, **kwargs)
731
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
732
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
733
+ [rank1]: return forward_call(*args, **kwargs)
734
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
735
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 217, in forward
736
+ [rank1]: query_states = self.q_norm(self.q_proj(hidden_states).view(hidden_shape)).transpose(1, 2)
737
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
738
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
739
+ [rank1]: return self._call_impl(*args, **kwargs)
740
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
741
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
742
+ [rank1]: return forward_call(*args, **kwargs)
743
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
744
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 74, in forward
745
+ [rank1]: variance = hidden_states.pow(2).mean(-1, keepdim=True)
746
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
747
+ [rank1]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 210.00 MiB. GPU 1 has a total capacity of 79.25 GiB of which 129.94 MiB is free. Including non-PyTorch memory, this process has 79.10 GiB memory in use. Of the allocated memory 73.70 GiB is allocated by PyTorch, and 4.76 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
748
+ wandb: updating run metadata
749
+ wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
750
+ wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/tu7a0ujw
751
+ wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
752
+ wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
753
+ wandb: Find logs at: ./wandb/run-20260615_150511-tu7a0ujw/logs
754
+ Traceback (most recent call last):
755
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
756
+ run_yaml_experiment(
757
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
758
+ main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
759
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
760
+ experiment.train()
761
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
762
+ self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
763
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
764
+ return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
765
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
766
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
767
+ return inner_training_loop(
768
+ ^^^^^^^^^^^^^^^^^^^^
769
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
770
+ tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
771
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
772
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
773
+ loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
774
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
775
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
776
+ outputs = model(inputs)
777
+ ^^^^^^^^^^^^^
778
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
779
+ return self._call_impl(*args, **kwargs)
780
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
781
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
782
+ return forward_call(*args, **kwargs)
783
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
784
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
785
+ else self._run_ddp_forward(*inputs, **kwargs)
786
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
787
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
788
+ return self.module(*inputs, **kwargs) # type: ignore[index]
789
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
790
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
791
+ return self._call_impl(*args, **kwargs)
792
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
793
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
794
+ return forward_call(*args, **kwargs)
795
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
796
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
797
+ return model_forward(*args, **kwargs)
798
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
799
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
800
+ return convert_to_fp32(self.model_forward(*args, **kwargs))
801
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
802
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
803
+ return func(*args, **kwargs)
804
+ ^^^^^^^^^^^^^^^^^^^^^
805
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
806
+ backbone_outputs = self.backbone(backbone_inputs)
807
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
808
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
809
+ return self._call_impl(*args, **kwargs)
810
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
811
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
812
+ return forward_call(*args, **kwargs)
813
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
814
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 123, in forward
815
+ eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
816
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
817
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
818
+ eagle_output = self.eagle_model(
819
+ ^^^^^^^^^^^^^^^^^
820
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
821
+ return self._call_impl(*args, **kwargs)
822
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
823
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
824
+ return forward_call(*args, **kwargs)
825
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
826
+ File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 262, in forward
827
+ outputs = self.language_model(
828
+ ^^^^^^^^^^^^^^^^^^^^
829
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
830
+ return self._call_impl(*args, **kwargs)
831
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
832
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
833
+ return forward_call(*args, **kwargs)
834
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
835
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
836
+ output = func(self, *args, **kwargs)
837
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
838
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
839
+ return func(*args, **kwargs)
840
+ ^^^^^^^^^^^^^^^^^^^^^
841
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
842
+ outputs: BaseModelOutputWithPast = self.model(
843
+ ^^^^^^^^^^^
844
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
845
+ return self._call_impl(*args, **kwargs)
846
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
847
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
848
+ return forward_call(*args, **kwargs)
849
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
850
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
851
+ output = func(self, *args, **kwargs)
852
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
853
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
854
+ layer_outputs = decoder_layer(
855
+ ^^^^^^^^^^^^^^
856
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
857
+ return self._call_impl(*args, **kwargs)
858
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
859
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
860
+ return forward_call(*args, **kwargs)
861
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
862
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 289, in forward
863
+ hidden_states, self_attn_weights = self.self_attn(
864
+ ^^^^^^^^^^^^^^^
865
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
866
+ return self._call_impl(*args, **kwargs)
867
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
868
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
869
+ return forward_call(*args, **kwargs)
870
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
871
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 217, in forward
872
+ query_states = self.q_norm(self.q_proj(hidden_states).view(hidden_shape)).transpose(1, 2)
873
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
874
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
875
+ return self._call_impl(*args, **kwargs)
876
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
877
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
878
+ return forward_call(*args, **kwargs)
879
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
880
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 74, in forward
881
+ variance = hidden_states.pow(2).mean(-1, keepdim=True)
882
+ ^^^^^^^^^^^^^^^^^^^^
883
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 210.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 129.94 MiB is free. Including non-PyTorch memory, this process has 79.10 GiB memory in use. Of the allocated memory 73.70 GiB is allocated by PyTorch, and 4.76 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
884
+ [rank0]: Traceback (most recent call last):
885
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
886
+ [rank0]: run_yaml_experiment(
887
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
888
+ [rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
889
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
890
+ [rank0]: experiment.train()
891
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
892
+ [rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
893
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
894
+ [rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
895
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
896
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
897
+ [rank0]: return inner_training_loop(
898
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
899
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
900
+ [rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
901
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
902
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
903
+ [rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
904
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
905
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
906
+ [rank0]: outputs = model(inputs)
907
+ [rank0]: ^^^^^^^^^^^^^
908
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
909
+ [rank0]: return self._call_impl(*args, **kwargs)
910
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
911
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
912
+ [rank0]: return forward_call(*args, **kwargs)
913
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
914
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
915
+ [rank0]: else self._run_ddp_forward(*inputs, **kwargs)
916
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
917
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
918
+ [rank0]: return self.module(*inputs, **kwargs) # type: ignore[index]
919
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
920
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
921
+ [rank0]: return self._call_impl(*args, **kwargs)
922
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
923
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
924
+ [rank0]: return forward_call(*args, **kwargs)
925
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
926
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
927
+ [rank0]: return model_forward(*args, **kwargs)
928
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
929
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
930
+ [rank0]: return convert_to_fp32(self.model_forward(*args, **kwargs))
931
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
932
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
933
+ [rank0]: return func(*args, **kwargs)
934
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^
935
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
936
+ [rank0]: backbone_outputs = self.backbone(backbone_inputs)
937
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
938
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
939
+ [rank0]: return self._call_impl(*args, **kwargs)
940
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
941
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
942
+ [rank0]: return forward_call(*args, **kwargs)
943
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
944
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 123, in forward
945
+ [rank0]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
946
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
947
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
948
+ [rank0]: eagle_output = self.eagle_model(
949
+ [rank0]: ^^^^^^^^^^^^^^^^^
950
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
951
+ [rank0]: return self._call_impl(*args, **kwargs)
952
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
953
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
954
+ [rank0]: return forward_call(*args, **kwargs)
955
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
956
+ [rank0]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 262, in forward
957
+ [rank0]: outputs = self.language_model(
958
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
959
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
960
+ [rank0]: return self._call_impl(*args, **kwargs)
961
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
962
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
963
+ [rank0]: return forward_call(*args, **kwargs)
964
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
965
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
966
+ [rank0]: output = func(self, *args, **kwargs)
967
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
968
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
969
+ [rank0]: return func(*args, **kwargs)
970
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^
971
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
972
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
973
+ [rank0]: ^^^^^^^^^^^
974
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
975
+ [rank0]: return self._call_impl(*args, **kwargs)
976
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
977
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
978
+ [rank0]: return forward_call(*args, **kwargs)
979
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
980
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
981
+ [rank0]: output = func(self, *args, **kwargs)
982
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
983
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
984
+ [rank0]: layer_outputs = decoder_layer(
985
+ [rank0]: ^^^^^^^^^^^^^^
986
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
987
+ [rank0]: return self._call_impl(*args, **kwargs)
988
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
989
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
990
+ [rank0]: return forward_call(*args, **kwargs)
991
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
992
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 289, in forward
993
+ [rank0]: hidden_states, self_attn_weights = self.self_attn(
994
+ [rank0]: ^^^^^^^^^^^^^^^
995
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
996
+ [rank0]: return self._call_impl(*args, **kwargs)
997
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
998
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
999
+ [rank0]: return forward_call(*args, **kwargs)
1000
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1001
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 217, in forward
1002
+ [rank0]: query_states = self.q_norm(self.q_proj(hidden_states).view(hidden_shape)).transpose(1, 2)
1003
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1004
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
1005
+ [rank0]: return self._call_impl(*args, **kwargs)
1006
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1007
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
1008
+ [rank0]: return forward_call(*args, **kwargs)
1009
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1010
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 74, in forward
1011
+ [rank0]: variance = hidden_states.pow(2).mean(-1, keepdim=True)
1012
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
1013
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 210.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 129.94 MiB is free. Including non-PyTorch memory, this process has 79.10 GiB memory in use. Of the allocated memory 73.70 GiB is allocated by PyTorch, and 4.76 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
1014
+ [rank0]:[W615 15:05:24.539442388 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
1015
+ W0615 15:05:24.803000 2931837 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 2938608 closing signal SIGTERM
1016
+ E0615 15:05:25.318000 2931837 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 2938609) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
1017
+ Traceback (most recent call last):
1018
+ File "<frozen runpy>", line 198, in _run_module_as_main
1019
+ File "<frozen runpy>", line 88, in _run_code
1020
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
1021
+ main()
1022
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
1023
+ return f(*args, **kwargs)
1024
+ ^^^^^^^^^^^^^^^^^^
1025
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
1026
+ run(args)
1027
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
1028
+ elastic_launch(
1029
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
1030
+ return launch_agent(self._config, self._entrypoint, list(args))
1031
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1032
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
1033
+ raise ChildFailedError(
1034
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
1035
+ ============================================================
1036
+ /home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
1037
+ ------------------------------------------------------------
1038
+ Failures:
1039
+ <NO_OTHER_FAILURES>
1040
+ ------------------------------------------------------------
1041
+ Root Cause (first observed failure):
1042
+ [0]:
1043
+ time : 2026-06-15_15:05:24
1044
+ host : worker1
1045
+ rank : 1 (local_rank: 1)
1046
+ exitcode : 1 (pid: 2938609)
1047
+ error_file: <N/A>
1048
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
1049
+ ============================================================
1050
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
1051
+
1052
+ ==================================================
1053
+ GR00T FINE-TUNING CONFIGURATION:
1054
+ ==================================================
1055
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
1056
+ dataset_soup: None
1057
+ output_dir: /tmp/gr00t
1058
+ output_root: None
1059
+ data_config: panda_omron
1060
+ batch_size: 32
1061
+ max_steps: 300000
1062
+ num_gpus: 2
1063
+ save_steps: 20000
1064
+ run_name: None
1065
+ save_total_limit: 100
1066
+ seed: 42
1067
+ base_model_path: nvidia/GR00T-N1.5-3B
1068
+ tune_llm: False
1069
+ tune_visual: False
1070
+ tune_projector: True
1071
+ tune_diffusion_model: True
1072
+ resume: False
1073
+ learning_rate: 3e-05
1074
+ weight_decay: 1e-05
1075
+ warmup_ratio: 0.05
1076
+ lora_rank: 0
1077
+ lora_alpha: 16
1078
+ lora_dropout: 0.1
1079
+ lora_full_model: False
1080
+ dataloader_num_workers: 8
1081
+ report_to: wandb
1082
+ embodiment_tag: new_embodiment
1083
+ video_backend: opencv
1084
+ balance_dataset_weights: True
1085
+ balance_trajectory_weights: True
1086
+ ds_weights_alpha: 0.4
1087
+ ==================================================
1088
+
1089
+ Using 2 GPUs
1090
+ Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '32', '--num-gpus', '2']
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_3_probe.log ADDED
@@ -0,0 +1,997 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [robosuite WARNING] No private macro file found! (macros.py:57)
2
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
3
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
4
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
5
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
6
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
7
+ check_for_updates()
8
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
9
+
10
+ *****************************************
11
+ Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
12
+ *****************************************
13
+ [robosuite WARNING] No private macro file found! (macros.py:57)
14
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
15
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
16
+ [robosuite WARNING] No private macro file found! (macros.py:57)
17
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
18
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
19
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
20
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
21
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
22
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
23
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
24
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
25
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
26
+ check_for_updates()
27
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
28
+ check_for_updates()
29
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
30
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
31
+
32
+ ==================================================
33
+ GR00T FINE-TUNING CONFIGURATION:
34
+ ==================================================
35
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
36
+ dataset_soup: None
37
+ output_dir: /tmp/gr00t
38
+ output_root: None
39
+ data_config: panda_omron
40
+ batch_size: 32
41
+ max_steps: 300000
42
+ num_gpus: 2
43
+ save_steps: 20000
44
+ run_name: None
45
+ save_total_limit: 100
46
+ seed: 42
47
+ base_model_path: nvidia/GR00T-N1.5-3B
48
+ tune_llm: False
49
+ tune_visual: False
50
+ tune_projector: True
51
+ tune_diffusion_model: True
52
+ resume: False
53
+ learning_rate: 3e-05
54
+ weight_decay: 1e-05
55
+ warmup_ratio: 0.05
56
+ lora_rank: 0
57
+ lora_alpha: 16
58
+ lora_dropout: 0.1
59
+ lora_full_model: False
60
+ dataloader_num_workers: 8
61
+ report_to: wandb
62
+ embodiment_tag: new_embodiment
63
+ video_backend: opencv
64
+ balance_dataset_weights: True
65
+ balance_trajectory_weights: True
66
+ ds_weights_alpha: 0.4
67
+ ==================================================
68
+
69
+ Using 2 GPUs
70
+
71
+ ================================================================================
72
+ Starting sweep branch: default
73
+ Sweep vars: {}
74
+ ================================================================================
75
+
76
+ --------------------------------------------------------------------------------
77
+ Running phase 1: phase2_rkd_a_vlm_only
78
+ Policy type: groot_rkd_v2_raw
79
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
80
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
81
+ Trainable preset: freeze_processing_line
82
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
83
+ --------------------------------------------------------------------------------
84
+
85
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
86
+ Using 100 subset demos for filter_key: 100_demos
87
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
88
+ self.statistics[key] = torch.tensor(value)
89
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
90
+ Using 100 subset demos for filter_key: 100_demos
91
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
92
+ Using 100 subset demos for filter_key: 100_demos
93
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
94
+ Using 100 subset demos for filter_key: 100_demos
95
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
96
+ Using 100 subset demos for filter_key: 100_demos
97
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
98
+ Using 100 subset demos for filter_key: 100_demos
99
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
100
+ Using 100 subset demos for filter_key: 100_demos
101
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
102
+ Using 100 subset demos for filter_key: 100_demos
103
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
104
+ Using 100 subset demos for filter_key: 100_demos
105
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
106
+ Using 100 subset demos for filter_key: 100_demos
107
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
108
+ Using 100 subset demos for filter_key: 100_demos
109
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
110
+ Using 100 subset demos for filter_key: 100_demos
111
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
112
+ Using 100 subset demos for filter_key: 100_demos
113
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
114
+ Using 100 subset demos for filter_key: 100_demos
115
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
116
+ Using 100 subset demos for filter_key: 100_demos
117
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
118
+ Using 100 subset demos for filter_key: 100_demos
119
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
120
+ Using 100 subset demos for filter_key: 100_demos
121
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
122
+ Using 100 subset demos for filter_key: 100_demos
123
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
124
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
125
+ Using 100 subset demos for filter_key: 100_demos
126
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
127
+ Using 100 subset demos for filter_key: 100_demos
128
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
129
+ Using 100 subset demos for filter_key: 100_demos
130
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
131
+ Using 100 subset demos for filter_key: 100_demos
132
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
133
+ Using 100 subset demos for filter_key: 100_demos
134
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
135
+ Using 100 subset demos for filter_key: 100_demos
136
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
137
+ Using 100 subset demos for filter_key: 100_demos
138
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
139
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
140
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
141
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
142
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
143
+ 0.75517122 0.7973985 ]
144
+ Loaded 26 datasets
145
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
146
+ Tune backbone vision tower: True
147
+ Tune backbone LLM: True
148
+ Tune action head projector: False
149
+ Tune action head DiT: False
150
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
151
+
152
+ ==================================================
153
+ GR00T FINE-TUNING CONFIGURATION:
154
+ ==================================================
155
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
156
+ dataset_soup: None
157
+ output_dir: /tmp/gr00t
158
+ output_root: None
159
+ data_config: panda_omron
160
+ batch_size: 32
161
+ max_steps: 300000
162
+ num_gpus: 2
163
+ save_steps: 20000
164
+ run_name: None
165
+ save_total_limit: 100
166
+ seed: 42
167
+ base_model_path: nvidia/GR00T-N1.5-3B
168
+ tune_llm: False
169
+ tune_visual: False
170
+ tune_projector: True
171
+ tune_diffusion_model: True
172
+ resume: False
173
+ learning_rate: 3e-05
174
+ weight_decay: 1e-05
175
+ warmup_ratio: 0.05
176
+ lora_rank: 0
177
+ lora_alpha: 16
178
+ lora_dropout: 0.1
179
+ lora_full_model: False
180
+ dataloader_num_workers: 8
181
+ report_to: wandb
182
+ embodiment_tag: new_embodiment
183
+ video_backend: opencv
184
+ balance_dataset_weights: True
185
+ balance_trajectory_weights: True
186
+ ds_weights_alpha: 0.4
187
+ ==================================================
188
+
189
+ Using 2 GPUs
190
+
191
+ ================================================================================
192
+ Starting sweep branch: default
193
+ Sweep vars: {}
194
+ ================================================================================
195
+
196
+ --------------------------------------------------------------------------------
197
+ Running phase 1: phase2_rkd_a_vlm_only
198
+ Policy type: groot_rkd_v2_raw
199
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
200
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
201
+ Trainable preset: freeze_processing_line
202
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
203
+ --------------------------------------------------------------------------------
204
+
205
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
206
+ Using 100 subset demos for filter_key: 100_demos
207
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
208
+ self.statistics[key] = torch.tensor(value)
209
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
210
+ Using 100 subset demos for filter_key: 100_demos
211
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
212
+ Using 100 subset demos for filter_key: 100_demos
213
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
214
+ Using 100 subset demos for filter_key: 100_demos
215
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
216
+ Using 100 subset demos for filter_key: 100_demos
217
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
218
+ Using 100 subset demos for filter_key: 100_demos
219
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
220
+ Using 100 subset demos for filter_key: 100_demos
221
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
222
+ Using 100 subset demos for filter_key: 100_demos
223
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
224
+ Using 100 subset demos for filter_key: 100_demos
225
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
226
+ Using 100 subset demos for filter_key: 100_demos
227
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
228
+ Using 100 subset demos for filter_key: 100_demos
229
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
230
+ Using 100 subset demos for filter_key: 100_demos
231
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
232
+ Using 100 subset demos for filter_key: 100_demos
233
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
234
+ Using 100 subset demos for filter_key: 100_demos
235
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
236
+ Using 100 subset demos for filter_key: 100_demos
237
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
238
+ Using 100 subset demos for filter_key: 100_demos
239
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
240
+ Using 100 subset demos for filter_key: 100_demos
241
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
242
+ Using 100 subset demos for filter_key: 100_demos
243
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
244
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
245
+ Using 100 subset demos for filter_key: 100_demos
246
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
247
+ Using 100 subset demos for filter_key: 100_demos
248
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
249
+ Using 100 subset demos for filter_key: 100_demos
250
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
251
+ Using 100 subset demos for filter_key: 100_demos
252
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
253
+ Using 100 subset demos for filter_key: 100_demos
254
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
255
+ Using 100 subset demos for filter_key: 100_demos
256
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
257
+ Using 100 subset demos for filter_key: 100_demos
258
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
259
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
260
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
261
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
262
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
263
+ 0.75517122 0.7973985 ]
264
+ Loaded 26 datasets
265
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
266
+ Tune backbone vision tower: True
267
+ Tune backbone LLM: True
268
+ Tune action head projector: False
269
+ Tune action head DiT: False
270
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
271
+ Tune backbone llm: False
272
+ Tune backbone visual: True
273
+ Total number of DiT parameters: 550386688
274
+ Tune backbone llm: False
275
+ Tune backbone visual: True
276
+ Total number of DiT parameters: 550386688
277
+ Total number of SelfAttentionTransformer parameters: 201433088
278
+ Tune action head projector: True
279
+ Tune action head diffusion model: True
280
+
281
+ Tune backbone llm: True
282
+ Tune backbone visual: True
283
+ Tune action head projector: False
284
+ Tune action head diffusion model: False
285
+ Action head trainable parameter: future_tokens.weight
286
+ Action head trainable parameter: vlln.weight
287
+ Action head trainable parameter: vlln.bias
288
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
289
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
290
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
291
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
292
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
293
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
294
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
295
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
296
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
297
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
298
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
299
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
300
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
301
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
302
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
303
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
304
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
305
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
306
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
307
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
308
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
309
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
310
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
311
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
312
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
313
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
314
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
315
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
316
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
317
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
318
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
319
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
320
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
321
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
322
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
323
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
324
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
325
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
326
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
327
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
328
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
329
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
330
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
331
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
332
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
333
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
334
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
335
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
336
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
337
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
338
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
339
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
340
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
341
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
342
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
343
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
344
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
345
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
346
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
347
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
348
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
349
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
350
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
351
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
352
+ Applied trainable preset: freeze_processing_line
353
+ Trainable parameter tensors after preset: 585
354
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
355
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
356
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
357
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
358
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
359
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
360
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
361
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
362
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
363
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
364
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
365
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
366
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
367
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
368
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
369
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
370
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
371
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
372
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
373
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
374
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
375
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
376
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
377
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
378
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
379
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
380
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
381
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
382
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
383
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
384
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
385
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
386
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
387
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
388
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
389
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
390
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
391
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
392
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
393
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
394
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
395
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
396
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
397
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
398
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
399
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
400
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
401
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
402
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
403
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
404
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
405
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
406
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
407
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
408
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
409
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
410
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
411
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
412
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
413
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
414
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
415
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
416
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
417
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
418
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
419
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
420
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
421
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
422
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
423
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
424
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
425
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
426
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
427
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
428
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
429
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
430
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
431
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
432
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
433
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
434
+ ... 505 more
435
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
436
+ Total number of SelfAttentionTransformer parameters: 201433088
437
+ Tune action head projector: True
438
+ Tune action head diffusion model: True
439
+
440
+ Tune backbone llm: True
441
+ Tune backbone visual: True
442
+ Tune action head projector: False
443
+ Tune action head diffusion model: False
444
+ Action head trainable parameter: future_tokens.weight
445
+ Action head trainable parameter: vlln.weight
446
+ Action head trainable parameter: vlln.bias
447
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
448
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
449
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
450
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
451
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
452
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
453
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
454
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
455
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
456
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
457
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
458
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
459
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
460
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
461
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
462
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
463
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
464
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
465
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
466
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
467
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
468
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
469
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
470
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
471
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
472
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
473
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
474
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
475
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
476
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
477
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
478
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
479
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
480
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
481
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
482
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
483
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
484
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
485
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
486
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
487
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
488
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
489
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
490
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
491
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
492
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
493
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
494
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
495
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
496
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
497
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
498
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
499
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
500
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
501
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
502
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
503
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
504
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
505
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
506
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
507
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
508
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
509
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
510
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
511
+ Applied trainable preset: freeze_processing_line
512
+ Trainable parameter tensors after preset: 585
513
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
514
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
515
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
516
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
517
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
518
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
519
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
520
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
521
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
522
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
523
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
524
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
525
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
526
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
527
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
528
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
529
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
530
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
531
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
532
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
533
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
534
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
535
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
536
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
537
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
538
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
539
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
540
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
541
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
542
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
543
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
544
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
545
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
546
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
547
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
548
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
549
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
550
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
551
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
552
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
553
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
554
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
555
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
556
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
557
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
558
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
559
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
560
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
561
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
562
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
563
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
564
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
565
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
566
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
567
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
568
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
569
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
570
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
571
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
572
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
573
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
574
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
575
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
576
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
577
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
578
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
579
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
580
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
581
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
582
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
583
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
584
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
585
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
586
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
587
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
588
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
589
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
590
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
591
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
592
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
593
+ ... 505 more
594
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
595
+ Run name: default_phase2
596
+ Run name: default_phase2
597
+ train dataloader length: 6873
598
+ train dataset length: 439854
599
+ GPU memory before training: 7.076685905456543 GB
600
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
601
+ train dataloader length: 6873
602
+ train dataset length: 439854
603
+ GPU memory before training: 7.076685905456543 GB
604
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
605
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
606
+ wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
607
+ wandb: setting up run e2phs2yb
608
+ wandb: Tracking run with wandb version 0.25.0
609
+ wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_141733-e2phs2yb
610
+ wandb: Run `wandb offline` to turn off syncing.
611
+ wandb: Syncing run default_phase2
612
+ wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
613
+ wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/e2phs2yb
614
+
615
  0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
616
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
617
+ [rank1]: run_yaml_experiment(
618
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
619
+ [rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
620
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
621
+ [rank1]: experiment.train()
622
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
623
+ [rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
624
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
625
+ [rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
626
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
627
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
628
+ [rank1]: return inner_training_loop(
629
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
630
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
631
+ [rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
632
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
633
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
634
+ [rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
635
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
636
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
637
+ [rank1]: outputs = model(inputs)
638
+ [rank1]: ^^^^^^^^^^^^^
639
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
640
+ [rank1]: return self._call_impl(*args, **kwargs)
641
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
642
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
643
+ [rank1]: return forward_call(*args, **kwargs)
644
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
645
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
646
+ [rank1]: else self._run_ddp_forward(*inputs, **kwargs)
647
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
648
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
649
+ [rank1]: return self.module(*inputs, **kwargs) # type: ignore[index]
650
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
651
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
652
+ [rank1]: return self._call_impl(*args, **kwargs)
653
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
654
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
655
+ [rank1]: return forward_call(*args, **kwargs)
656
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
657
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
658
+ [rank1]: return model_forward(*args, **kwargs)
659
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
660
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
661
+ [rank1]: return convert_to_fp32(self.model_forward(*args, **kwargs))
662
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
663
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
664
+ [rank1]: return func(*args, **kwargs)
665
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^
666
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
667
+ [rank1]: backbone_outputs = self.backbone(backbone_inputs)
668
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
669
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
670
+ [rank1]: return self._call_impl(*args, **kwargs)
671
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
672
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
673
+ [rank1]: return forward_call(*args, **kwargs)
674
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
675
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
676
+ [rank1]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
677
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
678
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
679
+ [rank1]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
680
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
681
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
682
+ [rank1]: return self._call_impl(*args, **kwargs)
683
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
684
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
685
+ [rank1]: return forward_call(*args, **kwargs)
686
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
687
+ [rank1]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
688
+ [rank1]: outputs = self.language_model(
689
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
690
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
691
+ [rank1]: return self._call_impl(*args, **kwargs)
692
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
693
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
694
+ [rank1]: return forward_call(*args, **kwargs)
695
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
696
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
697
+ [rank1]: output = func(self, *args, **kwargs)
698
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
699
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
700
+ [rank1]: return func(*args, **kwargs)
701
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^
702
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 866, in forward
703
+ [rank1]: logits = self.lm_head(hidden_states[:, slice_indices, :])
704
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
705
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
706
+ [rank1]: return self._call_impl(*args, **kwargs)
707
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
708
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
709
+ [rank1]: return forward_call(*args, **kwargs)
710
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
711
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
712
+ [rank1]: return F.linear(input, self.weight, self.bias)
713
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
714
+ [rank1]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 7.55 GiB. GPU 1 has a total capacity of 79.25 GiB of which 2.73 GiB is free. Including non-PyTorch memory, this process has 76.50 GiB memory in use. Of the allocated memory 74.82 GiB is allocated by PyTorch, and 1.05 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
715
+ wandb: updating run metadata
716
+ wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
717
+ wandb: uploading config.yaml
718
+ wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/e2phs2yb
719
+ wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
720
+ wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
721
+ wandb: Find logs at: ./wandb/run-20260615_141733-e2phs2yb/logs
722
+ Traceback (most recent call last):
723
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
724
+ run_yaml_experiment(
725
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
726
+ main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
727
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
728
+ experiment.train()
729
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
730
+ self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
731
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
732
+ return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
733
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
734
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
735
+ return inner_training_loop(
736
+ ^^^^^^^^^^^^^^^^^^^^
737
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
738
+ tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
739
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
740
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
741
+ loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
742
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
743
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
744
+ outputs = model(inputs)
745
+ ^^^^^^^^^^^^^
746
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
747
+ return self._call_impl(*args, **kwargs)
748
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
749
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
750
+ return forward_call(*args, **kwargs)
751
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
752
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
753
+ else self._run_ddp_forward(*inputs, **kwargs)
754
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
755
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
756
+ return self.module(*inputs, **kwargs) # type: ignore[index]
757
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
758
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
759
+ return self._call_impl(*args, **kwargs)
760
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
761
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
762
+ return forward_call(*args, **kwargs)
763
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
764
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
765
+ return model_forward(*args, **kwargs)
766
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
767
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
768
+ return convert_to_fp32(self.model_forward(*args, **kwargs))
769
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
770
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
771
+ return func(*args, **kwargs)
772
+ ^^^^^^^^^^^^^^^^^^^^^
773
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
774
+ backbone_outputs = self.backbone(backbone_inputs)
775
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
776
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
777
+ return self._call_impl(*args, **kwargs)
778
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
779
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
780
+ return forward_call(*args, **kwargs)
781
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
782
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
783
+ eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
784
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
785
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
786
+ eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
787
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
788
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
789
+ return self._call_impl(*args, **kwargs)
790
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
791
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
792
+ return forward_call(*args, **kwargs)
793
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
794
+ File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
795
+ outputs = self.language_model(
796
+ ^^^^^^^^^^^^^^^^^^^^
797
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
798
+ return self._call_impl(*args, **kwargs)
799
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
800
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
801
+ return forward_call(*args, **kwargs)
802
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
803
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
804
+ output = func(self, *args, **kwargs)
805
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
806
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
807
+ return func(*args, **kwargs)
808
+ ^^^^^^^^^^^^^^^^^^^^^
809
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 866, in forward
810
+ logits = self.lm_head(hidden_states[:, slice_indices, :])
811
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
812
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
813
+ return self._call_impl(*args, **kwargs)
814
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
815
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
816
+ return forward_call(*args, **kwargs)
817
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
818
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
819
+ return F.linear(input, self.weight, self.bias)
820
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
821
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 7.55 GiB. GPU 0 has a total capacity of 79.25 GiB of which 2.73 GiB is free. Including non-PyTorch memory, this process has 76.50 GiB memory in use. Of the allocated memory 74.82 GiB is allocated by PyTorch, and 1.05 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
822
+ [rank0]: Traceback (most recent call last):
823
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
824
+ [rank0]: run_yaml_experiment(
825
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
826
+ [rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
827
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
828
+ [rank0]: experiment.train()
829
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
830
+ [rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
831
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
832
+ [rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
833
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
834
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
835
+ [rank0]: return inner_training_loop(
836
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
837
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
838
+ [rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
839
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
840
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
841
+ [rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
842
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
843
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
844
+ [rank0]: outputs = model(inputs)
845
+ [rank0]: ^^^^^^^^^^^^^
846
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
847
+ [rank0]: return self._call_impl(*args, **kwargs)
848
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
849
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
850
+ [rank0]: return forward_call(*args, **kwargs)
851
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
852
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
853
+ [rank0]: else self._run_ddp_forward(*inputs, **kwargs)
854
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
855
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
856
+ [rank0]: return self.module(*inputs, **kwargs) # type: ignore[index]
857
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
858
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
859
+ [rank0]: return self._call_impl(*args, **kwargs)
860
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
861
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
862
+ [rank0]: return forward_call(*args, **kwargs)
863
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
864
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
865
+ [rank0]: return model_forward(*args, **kwargs)
866
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
867
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
868
+ [rank0]: return convert_to_fp32(self.model_forward(*args, **kwargs))
869
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
870
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
871
+ [rank0]: return func(*args, **kwargs)
872
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^
873
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
874
+ [rank0]: backbone_outputs = self.backbone(backbone_inputs)
875
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
876
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
877
+ [rank0]: return self._call_impl(*args, **kwargs)
878
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
879
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
880
+ [rank0]: return forward_call(*args, **kwargs)
881
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
882
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
883
+ [rank0]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
884
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
885
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
886
+ [rank0]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
887
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
888
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
889
+ [rank0]: return self._call_impl(*args, **kwargs)
890
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
891
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
892
+ [rank0]: return forward_call(*args, **kwargs)
893
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
894
+ [rank0]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
895
+ [rank0]: outputs = self.language_model(
896
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
897
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
898
+ [rank0]: return self._call_impl(*args, **kwargs)
899
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
900
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
901
+ [rank0]: return forward_call(*args, **kwargs)
902
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
903
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
904
+ [rank0]: output = func(self, *args, **kwargs)
905
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
906
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
907
+ [rank0]: return func(*args, **kwargs)
908
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^
909
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 866, in forward
910
+ [rank0]: logits = self.lm_head(hidden_states[:, slice_indices, :])
911
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
912
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
913
+ [rank0]: return self._call_impl(*args, **kwargs)
914
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
915
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
916
+ [rank0]: return forward_call(*args, **kwargs)
917
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
918
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
919
+ [rank0]: return F.linear(input, self.weight, self.bias)
920
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
921
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 7.55 GiB. GPU 0 has a total capacity of 79.25 GiB of which 2.73 GiB is free. Including non-PyTorch memory, this process has 76.50 GiB memory in use. Of the allocated memory 74.82 GiB is allocated by PyTorch, and 1.05 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
922
+ [rank0]:[W615 14:17:43.385716740 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
923
+ W0615 14:17:43.882000 478197 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 484952 closing signal SIGTERM
924
+ E0615 14:17:44.246000 478197 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 484957) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
925
+ Traceback (most recent call last):
926
+ File "<frozen runpy>", line 198, in _run_module_as_main
927
+ File "<frozen runpy>", line 88, in _run_code
928
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
929
+ main()
930
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
931
+ return f(*args, **kwargs)
932
+ ^^^^^^^^^^^^^^^^^^
933
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
934
+ run(args)
935
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
936
+ elastic_launch(
937
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
938
+ return launch_agent(self._config, self._entrypoint, list(args))
939
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
940
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
941
+ raise ChildFailedError(
942
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
943
+ ============================================================
944
+ /home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
945
+ ------------------------------------------------------------
946
+ Failures:
947
+ <NO_OTHER_FAILURES>
948
+ ------------------------------------------------------------
949
+ Root Cause (first observed failure):
950
+ [0]:
951
+ time : 2026-06-15_14:17:43
952
+ host : worker1
953
+ rank : 1 (local_rank: 1)
954
+ exitcode : 1 (pid: 484957)
955
+ error_file: <N/A>
956
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
957
+ ============================================================
958
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
959
+
960
+ ==================================================
961
+ GR00T FINE-TUNING CONFIGURATION:
962
+ ==================================================
963
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
964
+ dataset_soup: None
965
+ output_dir: /tmp/gr00t
966
+ output_root: None
967
+ data_config: panda_omron
968
+ batch_size: 32
969
+ max_steps: 300000
970
+ num_gpus: 2
971
+ save_steps: 20000
972
+ run_name: None
973
+ save_total_limit: 100
974
+ seed: 42
975
+ base_model_path: nvidia/GR00T-N1.5-3B
976
+ tune_llm: False
977
+ tune_visual: False
978
+ tune_projector: True
979
+ tune_diffusion_model: True
980
+ resume: False
981
+ learning_rate: 3e-05
982
+ weight_decay: 1e-05
983
+ warmup_ratio: 0.05
984
+ lora_rank: 0
985
+ lora_alpha: 16
986
+ lora_dropout: 0.1
987
+ lora_full_model: False
988
+ dataloader_num_workers: 8
989
+ report_to: wandb
990
+ embodiment_tag: new_embodiment
991
+ video_backend: opencv
992
+ balance_dataset_weights: True
993
+ balance_trajectory_weights: True
994
+ ds_weights_alpha: 0.4
995
+ ==================================================
996
+
997
+ Using 2 GPUs
998
+ Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '32', '--num-gpus', '2']
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_single_probe.log ADDED
@@ -0,0 +1,401 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/30000 [00:00<?, ?it/s]wandb: updating run metadata; uploading wandb-metadata.json; uploading requirements.txt
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [robosuite WARNING] No private macro file found! (macros.py:57)
2
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
3
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
4
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
5
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
6
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
7
+ check_for_updates()
8
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
9
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
10
+
11
+ ==================================================
12
+ GR00T FINE-TUNING CONFIGURATION:
13
+ ==================================================
14
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
15
+ dataset_soup: None
16
+ output_dir: /tmp/gr00t
17
+ output_root: None
18
+ data_config: panda_omron
19
+ batch_size: 32
20
+ max_steps: 300000
21
+ num_gpus: 1
22
+ save_steps: 20000
23
+ run_name: None
24
+ save_total_limit: 100
25
+ seed: 42
26
+ base_model_path: nvidia/GR00T-N1.5-3B
27
+ tune_llm: False
28
+ tune_visual: False
29
+ tune_projector: True
30
+ tune_diffusion_model: True
31
+ resume: False
32
+ learning_rate: 3e-05
33
+ weight_decay: 1e-05
34
+ warmup_ratio: 0.05
35
+ lora_rank: 0
36
+ lora_alpha: 16
37
+ lora_dropout: 0.1
38
+ lora_full_model: False
39
+ dataloader_num_workers: 8
40
+ report_to: wandb
41
+ embodiment_tag: new_embodiment
42
+ video_backend: opencv
43
+ balance_dataset_weights: True
44
+ balance_trajectory_weights: True
45
+ ds_weights_alpha: 0.4
46
+ ==================================================
47
+
48
+ Using 1 GPUs
49
+
50
+ ================================================================================
51
+ Starting sweep branch: default
52
+ Sweep vars: {}
53
+ ================================================================================
54
+
55
+ --------------------------------------------------------------------------------
56
+ Running phase 1: phase2_rkd_a_vlm_only
57
+ Policy type: groot_rkd_v2_raw
58
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
59
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
60
+ Trainable preset: freeze_processing_line
61
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
62
+ --------------------------------------------------------------------------------
63
+
64
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
65
+ Using 100 subset demos for filter_key: 100_demos/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
66
+ self.statistics[key] = torch.tensor(value)
67
+
68
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
69
+ Using 100 subset demos for filter_key: 100_demos
70
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
71
+ Using 100 subset demos for filter_key: 100_demos
72
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
73
+ Using 100 subset demos for filter_key: 100_demos
74
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
75
+ Using 100 subset demos for filter_key: 100_demos
76
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
77
+ Using 100 subset demos for filter_key: 100_demos
78
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
79
+ Using 100 subset demos for filter_key: 100_demos
80
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
81
+ Using 100 subset demos for filter_key: 100_demos
82
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
83
+ Using 100 subset demos for filter_key: 100_demos
84
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
85
+ Using 100 subset demos for filter_key: 100_demos
86
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
87
+ Using 100 subset demos for filter_key: 100_demos
88
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
89
+ Using 100 subset demos for filter_key: 100_demos
90
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
91
+ Using 100 subset demos for filter_key: 100_demos
92
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
93
+ Using 100 subset demos for filter_key: 100_demos
94
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
95
+ Using 100 subset demos for filter_key: 100_demos
96
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
97
+ Using 100 subset demos for filter_key: 100_demos
98
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
99
+ Using 100 subset demos for filter_key: 100_demos
100
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
101
+ Using 100 subset demos for filter_key: 100_demos
102
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
103
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
104
+ Using 100 subset demos for filter_key: 100_demos
105
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
106
+ Using 100 subset demos for filter_key: 100_demos
107
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
108
+ Using 100 subset demos for filter_key: 100_demos
109
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
110
+ Using 100 subset demos for filter_key: 100_demos
111
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
112
+ Using 100 subset demos for filter_key: 100_demos
113
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
114
+ Using 100 subset demos for filter_key: 100_demos
115
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
116
+ Using 100 subset demos for filter_key: 100_demos
117
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
118
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
119
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
120
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
121
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
122
+ 0.75517122 0.7973985 ]
123
+ Loaded 26 datasets
124
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
125
+ Tune backbone vision tower: True
126
+ Tune backbone LLM: True
127
+ Tune action head projector: False
128
+ Tune action head DiT: False
129
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
130
+ Tune backbone llm: False
131
+ Tune backbone visual: True
132
+ Total number of DiT parameters: 550386688
133
+ Total number of SelfAttentionTransformer parameters: 201433088
134
+ Tune action head projector: True
135
+ Tune action head diffusion model: True
136
+
137
+ Tune backbone llm: True
138
+ Tune backbone visual: True
139
+ Tune action head projector: False
140
+ Tune action head diffusion model: False
141
+ Action head trainable parameter: future_tokens.weight
142
+ Action head trainable parameter: vlln.weight
143
+ Action head trainable parameter: vlln.bias
144
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
145
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
146
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
147
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
148
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
149
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
150
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
151
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
152
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
153
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
154
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
155
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
156
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
157
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
158
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
159
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
160
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
161
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
162
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
163
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
164
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
165
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
166
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
167
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
168
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
169
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
170
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
171
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
172
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
173
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
174
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
175
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
176
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
177
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
178
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
179
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
180
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
181
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
182
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
183
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
184
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
185
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
186
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
187
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
188
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
189
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
190
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
191
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
192
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
193
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
194
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
195
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
196
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
197
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
198
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
199
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
200
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
201
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
202
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
203
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
204
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
205
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
206
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
207
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
208
+ Applied trainable preset: freeze_processing_line
209
+ Trainable parameter tensors after preset: 584
210
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
211
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
212
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
213
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
214
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
215
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
216
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
217
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
218
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
219
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
220
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
221
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
222
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
223
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
224
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
225
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
226
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
227
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
228
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
229
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
230
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
231
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
232
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
233
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
234
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
235
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
236
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
237
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
238
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
239
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
240
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
241
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
242
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
243
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
244
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
245
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
246
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
247
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
248
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
249
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
250
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
251
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
252
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
253
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
254
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
255
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
256
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
257
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
258
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
259
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
260
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
261
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
262
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
263
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
264
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
265
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
266
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
267
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
268
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
269
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
270
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
271
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
272
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
273
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
274
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
275
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
276
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
277
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
278
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
279
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
280
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
281
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
282
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
283
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
284
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
285
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
286
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
287
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
288
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
289
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
290
+ ... 504 more
291
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
292
+ Run name: default_phase2
293
+ train dataloader length: 13746
294
+ train dataset length: 439854
295
+ GPU memory before training: 7.076685905456543 GB
296
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
297
+ wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
298
+ wandb: setting up run kloq3gk7
299
+ wandb: Tracking run with wandb version 0.25.0
300
+ wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_143415-kloq3gk7
301
+ wandb: Run `wandb offline` to turn off syncing.
302
+ wandb: Syncing run default_phase2
303
+ wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
304
+ wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/kloq3gk7
305
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
306
+
307
  0%| | 0/30000 [00:00<?, ?it/s]wandb: updating run metadata; uploading wandb-metadata.json; uploading requirements.txt
308
+ wandb: uploading wandb-metadata.json; uploading requirements.txt
309
+ wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
310
+ wandb: uploading summary
311
+ wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/kloq3gk7
312
+ wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
313
+ wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
314
+ wandb: Find logs at: ./wandb/run-20260615_143415-kloq3gk7/logs
315
+ Traceback (most recent call last):
316
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1015, in <module>
317
+ run_yaml_experiment(
318
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
319
+ main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
320
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
321
+ experiment.train()
322
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
323
+ self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
324
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
325
+ return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
326
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
327
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
328
+ return inner_training_loop(
329
+ ^^^^^^^^^^^^^^^^^^^^
330
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
331
+ tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
332
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
333
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
334
+ loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
335
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
336
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
337
+ outputs = model(inputs)
338
+ ^^^^^^^^^^^^^
339
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
340
+ return self._call_impl(*args, **kwargs)
341
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
342
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
343
+ return forward_call(*args, **kwargs)
344
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
345
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
346
+ return model_forward(*args, **kwargs)
347
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
348
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
349
+ return convert_to_fp32(self.model_forward(*args, **kwargs))
350
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
351
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
352
+ return func(*args, **kwargs)
353
+ ^^^^^^^^^^^^^^^^^^^^^
354
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
355
+ backbone_outputs = self.backbone(backbone_inputs)
356
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
357
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
358
+ return self._call_impl(*args, **kwargs)
359
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
360
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
361
+ return forward_call(*args, **kwargs)
362
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
363
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
364
+ eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
365
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
366
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
367
+ eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
368
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
369
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
370
+ return self._call_impl(*args, **kwargs)
371
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
372
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
373
+ return forward_call(*args, **kwargs)
374
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
375
+ File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
376
+ outputs = self.language_model(
377
+ ^^^^^^^^^^^^^^^^^^^^
378
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
379
+ return self._call_impl(*args, **kwargs)
380
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
381
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
382
+ return forward_call(*args, **kwargs)
383
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
384
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
385
+ output = func(self, *args, **kwargs)
386
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
387
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
388
+ return func(*args, **kwargs)
389
+ ^^^^^^^^^^^^^^^^^^^^^
390
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 866, in forward
391
+ logits = self.lm_head(hidden_states[:, slice_indices, :])
392
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
393
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
394
+ return self._call_impl(*args, **kwargs)
395
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
396
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
397
+ return forward_call(*args, **kwargs)
398
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
399
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
400
+ return F.linear(input, self.weight, self.bias)
401
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
402
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 7.55 GiB. GPU 0 has a total capacity of 79.25 GiB of which 6.50 GiB is free. Including non-PyTorch memory, this process has 72.73 GiB memory in use. Of the allocated memory 71.75 GiB is allocated by PyTorch, and 492.34 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs64_gpu2_3.log ADDED
@@ -0,0 +1,1095 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [robosuite WARNING] No private macro file found! (macros.py:57)
2
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
3
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
4
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
5
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
6
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
7
+ check_for_updates()
8
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
9
+
10
+ *****************************************
11
+ Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
12
+ *****************************************
13
+ [robosuite WARNING] No private macro file found! (macros.py:57)
14
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
15
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
16
+ [robosuite WARNING] No private macro file found! (macros.py:57)
17
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
18
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
19
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
20
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
21
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
22
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
23
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
24
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
25
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
26
+ check_for_updates()
27
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
28
+ check_for_updates()
29
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
30
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
31
+
32
+ ==================================================
33
+ GR00T FINE-TUNING CONFIGURATION:
34
+ ==================================================
35
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
36
+ dataset_soup: None
37
+ output_dir: /tmp/gr00t
38
+ output_root: None
39
+ data_config: panda_omron
40
+ batch_size: 64
41
+ max_steps: 300000
42
+ num_gpus: 2
43
+ save_steps: 20000
44
+ run_name: None
45
+ save_total_limit: 100
46
+ seed: 42
47
+ base_model_path: nvidia/GR00T-N1.5-3B
48
+ tune_llm: False
49
+ tune_visual: False
50
+ tune_projector: True
51
+ tune_diffusion_model: True
52
+ resume: False
53
+ learning_rate: 3e-05
54
+ weight_decay: 1e-05
55
+ warmup_ratio: 0.05
56
+ lora_rank: 0
57
+ lora_alpha: 16
58
+ lora_dropout: 0.1
59
+ lora_full_model: False
60
+ dataloader_num_workers: 8
61
+ report_to: wandb
62
+ embodiment_tag: new_embodiment
63
+ video_backend: opencv
64
+ balance_dataset_weights: True
65
+ balance_trajectory_weights: True
66
+ ds_weights_alpha: 0.4
67
+ ==================================================
68
+
69
+ Using 2 GPUs
70
+
71
+ ================================================================================
72
+ Starting sweep branch: default
73
+ Sweep vars: {}
74
+ ================================================================================
75
+
76
+ --------------------------------------------------------------------------------
77
+ Running phase 1: phase2_rkd_a_vlm_only
78
+ Policy type: groot_rkd_v2_raw
79
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
80
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
81
+ Trainable preset: freeze_processing_line
82
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
83
+ --------------------------------------------------------------------------------
84
+
85
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
86
+ Using 100 subset demos for filter_key: 100_demos
87
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
88
+ self.statistics[key] = torch.tensor(value)
89
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
90
+ Using 100 subset demos for filter_key: 100_demos
91
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
92
+ Using 100 subset demos for filter_key: 100_demos
93
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
94
+ Using 100 subset demos for filter_key: 100_demos
95
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
96
+ Using 100 subset demos for filter_key: 100_demos
97
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
98
+ Using 100 subset demos for filter_key: 100_demos
99
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
100
+ Using 100 subset demos for filter_key: 100_demos
101
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
102
+ Using 100 subset demos for filter_key: 100_demos
103
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
104
+ Using 100 subset demos for filter_key: 100_demos
105
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
106
+ Using 100 subset demos for filter_key: 100_demos
107
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
108
+ Using 100 subset demos for filter_key: 100_demos
109
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
110
+ Using 100 subset demos for filter_key: 100_demos
111
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
112
+ Using 100 subset demos for filter_key: 100_demos
113
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
114
+ Using 100 subset demos for filter_key: 100_demos
115
+
116
+ ==================================================
117
+ GR00T FINE-TUNING CONFIGURATION:
118
+ ==================================================
119
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
120
+ dataset_soup: None
121
+ output_dir: /tmp/gr00t
122
+ output_root: None
123
+ data_config: panda_omron
124
+ batch_size: 64
125
+ max_steps: 300000
126
+ num_gpus: 2
127
+ save_steps: 20000
128
+ run_name: None
129
+ save_total_limit: 100Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
130
+ seed: 42
131
+
132
+ base_model_path: nvidia/GR00T-N1.5-3B
133
+ tune_llm: False
134
+ tune_visual: False
135
+ tune_projector: True
136
+ tune_diffusion_model: True
137
+ resume: False
138
+ learning_rate: 3e-05
139
+ weight_decay: 1e-05
140
+ warmup_ratio: 0.05
141
+ lora_rank: 0
142
+ lora_alpha: 16
143
+ lora_dropout: 0.1
144
+ lora_full_model: False
145
+ dataloader_num_workers: 8
146
+ report_to: wandb
147
+ embodiment_tag: new_embodiment
148
+ video_backend: opencv
149
+ balance_dataset_weights: True
150
+ balance_trajectory_weights: True
151
+ ds_weights_alpha: 0.4
152
+ ==================================================
153
+
154
+ Using 100 subset demos for filter_key: 100_demos
155
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
156
+ Using 100 subset demos for filter_key: 100_demos
157
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
158
+ Using 100 subset demos for filter_key: 100_demos
159
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
160
+ Using 100 subset demos for filter_key: 100_demos
161
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
162
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
163
+ Using 100 subset demos for filter_key: 100_demos
164
+ Using 2 GPUs
165
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
166
+ Using 100 subset demos for filter_key: 100_demos
167
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
168
+
169
+ ================================================================================
170
+ Starting sweep branch: default
171
+ Sweep vars: {}
172
+ ================================================================================
173
+
174
+ --------------------------------------------------------------------------------
175
+ Running phase 1: phase2_rkd_a_vlm_only
176
+ Policy type: groot_rkd_v2_raw
177
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
178
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
179
+ Trainable preset: freeze_processing_line
180
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
181
+ --------------------------------------------------------------------------------
182
+
183
+ Using 100 subset demos for filter_key: 100_demos
184
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
185
+ Using 100 subset demos for filter_key: 100_demos
186
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
187
+ Using 100 subset demos for filter_key: 100_demos
188
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
189
+ self.statistics[key] = torch.tensor(value)
190
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
191
+ Using 100 subset demos for filter_key: 100_demos
192
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
193
+ Using 100 subset demos for filter_key: 100_demos
194
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
195
+ Using 100 subset demos for filter_key: 100_demos
196
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
197
+ Using 100 subset demos for filter_key: 100_demos
198
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
199
+ Using 100 subset demos for filter_key: 100_demos
200
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
201
+ Using 100 subset demos for filter_key: 100_demos
202
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
203
+ Using 100 subset demos for filter_key: 100_demos
204
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
205
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
206
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
207
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
208
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
209
+ 0.75517122 0.7973985 ]
210
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
211
+ Using 100 subset demos for filter_key: 100_demos
212
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
213
+ Using 100 subset demos for filter_key: 100_demos
214
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
215
+ Using 100 subset demos for filter_key: 100_demos
216
+ Loaded 26 datasets
217
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
218
+ Using 100 subset demos for filter_key: 100_demos
219
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
220
+ Using 100 subset demos for filter_key: 100_demos
221
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
222
+ Using 100 subset demos for filter_key: 100_demos
223
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
224
+ Using 100 subset demos for filter_key: 100_demos
225
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
226
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
227
+ Tune backbone vision tower: True
228
+ Tune backbone LLM: True
229
+ Tune action head projector: False
230
+ Tune action head DiT: False
231
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
232
+ Using 100 subset demos for filter_key: 100_demos
233
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
234
+ Using 100 subset demos for filter_key: 100_demos
235
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
236
+ Using 100 subset demos for filter_key: 100_demos
237
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
238
+ Using 100 subset demos for filter_key: 100_demos
239
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
240
+ Using 100 subset demos for filter_key: 100_demos
241
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
242
+ Using 100 subset demos for filter_key: 100_demos
243
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
244
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
245
+ Using 100 subset demos for filter_key: 100_demos
246
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
247
+ Using 100 subset demos for filter_key: 100_demos
248
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
249
+ Using 100 subset demos for filter_key: 100_demos
250
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
251
+ Using 100 subset demos for filter_key: 100_demos
252
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
253
+ Using 100 subset demos for filter_key: 100_demos
254
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
255
+ Using 100 subset demos for filter_key: 100_demos
256
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
257
+ Using 100 subset demos for filter_key: 100_demos
258
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
259
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
260
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
261
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
262
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
263
+ 0.75517122 0.7973985 ]
264
+ Loaded 26 datasets
265
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
266
+ Tune backbone vision tower: True
267
+ Tune backbone LLM: True
268
+ Tune action head projector: False
269
+ Tune action head DiT: False
270
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
271
+ Tune backbone llm: False
272
+ Tune backbone visual: True
273
+ Total number of DiT parameters: 550386688
274
+ Tune backbone llm: False
275
+ Tune backbone visual: True
276
+ Total number of DiT parameters: 550386688
277
+ Total number of SelfAttentionTransformer parameters: 201433088
278
+ Tune action head projector: True
279
+ Tune action head diffusion model: True
280
+
281
+ Tune backbone llm: True
282
+ Tune backbone visual: True
283
+ Tune action head projector: False
284
+ Tune action head diffusion model: False
285
+ Action head trainable parameter: future_tokens.weight
286
+ Action head trainable parameter: vlln.weight
287
+ Action head trainable parameter: vlln.bias
288
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
289
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
290
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
291
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
292
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
293
+ Total number of SelfAttentionTransformer parameters: Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
294
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
295
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
296
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
297
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
298
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
299
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias201433088
300
+
301
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
302
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
303
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
304
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
305
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
306
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
307
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
308
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
309
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
310
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
311
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
312
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
313
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
314
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
315
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
316
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
317
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
318
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
319
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
320
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
321
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
322
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
323
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
324
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
325
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
326
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
327
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
328
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
329
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
330
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
331
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
332
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
333
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
334
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
335
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
336
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
337
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
338
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
339
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
340
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
341
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
342
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
343
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
344
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
345
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
346
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
347
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
348
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
349
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
350
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
351
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
352
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
353
+ Tune action head projector: True
354
+ Tune action head diffusion model: True
355
+ Applied trainable preset: freeze_processing_line
356
+ Trainable parameter tensors after preset: 585
357
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
358
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
359
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
360
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
361
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
362
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
363
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
364
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
365
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
366
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
367
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
368
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
369
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
370
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
371
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
372
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
373
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
374
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
375
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
376
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
377
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
378
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
379
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
380
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
381
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
382
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
383
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
384
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
385
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
386
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
387
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
388
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
389
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
390
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
391
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
392
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
393
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
394
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
395
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
396
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
397
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
398
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
399
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
400
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
401
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
402
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
403
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
404
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
405
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
406
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
407
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
408
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
409
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
410
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
411
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
412
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
413
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
414
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
415
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
416
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
417
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
418
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
419
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
420
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
421
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
422
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
423
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
424
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
425
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
426
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
427
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
428
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
429
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
430
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
431
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
432
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
433
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
434
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
435
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
436
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
437
+ ... 505 more
438
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
439
+
440
+
441
+ Tune backbone llm: True
442
+ Tune backbone visual: True
443
+ Tune action head projector: False
444
+ Tune action head diffusion model: False
445
+ Action head trainable parameter: future_tokens.weight
446
+ Action head trainable parameter: vlln.weight
447
+ Action head trainable parameter: vlln.bias
448
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
449
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
450
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
451
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
452
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
453
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
454
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
455
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
456
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
457
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
458
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
459
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
460
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
461
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
462
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
463
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
464
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
465
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
466
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
467
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
468
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
469
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
470
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
471
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
472
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
473
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
474
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
475
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
476
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
477
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
478
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
479
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
480
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
481
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
482
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
483
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
484
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
485
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
486
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
487
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
488
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
489
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
490
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
491
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
492
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
493
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
494
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
495
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
496
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
497
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
498
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
499
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
500
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
501
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
502
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
503
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
504
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
505
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
506
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
507
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
508
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
509
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
510
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
511
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
512
+ Applied trainable preset: freeze_processing_line
513
+ Trainable parameter tensors after preset: 585
514
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
515
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
516
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
517
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
518
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
519
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
520
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
521
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
522
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
523
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
524
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
525
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
526
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
527
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
528
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
529
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
530
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
531
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
532
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
533
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
534
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
535
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
536
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
537
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
538
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
539
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
540
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
541
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
542
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
543
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
544
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
545
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
546
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
547
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
548
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
549
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
550
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
551
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
552
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
553
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
554
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
555
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
556
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
557
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
558
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
559
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
560
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
561
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
562
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
563
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
564
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
565
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
566
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
567
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
568
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
569
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
570
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
571
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
572
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
573
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
574
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
575
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
576
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
577
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
578
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
579
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
580
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
581
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
582
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
583
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
584
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
585
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
586
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
587
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
588
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
589
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
590
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
591
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
592
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
593
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
594
+ ... 505 more
595
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
596
+ Run name: default_phase2
597
+ train dataloader length: 3437
598
+ train dataset length: 439854
599
+ GPU memory before training: 7.076685905456543 GB
600
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
601
+ train dataloader length: 3437
602
+ train dataset length: 439854
603
+ GPU memory before training: 7.076685905456543 GB
604
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
605
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
606
+ wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
607
+ wandb: setting up run 3uyg5mqq
608
+ wandb: Tracking run with wandb version 0.25.0
609
+ wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_141245-3uyg5mqq
610
+ wandb: Run `wandb offline` to turn off syncing.
611
+ wandb: Syncing run default_phase2
612
+ wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
613
+ wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/3uyg5mqq
614
+
615
  0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
616
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
617
+ [rank1]: run_yaml_experiment(
618
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
619
+ [rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
620
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
621
+ [rank1]: experiment.train()
622
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
623
+ [rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
624
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
625
+ [rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
626
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
627
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
628
+ [rank1]: return inner_training_loop(
629
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
630
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
631
+ [rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
632
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
633
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
634
+ [rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
635
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
636
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
637
+ [rank1]: outputs = model(inputs)
638
+ [rank1]: ^^^^^^^^^^^^^
639
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
640
+ [rank1]: return self._call_impl(*args, **kwargs)
641
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
642
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
643
+ [rank1]: return forward_call(*args, **kwargs)
644
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
645
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
646
+ [rank1]: else self._run_ddp_forward(*inputs, **kwargs)
647
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
648
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
649
+ [rank1]: return self.module(*inputs, **kwargs) # type: ignore[index]
650
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
651
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
652
+ [rank1]: return self._call_impl(*args, **kwargs)
653
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
654
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
655
+ [rank1]: return forward_call(*args, **kwargs)
656
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
657
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
658
+ [rank1]: return model_forward(*args, **kwargs)
659
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
660
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
661
+ [rank1]: return convert_to_fp32(self.model_forward(*args, **kwargs))
662
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
663
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
664
+ [rank1]: return func(*args, **kwargs)
665
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^
666
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
667
+ [rank1]: backbone_outputs = self.backbone(backbone_inputs)
668
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
669
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
670
+ [rank1]: return self._call_impl(*args, **kwargs)
671
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
672
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
673
+ [rank1]: return forward_call(*args, **kwargs)
674
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
675
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
676
+ [rank1]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
677
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
678
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
679
+ [rank1]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
680
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
681
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
682
+ [rank1]: return self._call_impl(*args, **kwargs)
683
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
684
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
685
+ [rank1]: return forward_call(*args, **kwargs)
686
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
687
+ [rank1]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
688
+ [rank1]: outputs = self.language_model(
689
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
690
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
691
+ [rank1]: return self._call_impl(*args, **kwargs)
692
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
693
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
694
+ [rank1]: return forward_call(*args, **kwargs)
695
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
696
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
697
+ [rank1]: output = func(self, *args, **kwargs)
698
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
699
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
700
+ [rank1]: return func(*args, **kwargs)
701
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^
702
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
703
+ [rank1]: outputs: BaseModelOutputWithPast = self.model(
704
+ [rank1]: ^^^^^^^^^^^
705
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
706
+ [rank1]: return self._call_impl(*args, **kwargs)
707
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
708
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
709
+ [rank1]: return forward_call(*args, **kwargs)
710
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
711
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
712
+ [rank1]: output = func(self, *args, **kwargs)
713
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
714
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
715
+ [rank1]: layer_outputs = decoder_layer(
716
+ [rank1]: ^^^^^^^^^^^^^^
717
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
718
+ [rank1]: return self._call_impl(*args, **kwargs)
719
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
720
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
721
+ [rank1]: return forward_call(*args, **kwargs)
722
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
723
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
724
+ [rank1]: hidden_states = self.mlp(hidden_states)
725
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^
726
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
727
+ [rank1]: return self._call_impl(*args, **kwargs)
728
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
729
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
730
+ [rank1]: return forward_call(*args, **kwargs)
731
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
732
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
733
+ [rank1]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
734
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
735
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
736
+ [rank1]: return self._call_impl(*args, **kwargs)
737
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
738
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
739
+ [rank1]: return forward_call(*args, **kwargs)
740
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
741
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/activation.py", line 432, in forward
742
+ [rank1]: return F.silu(input, inplace=self.inplace)
743
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
744
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/functional.py", line 2380, in silu
745
+ [rank1]: return torch._C._nn.silu(input)
746
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^
747
+ [rank1]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 1 has a total capacity of 79.25 GiB of which 469.94 MiB is free. Including non-PyTorch memory, this process has 78.77 GiB memory in use. Of the allocated memory 76.81 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
748
+ wandb: updating run metadata
749
+ wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
750
+ wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/3uyg5mqq
751
+ wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
752
+ wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
753
+ wandb: Find logs at: ./wandb/run-20260615_141245-3uyg5mqq/logs
754
+ Traceback (most recent call last):
755
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
756
+ run_yaml_experiment(
757
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
758
+ main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
759
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
760
+ experiment.train()
761
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
762
+ self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
763
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
764
+ return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
765
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
766
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
767
+ return inner_training_loop(
768
+ ^^^^^^^^^^^^^^^^^^^^
769
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
770
+ tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
771
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
772
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
773
+ loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
774
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
775
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
776
+ outputs = model(inputs)
777
+ ^^^^^^^^^^^^^
778
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
779
+ return self._call_impl(*args, **kwargs)
780
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
781
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
782
+ return forward_call(*args, **kwargs)
783
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
784
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
785
+ else self._run_ddp_forward(*inputs, **kwargs)
786
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
787
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
788
+ return self.module(*inputs, **kwargs) # type: ignore[index]
789
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
790
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
791
+ return self._call_impl(*args, **kwargs)
792
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
793
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
794
+ return forward_call(*args, **kwargs)
795
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
796
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
797
+ return model_forward(*args, **kwargs)
798
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
799
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
800
+ return convert_to_fp32(self.model_forward(*args, **kwargs))
801
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
802
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
803
+ return func(*args, **kwargs)
804
+ ^^^^^^^^^^^^^^^^^^^^^
805
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
806
+ backbone_outputs = self.backbone(backbone_inputs)
807
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
808
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
809
+ return self._call_impl(*args, **kwargs)
810
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
811
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
812
+ return forward_call(*args, **kwargs)
813
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
814
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
815
+ eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
816
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
817
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
818
+ eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
819
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
820
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
821
+ return self._call_impl(*args, **kwargs)
822
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
823
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
824
+ return forward_call(*args, **kwargs)
825
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
826
+ File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
827
+ outputs = self.language_model(
828
+ ^^^^^^^^^^^^^^^^^^^^
829
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
830
+ return self._call_impl(*args, **kwargs)
831
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
832
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
833
+ return forward_call(*args, **kwargs)
834
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
835
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
836
+ output = func(self, *args, **kwargs)
837
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
838
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
839
+ return func(*args, **kwargs)
840
+ ^^^^^^^^^^^^^^^^^^^^^
841
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
842
+ outputs: BaseModelOutputWithPast = self.model(
843
+ ^^^^^^^^^^^
844
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
845
+ return self._call_impl(*args, **kwargs)
846
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
847
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
848
+ return forward_call(*args, **kwargs)
849
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
850
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
851
+ output = func(self, *args, **kwargs)
852
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
853
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
854
+ layer_outputs = decoder_layer(
855
+ ^^^^^^^^^^^^^^
856
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
857
+ return self._call_impl(*args, **kwargs)
858
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
859
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
860
+ return forward_call(*args, **kwargs)
861
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
862
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
863
+ hidden_states = self.mlp(hidden_states)
864
+ ^^^^^^^^^^^^^^^^^^^^^^^
865
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
866
+ return self._call_impl(*args, **kwargs)
867
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
868
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
869
+ return forward_call(*args, **kwargs)
870
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
871
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
872
+ down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
873
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
874
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
875
+ return self._call_impl(*args, **kwargs)
876
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
877
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
878
+ return forward_call(*args, **kwargs)
879
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
880
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/activation.py", line 432, in forward
881
+ return F.silu(input, inplace=self.inplace)
882
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
883
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/functional.py", line 2380, in silu
884
+ return torch._C._nn.silu(input)
885
+ ^^^^^^^^^^^^^^^^^^^^^^^^
886
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 469.94 MiB is free. Including non-PyTorch memory, this process has 78.77 GiB memory in use. Of the allocated memory 76.81 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
887
+ [rank0]: Traceback (most recent call last):
888
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
889
+ [rank0]: run_yaml_experiment(
890
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
891
+ [rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
892
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
893
+ [rank0]: experiment.train()
894
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
895
+ [rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
896
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
897
+ [rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
898
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
899
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
900
+ [rank0]: return inner_training_loop(
901
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
902
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
903
+ [rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
904
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
905
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
906
+ [rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
907
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
908
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
909
+ [rank0]: outputs = model(inputs)
910
+ [rank0]: ^^^^^^^^^^^^^
911
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
912
+ [rank0]: return self._call_impl(*args, **kwargs)
913
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
914
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
915
+ [rank0]: return forward_call(*args, **kwargs)
916
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
917
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
918
+ [rank0]: else self._run_ddp_forward(*inputs, **kwargs)
919
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
920
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
921
+ [rank0]: return self.module(*inputs, **kwargs) # type: ignore[index]
922
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
923
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
924
+ [rank0]: return self._call_impl(*args, **kwargs)
925
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
926
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
927
+ [rank0]: return forward_call(*args, **kwargs)
928
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
929
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
930
+ [rank0]: return model_forward(*args, **kwargs)
931
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
932
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
933
+ [rank0]: return convert_to_fp32(self.model_forward(*args, **kwargs))
934
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
935
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
936
+ [rank0]: return func(*args, **kwargs)
937
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^
938
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
939
+ [rank0]: backbone_outputs = self.backbone(backbone_inputs)
940
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
941
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
942
+ [rank0]: return self._call_impl(*args, **kwargs)
943
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
944
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
945
+ [rank0]: return forward_call(*args, **kwargs)
946
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
947
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
948
+ [rank0]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
949
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
950
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
951
+ [rank0]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
952
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
953
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
954
+ [rank0]: return self._call_impl(*args, **kwargs)
955
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
956
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
957
+ [rank0]: return forward_call(*args, **kwargs)
958
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
959
+ [rank0]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
960
+ [rank0]: outputs = self.language_model(
961
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
962
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
963
+ [rank0]: return self._call_impl(*args, **kwargs)
964
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
965
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
966
+ [rank0]: return forward_call(*args, **kwargs)
967
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
968
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
969
+ [rank0]: output = func(self, *args, **kwargs)
970
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
971
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
972
+ [rank0]: return func(*args, **kwargs)
973
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^
974
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
975
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
976
+ [rank0]: ^^^^^^^^^^^
977
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
978
+ [rank0]: return self._call_impl(*args, **kwargs)
979
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
980
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
981
+ [rank0]: return forward_call(*args, **kwargs)
982
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
983
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
984
+ [rank0]: output = func(self, *args, **kwargs)
985
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
986
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
987
+ [rank0]: layer_outputs = decoder_layer(
988
+ [rank0]: ^^^^^^^^^^^^^^
989
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
990
+ [rank0]: return self._call_impl(*args, **kwargs)
991
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
992
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
993
+ [rank0]: return forward_call(*args, **kwargs)
994
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
995
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
996
+ [rank0]: hidden_states = self.mlp(hidden_states)
997
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
998
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
999
+ [rank0]: return self._call_impl(*args, **kwargs)
1000
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1001
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
1002
+ [rank0]: return forward_call(*args, **kwargs)
1003
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1004
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
1005
+ [rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
1006
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1007
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
1008
+ [rank0]: return self._call_impl(*args, **kwargs)
1009
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1010
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
1011
+ [rank0]: return forward_call(*args, **kwargs)
1012
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1013
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/activation.py", line 432, in forward
1014
+ [rank0]: return F.silu(input, inplace=self.inplace)
1015
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1016
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/functional.py", line 2380, in silu
1017
+ [rank0]: return torch._C._nn.silu(input)
1018
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^
1019
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 469.94 MiB is free. Including non-PyTorch memory, this process has 78.77 GiB memory in use. Of the allocated memory 76.81 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
1020
+ [rank0]:[W615 14:13:02.012395307 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
1021
+ W0615 14:13:02.646000 3538434 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 3544058 closing signal SIGTERM
1022
+ E0615 14:13:03.061000 3538434 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 3544064) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
1023
+ Traceback (most recent call last):
1024
+ File "<frozen runpy>", line 198, in _run_module_as_main
1025
+ File "<frozen runpy>", line 88, in _run_code
1026
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
1027
+ main()
1028
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
1029
+ return f(*args, **kwargs)
1030
+ ^^^^^^^^^^^^^^^^^^
1031
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
1032
+ run(args)
1033
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
1034
+ elastic_launch(
1035
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
1036
+ return launch_agent(self._config, self._entrypoint, list(args))
1037
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1038
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
1039
+ raise ChildFailedError(
1040
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
1041
+ ============================================================
1042
+ /home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
1043
+ ------------------------------------------------------------
1044
+ Failures:
1045
+ <NO_OTHER_FAILURES>
1046
+ ------------------------------------------------------------
1047
+ Root Cause (first observed failure):
1048
+ [0]:
1049
+ time : 2026-06-15_14:13:02
1050
+ host : worker1
1051
+ rank : 1 (local_rank: 1)
1052
+ exitcode : 1 (pid: 3544064)
1053
+ error_file: <N/A>
1054
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
1055
+ ============================================================
1056
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
1057
+
1058
+ ==================================================
1059
+ GR00T FINE-TUNING CONFIGURATION:
1060
+ ==================================================
1061
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
1062
+ dataset_soup: None
1063
+ output_dir: /tmp/gr00t
1064
+ output_root: None
1065
+ data_config: panda_omron
1066
+ batch_size: 64
1067
+ max_steps: 300000
1068
+ num_gpus: 2
1069
+ save_steps: 20000
1070
+ run_name: None
1071
+ save_total_limit: 100
1072
+ seed: 42
1073
+ base_model_path: nvidia/GR00T-N1.5-3B
1074
+ tune_llm: False
1075
+ tune_visual: False
1076
+ tune_projector: True
1077
+ tune_diffusion_model: True
1078
+ resume: False
1079
+ learning_rate: 3e-05
1080
+ weight_decay: 1e-05
1081
+ warmup_ratio: 0.05
1082
+ lora_rank: 0
1083
+ lora_alpha: 16
1084
+ lora_dropout: 0.1
1085
+ lora_full_model: False
1086
+ dataloader_num_workers: 8
1087
+ report_to: wandb
1088
+ embodiment_tag: new_embodiment
1089
+ video_backend: opencv
1090
+ balance_dataset_weights: True
1091
+ balance_trajectory_weights: True
1092
+ ds_weights_alpha: 0.4
1093
+ ==================================================
1094
+
1095
+ Using 2 GPUs
1096
+ Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '64', '--num-gpus', '2']
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs64_gpu2_3_lmheadfreeze.log ADDED
@@ -0,0 +1,1087 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [robosuite WARNING] No private macro file found! (macros.py:57)
2
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
3
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
4
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
5
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
6
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
7
+ check_for_updates()
8
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
9
+
10
+ *****************************************
11
+ Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
12
+ *****************************************
13
+ [robosuite WARNING] No private macro file found! (macros.py:57)
14
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
15
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
16
+ [robosuite WARNING] No private macro file found! (macros.py:57)
17
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
18
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
19
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
20
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
21
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
22
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
23
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
24
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
25
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
26
+ check_for_updates()
27
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
28
+ check_for_updates()
29
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
30
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
31
+
32
+ ==================================================
33
+ GR00T FINE-TUNING CONFIGURATION:
34
+ ==================================================
35
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
36
+ dataset_soup: None
37
+ output_dir: /tmp/gr00t
38
+ output_root: None
39
+ data_config: panda_omron
40
+ batch_size: 64
41
+ max_steps: 300000
42
+ num_gpus: 2
43
+ save_steps: 20000
44
+ run_name: None
45
+ save_total_limit: 100
46
+ seed: 42
47
+ base_model_path: nvidia/GR00T-N1.5-3B
48
+ tune_llm: False
49
+ tune_visual: False
50
+ tune_projector: True
51
+ tune_diffusion_model: True
52
+ resume: False
53
+ learning_rate: 3e-05
54
+ weight_decay: 1e-05
55
+ warmup_ratio: 0.05
56
+ lora_rank: 0
57
+ lora_alpha: 16
58
+ lora_dropout: 0.1
59
+ lora_full_model: False
60
+ dataloader_num_workers: 8
61
+ report_to: wandb
62
+ embodiment_tag: new_embodiment
63
+ video_backend: opencv
64
+ balance_dataset_weights: True
65
+ balance_trajectory_weights: True
66
+ ds_weights_alpha: 0.4
67
+ ==================================================
68
+
69
+ Using 2 GPUs
70
+
71
+ ================================================================================
72
+ Starting sweep branch: default
73
+ Sweep vars: {}
74
+ ================================================================================
75
+
76
+ --------------------------------------------------------------------------------
77
+ Running phase 1: phase2_rkd_a_vlm_only
78
+ Policy type: groot_rkd_v2_raw
79
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
80
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
81
+ Trainable preset: freeze_processing_line
82
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
83
+ --------------------------------------------------------------------------------
84
+
85
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
86
+ Using 100 subset demos for filter_key: 100_demos
87
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
88
+ self.statistics[key] = torch.tensor(value)
89
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
90
+ Using 100 subset demos for filter_key: 100_demos
91
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
92
+ Using 100 subset demos for filter_key: 100_demos
93
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
94
+ Using 100 subset demos for filter_key: 100_demos
95
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
96
+ Using 100 subset demos for filter_key: 100_demos
97
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
98
+ Using 100 subset demos for filter_key: 100_demos
99
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
100
+ Using 100 subset demos for filter_key: 100_demos
101
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
102
+ Using 100 subset demos for filter_key: 100_demos
103
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
104
+ Using 100 subset demos for filter_key: 100_demos
105
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
106
+ Using 100 subset demos for filter_key: 100_demos
107
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
108
+ Using 100 subset demos for filter_key: 100_demos
109
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
110
+ Using 100 subset demos for filter_key: 100_demos
111
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
112
+ Using 100 subset demos for filter_key: 100_demos
113
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
114
+ Using 100 subset demos for filter_key: 100_demos
115
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
116
+ Using 100 subset demos for filter_key: 100_demos
117
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
118
+ Using 100 subset demos for filter_key: 100_demos
119
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
120
+ Using 100 subset demos for filter_key: 100_demos
121
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
122
+ Using 100 subset demos for filter_key: 100_demos
123
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
124
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
125
+ Using 100 subset demos for filter_key: 100_demos
126
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
127
+ Using 100 subset demos for filter_key: 100_demos
128
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
129
+ Using 100 subset demos for filter_key: 100_demos
130
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
131
+ Using 100 subset demos for filter_key: 100_demos
132
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
133
+ Using 100 subset demos for filter_key: 100_demos
134
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
135
+ Using 100 subset demos for filter_key: 100_demos
136
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
137
+ Using 100 subset demos for filter_key: 100_demos
138
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
139
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
140
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
141
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
142
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
143
+ 0.75517122 0.7973985 ]
144
+ Loaded 26 datasets
145
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
146
+ Tune backbone vision tower: True
147
+ Tune backbone LLM: True
148
+ Tune action head projector: False
149
+ Tune action head DiT: False
150
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
151
+
152
+ ==================================================
153
+ GR00T FINE-TUNING CONFIGURATION:
154
+ ==================================================
155
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
156
+ dataset_soup: None
157
+ output_dir: /tmp/gr00t
158
+ output_root: None
159
+ data_config: panda_omron
160
+ batch_size: 64
161
+ max_steps: 300000
162
+ num_gpus: 2
163
+ save_steps: 20000
164
+ run_name: None
165
+ save_total_limit: 100
166
+ seed: 42
167
+ base_model_path: nvidia/GR00T-N1.5-3B
168
+ tune_llm: False
169
+ tune_visual: False
170
+ tune_projector: True
171
+ tune_diffusion_model: True
172
+ resume: False
173
+ learning_rate: 3e-05
174
+ weight_decay: 1e-05
175
+ warmup_ratio: 0.05
176
+ lora_rank: 0
177
+ lora_alpha: 16
178
+ lora_dropout: 0.1
179
+ lora_full_model: False
180
+ dataloader_num_workers: 8
181
+ report_to: wandb
182
+ embodiment_tag: new_embodiment
183
+ video_backend: opencv
184
+ balance_dataset_weights: True
185
+ balance_trajectory_weights: True
186
+ ds_weights_alpha: 0.4
187
+ ==================================================
188
+
189
+ Using 2 GPUs
190
+
191
+ ================================================================================
192
+ Starting sweep branch: default
193
+ Sweep vars: {}
194
+ ================================================================================
195
+
196
+ --------------------------------------------------------------------------------
197
+ Running phase 1: phase2_rkd_a_vlm_only
198
+ Policy type: groot_rkd_v2_raw
199
+ Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
200
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
201
+ Trainable preset: freeze_processing_line
202
+ Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
203
+ --------------------------------------------------------------------------------
204
+
205
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
206
+ Using 100 subset demos for filter_key: 100_demos
207
+ /home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
208
+ self.statistics[key] = torch.tensor(value)
209
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
210
+ Using 100 subset demos for filter_key: 100_demos
211
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
212
+ Using 100 subset demos for filter_key: 100_demos
213
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
214
+ Using 100 subset demos for filter_key: 100_demos
215
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
216
+ Using 100 subset demos for filter_key: 100_demos
217
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
218
+ Using 100 subset demos for filter_key: 100_demos
219
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
220
+ Using 100 subset demos for filter_key: 100_demos
221
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
222
+ Using 100 subset demos for filter_key: 100_demos
223
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
224
+ Using 100 subset demos for filter_key: 100_demos
225
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
226
+ Using 100 subset demos for filter_key: 100_demos
227
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
228
+ Using 100 subset demos for filter_key: 100_demos
229
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
230
+ Using 100 subset demos for filter_key: 100_demos
231
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
232
+ Using 100 subset demos for filter_key: 100_demos
233
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
234
+ Using 100 subset demos for filter_key: 100_demos
235
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
236
+ Using 100 subset demos for filter_key: 100_demos
237
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
238
+ Using 100 subset demos for filter_key: 100_demos
239
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
240
+ Using 100 subset demos for filter_key: 100_demos
241
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
242
+ Using 100 subset demos for filter_key: 100_demos
243
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
244
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
245
+ Using 100 subset demos for filter_key: 100_demos
246
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
247
+ Using 100 subset demos for filter_key: 100_demos
248
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
249
+ Using 100 subset demos for filter_key: 100_demos
250
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
251
+ Using 100 subset demos for filter_key: 100_demos
252
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
253
+ Using 100 subset demos for filter_key: 100_demos
254
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
255
+ Using 100 subset demos for filter_key: 100_demos
256
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
257
+ Using 100 subset demos for filter_key: 100_demos
258
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
259
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
260
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
261
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
262
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
263
+ 0.75517122 0.7973985 ]
264
+ Loaded 26 datasets
265
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
266
+ Tune backbone vision tower: True
267
+ Tune backbone LLM: True
268
+ Tune action head projector: False
269
+ Tune action head DiT: False
270
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
271
+ Tune backbone llm: False
272
+ Tune backbone visual: True
273
+ Total number of DiT parameters: 550386688
274
+ Tune backbone llm: False
275
+ Tune backbone visual: True
276
+ Total number of DiT parameters: 550386688
277
+ Total number of SelfAttentionTransformer parameters: 201433088
278
+ Tune action head projector: True
279
+ Tune action head diffusion model: True
280
+
281
+ Tune backbone llm: True
282
+ Tune backbone visual: True
283
+ Tune action head projector: False
284
+ Tune action head diffusion model: False
285
+ Action head trainable parameter: future_tokens.weight
286
+ Action head trainable parameter: vlln.weight
287
+ Action head trainable parameter: vlln.bias
288
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
289
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
290
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
291
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
292
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
293
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
294
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
295
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
296
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
297
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
298
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
299
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
300
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
301
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
302
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
303
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
304
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
305
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
306
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
307
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
308
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
309
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
310
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
311
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
312
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
313
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
314
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
315
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
316
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
317
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
318
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
319
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
320
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
321
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
322
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
323
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
324
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
325
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
326
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
327
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
328
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
329
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
330
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
331
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
332
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
333
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
334
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
335
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
336
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
337
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
338
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
339
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
340
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
341
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
342
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
343
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
344
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
345
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
346
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
347
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
348
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
349
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
350
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
351
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
352
+ Applied trainable preset: freeze_processing_line
353
+ Trainable parameter tensors after preset: 584
354
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
355
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
356
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
357
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
358
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
359
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
360
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
361
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
362
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
363
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
364
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
365
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
366
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
367
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
368
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
369
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
370
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
371
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
372
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
373
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
374
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
375
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
376
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
377
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
378
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
379
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
380
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
381
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
382
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
383
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
384
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
385
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
386
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
387
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
388
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
389
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
390
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
391
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
392
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
393
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
394
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
395
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
396
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
397
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
398
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
399
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
400
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
401
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
402
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
403
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
404
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
405
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
406
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
407
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
408
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
409
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
410
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
411
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
412
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
413
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
414
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
415
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
416
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
417
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
418
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
419
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
420
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
421
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
422
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
423
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
424
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
425
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
426
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
427
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
428
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
429
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
430
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
431
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
432
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
433
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
434
+ ... 504 more
435
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
436
+ Run name: default_phase2
437
+ Total number of SelfAttentionTransformer parameters: 201433088
438
+ Tune action head projector: True
439
+ Tune action head diffusion model: True
440
+
441
+ Tune backbone llm: True
442
+ Tune backbone visual: True
443
+ Tune action head projector: False
444
+ Tune action head diffusion model: False
445
+ Action head trainable parameter: future_tokens.weight
446
+ Action head trainable parameter: vlln.weight
447
+ Action head trainable parameter: vlln.bias
448
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
449
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
450
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
451
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
452
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
453
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
454
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
455
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
456
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
457
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
458
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
459
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
460
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
461
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
462
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
463
+ Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
464
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
465
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
466
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
467
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
468
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
469
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
470
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
471
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
472
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
473
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
474
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
475
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
476
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
477
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
478
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
479
+ Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
480
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
481
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
482
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
483
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
484
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
485
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
486
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
487
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
488
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
489
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
490
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
491
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
492
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
493
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
494
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
495
+ Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
496
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
497
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
498
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
499
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
500
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
501
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
502
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
503
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
504
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
505
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
506
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
507
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
508
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
509
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
510
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
511
+ Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
512
+ Applied trainable preset: freeze_processing_line
513
+ Trainable parameter tensors after preset: 584
514
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
515
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
516
+ trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
517
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
518
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
519
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
520
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
521
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
522
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
523
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
524
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
525
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
526
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
527
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
528
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
529
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
530
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
531
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
532
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
533
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
534
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
535
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
536
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
537
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
538
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
539
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
540
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
541
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
542
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
543
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
544
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
545
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
546
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
547
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
548
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
549
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
550
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
551
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
552
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
553
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
554
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
555
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
556
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
557
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
558
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
559
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
560
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
561
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
562
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
563
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
564
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
565
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
566
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
567
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
568
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
569
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
570
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
571
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
572
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
573
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
574
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
575
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
576
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
577
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
578
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
579
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
580
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
581
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
582
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
583
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
584
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
585
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
586
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
587
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
588
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
589
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
590
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
591
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
592
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
593
+ trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
594
+ ... 504 more
595
+ Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
596
+ Run name: default_phase2
597
+ train dataloader length: 3437
598
+ train dataset length: 439854
599
+ GPU memory before training: 7.076685905456543 GB
600
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
601
+ train dataloader length: 3437
602
+ train dataset length: 439854
603
+ GPU memory before training: 7.076685905456543 GB
604
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
605
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
606
+ wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
607
+ wandb: setting up run 8wkyy6nk
608
+ wandb: Tracking run with wandb version 0.25.0
609
+ wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_143025-8wkyy6nk
610
+ wandb: Run `wandb offline` to turn off syncing.
611
+ wandb: Syncing run default_phase2
612
+ wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
613
+ wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/8wkyy6nk
614
+
615
  0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
616
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
617
+ [rank1]: run_yaml_experiment(
618
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
619
+ [rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
620
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
621
+ [rank1]: experiment.train()
622
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
623
+ [rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
624
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
625
+ [rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
626
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
627
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
628
+ [rank1]: return inner_training_loop(
629
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
630
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
631
+ [rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
632
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
633
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
634
+ [rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
635
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
636
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
637
+ [rank1]: outputs = model(inputs)
638
+ [rank1]: ^^^^^^^^^^^^^
639
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
640
+ [rank1]: return self._call_impl(*args, **kwargs)
641
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
642
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
643
+ [rank1]: return forward_call(*args, **kwargs)
644
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
645
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
646
+ [rank1]: else self._run_ddp_forward(*inputs, **kwargs)
647
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
648
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
649
+ [rank1]: return self.module(*inputs, **kwargs) # type: ignore[index]
650
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
651
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
652
+ [rank1]: return self._call_impl(*args, **kwargs)
653
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
654
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
655
+ [rank1]: return forward_call(*args, **kwargs)
656
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
657
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
658
+ [rank1]: return model_forward(*args, **kwargs)
659
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
660
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
661
+ [rank1]: return convert_to_fp32(self.model_forward(*args, **kwargs))
662
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
663
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
664
+ [rank1]: return func(*args, **kwargs)
665
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^
666
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
667
+ [rank1]: backbone_outputs = self.backbone(backbone_inputs)
668
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
669
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
670
+ [rank1]: return self._call_impl(*args, **kwargs)
671
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
672
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
673
+ [rank1]: return forward_call(*args, **kwargs)
674
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
675
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
676
+ [rank1]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
677
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
678
+ [rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
679
+ [rank1]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
680
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
681
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
682
+ [rank1]: return self._call_impl(*args, **kwargs)
683
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
684
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
685
+ [rank1]: return forward_call(*args, **kwargs)
686
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
687
+ [rank1]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
688
+ [rank1]: outputs = self.language_model(
689
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^
690
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
691
+ [rank1]: return self._call_impl(*args, **kwargs)
692
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
693
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
694
+ [rank1]: return forward_call(*args, **kwargs)
695
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
696
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
697
+ [rank1]: output = func(self, *args, **kwargs)
698
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
699
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
700
+ [rank1]: return func(*args, **kwargs)
701
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^
702
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
703
+ [rank1]: outputs: BaseModelOutputWithPast = self.model(
704
+ [rank1]: ^^^^^^^^^^^
705
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
706
+ [rank1]: return self._call_impl(*args, **kwargs)
707
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
708
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
709
+ [rank1]: return forward_call(*args, **kwargs)
710
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
711
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
712
+ [rank1]: output = func(self, *args, **kwargs)
713
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
714
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
715
+ [rank1]: layer_outputs = decoder_layer(
716
+ [rank1]: ^^^^^^^^^^^^^^
717
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
718
+ [rank1]: return self._call_impl(*args, **kwargs)
719
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
720
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
721
+ [rank1]: return forward_call(*args, **kwargs)
722
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
723
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
724
+ [rank1]: hidden_states = self.mlp(hidden_states)
725
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^
726
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
727
+ [rank1]: return self._call_impl(*args, **kwargs)
728
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
729
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
730
+ [rank1]: return forward_call(*args, **kwargs)
731
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
732
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
733
+ [rank1]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
734
+ [rank1]: ^^^^^^^^^^^^^^^
735
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
736
+ [rank1]: return self._call_impl(*args, **kwargs)
737
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
738
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
739
+ [rank1]: return forward_call(*args, **kwargs)
740
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
741
+ [rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
742
+ [rank1]: return F.linear(input, self.weight, self.bias)
743
+ [rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
744
+ [rank1]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 1 has a total capacity of 79.25 GiB of which 433.94 MiB is free. Including non-PyTorch memory, this process has 78.80 GiB memory in use. Of the allocated memory 76.84 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
745
+ wandb: updating run metadata
746
+ wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
747
+ wandb: uploading wandb-summary.json; uploading config.yaml
748
+ wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/8wkyy6nk
749
+ wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
750
+ wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
751
+ wandb: Find logs at: ./wandb/run-20260615_143025-8wkyy6nk/logs
752
+ Traceback (most recent call last):
753
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
754
+ run_yaml_experiment(
755
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
756
+ main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
757
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
758
+ experiment.train()
759
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
760
+ self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
761
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
762
+ return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
763
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
764
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
765
+ return inner_training_loop(
766
+ ^^^^^^^^^^^^^^^^^^^^
767
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
768
+ tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
769
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
770
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
771
+ loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
772
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
773
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
774
+ outputs = model(inputs)
775
+ ^^^^^^^^^^^^^
776
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
777
+ return self._call_impl(*args, **kwargs)
778
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
779
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
780
+ return forward_call(*args, **kwargs)
781
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
782
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
783
+ else self._run_ddp_forward(*inputs, **kwargs)
784
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
785
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
786
+ return self.module(*inputs, **kwargs) # type: ignore[index]
787
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
788
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
789
+ return self._call_impl(*args, **kwargs)
790
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
791
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
792
+ return forward_call(*args, **kwargs)
793
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
794
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
795
+ return model_forward(*args, **kwargs)
796
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
797
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
798
+ return convert_to_fp32(self.model_forward(*args, **kwargs))
799
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
800
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
801
+ return func(*args, **kwargs)
802
+ ^^^^^^^^^^^^^^^^^^^^^
803
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
804
+ backbone_outputs = self.backbone(backbone_inputs)
805
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
806
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
807
+ return self._call_impl(*args, **kwargs)
808
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
809
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
810
+ return forward_call(*args, **kwargs)
811
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
812
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
813
+ eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
814
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
815
+ File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
816
+ eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
817
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
818
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
819
+ return self._call_impl(*args, **kwargs)
820
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
821
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
822
+ return forward_call(*args, **kwargs)
823
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
824
+ File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
825
+ outputs = self.language_model(
826
+ ^^^^^^^^^^^^^^^^^^^^
827
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
828
+ return self._call_impl(*args, **kwargs)
829
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
830
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
831
+ return forward_call(*args, **kwargs)
832
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
833
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
834
+ output = func(self, *args, **kwargs)
835
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
836
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
837
+ return func(*args, **kwargs)
838
+ ^^^^^^^^^^^^^^^^^^^^^
839
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
840
+ outputs: BaseModelOutputWithPast = self.model(
841
+ ^^^^^^^^^^^
842
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
843
+ return self._call_impl(*args, **kwargs)
844
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
845
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
846
+ return forward_call(*args, **kwargs)
847
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
848
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
849
+ output = func(self, *args, **kwargs)
850
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
851
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
852
+ layer_outputs = decoder_layer(
853
+ ^^^^^^^^^^^^^^
854
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
855
+ return self._call_impl(*args, **kwargs)
856
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
857
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
858
+ return forward_call(*args, **kwargs)
859
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
860
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
861
+ hidden_states = self.mlp(hidden_states)
862
+ ^^^^^^^^^^^^^^^^^^^^^^^
863
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
864
+ return self._call_impl(*args, **kwargs)
865
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
866
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
867
+ return forward_call(*args, **kwargs)
868
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
869
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
870
+ down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
871
+ ^^^^^^^^^^^^^^^
872
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
873
+ return self._call_impl(*args, **kwargs)
874
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
875
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
876
+ return forward_call(*args, **kwargs)
877
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
878
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
879
+ return F.linear(input, self.weight, self.bias)
880
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
881
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 433.94 MiB is free. Including non-PyTorch memory, this process has 78.80 GiB memory in use. Of the allocated memory 76.84 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
882
+ [rank0]: Traceback (most recent call last):
883
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
884
+ [rank0]: run_yaml_experiment(
885
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
886
+ [rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
887
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
888
+ [rank0]: experiment.train()
889
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
890
+ [rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
891
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
892
+ [rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
893
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
894
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
895
+ [rank0]: return inner_training_loop(
896
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
897
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
898
+ [rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
899
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
900
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
901
+ [rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
902
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
903
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
904
+ [rank0]: outputs = model(inputs)
905
+ [rank0]: ^^^^^^^^^^^^^
906
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
907
+ [rank0]: return self._call_impl(*args, **kwargs)
908
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
909
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
910
+ [rank0]: return forward_call(*args, **kwargs)
911
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
912
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
913
+ [rank0]: else self._run_ddp_forward(*inputs, **kwargs)
914
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
915
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
916
+ [rank0]: return self.module(*inputs, **kwargs) # type: ignore[index]
917
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
918
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
919
+ [rank0]: return self._call_impl(*args, **kwargs)
920
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
921
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
922
+ [rank0]: return forward_call(*args, **kwargs)
923
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
924
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
925
+ [rank0]: return model_forward(*args, **kwargs)
926
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
927
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
928
+ [rank0]: return convert_to_fp32(self.model_forward(*args, **kwargs))
929
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
930
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
931
+ [rank0]: return func(*args, **kwargs)
932
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^
933
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
934
+ [rank0]: backbone_outputs = self.backbone(backbone_inputs)
935
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
936
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
937
+ [rank0]: return self._call_impl(*args, **kwargs)
938
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
939
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
940
+ [rank0]: return forward_call(*args, **kwargs)
941
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
942
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
943
+ [rank0]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
944
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
945
+ [rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
946
+ [rank0]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
947
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
948
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
949
+ [rank0]: return self._call_impl(*args, **kwargs)
950
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
951
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
952
+ [rank0]: return forward_call(*args, **kwargs)
953
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
954
+ [rank0]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
955
+ [rank0]: outputs = self.language_model(
956
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
957
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
958
+ [rank0]: return self._call_impl(*args, **kwargs)
959
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
960
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
961
+ [rank0]: return forward_call(*args, **kwargs)
962
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
963
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
964
+ [rank0]: output = func(self, *args, **kwargs)
965
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
966
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
967
+ [rank0]: return func(*args, **kwargs)
968
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^
969
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
970
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
971
+ [rank0]: ^^^^^^^^^^^
972
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
973
+ [rank0]: return self._call_impl(*args, **kwargs)
974
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
975
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
976
+ [rank0]: return forward_call(*args, **kwargs)
977
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
978
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
979
+ [rank0]: output = func(self, *args, **kwargs)
980
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
981
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
982
+ [rank0]: layer_outputs = decoder_layer(
983
+ [rank0]: ^^^^^^^^^^^^^^
984
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
985
+ [rank0]: return self._call_impl(*args, **kwargs)
986
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
987
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
988
+ [rank0]: return forward_call(*args, **kwargs)
989
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
990
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
991
+ [rank0]: hidden_states = self.mlp(hidden_states)
992
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
993
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
994
+ [rank0]: return self._call_impl(*args, **kwargs)
995
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
996
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
997
+ [rank0]: return forward_call(*args, **kwargs)
998
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
999
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
1000
+ [rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
1001
+ [rank0]: ^^^^^^^^^^^^^^^
1002
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
1003
+ [rank0]: return self._call_impl(*args, **kwargs)
1004
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1005
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
1006
+ [rank0]: return forward_call(*args, **kwargs)
1007
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1008
+ [rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
1009
+ [rank0]: return F.linear(input, self.weight, self.bias)
1010
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1011
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 433.94 MiB is free. Including non-PyTorch memory, this process has 78.80 GiB memory in use. Of the allocated memory 76.84 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
1012
+ [rank0]:[W615 14:30:41.759054055 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
1013
+ W0615 14:30:42.535000 3405061 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 3429266 closing signal SIGTERM
1014
+ E0615 14:30:42.850000 3405061 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 3429267) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
1015
+ Traceback (most recent call last):
1016
+ File "<frozen runpy>", line 198, in _run_module_as_main
1017
+ File "<frozen runpy>", line 88, in _run_code
1018
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
1019
+ main()
1020
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
1021
+ return f(*args, **kwargs)
1022
+ ^^^^^^^^^^^^^^^^^^
1023
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
1024
+ run(args)
1025
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
1026
+ elastic_launch(
1027
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
1028
+ return launch_agent(self._config, self._entrypoint, list(args))
1029
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
1030
+ File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
1031
+ raise ChildFailedError(
1032
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
1033
+ ============================================================
1034
+ /home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
1035
+ ------------------------------------------------------------
1036
+ Failures:
1037
+ <NO_OTHER_FAILURES>
1038
+ ------------------------------------------------------------
1039
+ Root Cause (first observed failure):
1040
+ [0]:
1041
+ time : 2026-06-15_14:30:42
1042
+ host : worker1
1043
+ rank : 1 (local_rank: 1)
1044
+ exitcode : 1 (pid: 3429267)
1045
+ error_file: <N/A>
1046
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
1047
+ ============================================================
1048
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
1049
+
1050
+ ==================================================
1051
+ GR00T FINE-TUNING CONFIGURATION:
1052
+ ==================================================
1053
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
1054
+ dataset_soup: None
1055
+ output_dir: /tmp/gr00t
1056
+ output_root: None
1057
+ data_config: panda_omron
1058
+ batch_size: 64
1059
+ max_steps: 300000
1060
+ num_gpus: 2
1061
+ save_steps: 20000
1062
+ run_name: None
1063
+ save_total_limit: 100
1064
+ seed: 42
1065
+ base_model_path: nvidia/GR00T-N1.5-3B
1066
+ tune_llm: False
1067
+ tune_visual: False
1068
+ tune_projector: True
1069
+ tune_diffusion_model: True
1070
+ resume: False
1071
+ learning_rate: 3e-05
1072
+ weight_decay: 1e-05
1073
+ warmup_ratio: 0.05
1074
+ lora_rank: 0
1075
+ lora_alpha: 16
1076
+ lora_dropout: 0.1
1077
+ lora_full_model: False
1078
+ dataloader_num_workers: 8
1079
+ report_to: wandb
1080
+ embodiment_tag: new_embodiment
1081
+ video_backend: opencv
1082
+ balance_dataset_weights: True
1083
+ balance_trajectory_weights: True
1084
+ ds_weights_alpha: 0.4
1085
+ ==================================================
1086
+
1087
+ Using 2 GPUs
1088
+ Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '64', '--num-gpus', '2']
experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p1_train.log ADDED
The diff for this file is too large to render. See raw diff
 
experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p2_train.log ADDED
The diff for this file is too large to render. See raw diff
 
experiment_cfg/processing_line_only_v2/MGD_v2/train.log ADDED
The diff for this file is too large to render. See raw diff
 
experiment_cfg/processing_line_only_v2/MGD_v2/train_mse.log ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/trainer_state.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/trainer_state.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_cosine_mse.yaml ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_loss_cosine_mse
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.3
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
31
+ mgd_use_cosine_loss: true
32
+ mgd_use_mse_loss: true
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
46
+ mgd_use_cosine_loss: true
47
+ mgd_use_mse_loss: true
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_cosine_mse_phase3_only.yaml ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_loss_cosine_mse_phase3
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.3
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase3_fm_mgd_fixed_0p5
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ tune_llm: false
21
+ tune_visual: false
22
+ tune_projector: true
23
+ tune_diffusion_model: true
24
+ losses:
25
+ mgd_enabled: true
26
+ mgd_fm_loss_weight: 1.0
27
+ mgd_loss_weight: 0.5
28
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
29
+ mgd_use_cosine_loss: true
30
+ mgd_use_mse_loss: true
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/config.json ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_MGD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "mgd_enabled": true,
61
+ "mgd_fm_loss_weight": 0.0,
62
+ "mgd_loss_weight": 1.0,
63
+ "mgd_loss_weight_end": 0.0,
64
+ "mgd_loss_weight_schedule": null,
65
+ "mgd_loss_weight_start": 0.05,
66
+ "mgd_pretrained_projector_path": null,
67
+ "mgd_sequence_hidden_dim": 512,
68
+ "mgd_target_dim": 512,
69
+ "mgd_target_pooling": "flatten",
70
+ "mgd_target_projection": "frozen_random",
71
+ "mgd_token_mask_ratio": 0.3,
72
+ "mgd_use_cosine_loss": true,
73
+ "mgd_use_mse_loss": true,
74
+ "model_dtype": "float32",
75
+ "model_type": "gr00t_n1_5",
76
+ "torch_dtype": "bfloat16",
77
+ "transformers_version": "4.51.3"
78
+ }
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/trainer_state.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/config.json ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_MGD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "mgd_enabled": true,
61
+ "mgd_fm_loss_weight": 1.0,
62
+ "mgd_loss_weight": 0.5,
63
+ "mgd_loss_weight_end": 0.0,
64
+ "mgd_loss_weight_schedule": null,
65
+ "mgd_loss_weight_start": 0.05,
66
+ "mgd_pretrained_projector_path": null,
67
+ "mgd_sequence_hidden_dim": 512,
68
+ "mgd_target_dim": 512,
69
+ "mgd_target_pooling": "flatten",
70
+ "mgd_target_projection": "frozen_random",
71
+ "mgd_token_mask_ratio": 0.3,
72
+ "mgd_use_cosine_loss": true,
73
+ "mgd_use_mse_loss": true,
74
+ "model_dtype": "float32",
75
+ "model_type": "gr00t_n1_5",
76
+ "torch_dtype": "bfloat16",
77
+ "transformers_version": "4.51.3"
78
+ }
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/trainer_state.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3_only_train.log ADDED
@@ -0,0 +1,158 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/30000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
 
1
  0%| | 1/30000 [00:12<105:19:47, 12.64s/it]
2
  0%| | 2/30000 [00:14<52:20:08, 6.28s/it]
3
  0%| | 3/30000 [00:16<35:11:45, 4.22s/it]
4
  0%| | 4/30000 [00:18<27:10:39, 3.26s/it]
5
  0%| | 5/30000 [00:19<22:43:51, 2.73s/it]
6
  0%| | 6/30000 [00:21<20:05:04, 2.41s/it]
7
  0%| | 7/30000 [00:23<18:23:19, 2.21s/it]
8
  0%| | 8/30000 [00:25<17:24:08, 2.09s/it]
9
  0%| | 9/30000 [00:27<16:37:23, 2.00s/it]
10
  0%| | 10/30000 [00:28<16:10:55, 1.94s/it]
11
 
12
  0%| | 10/30000 [00:28<16:10:55, 1.94s/it]
13
  0%| | 11/30000 [00:30<15:49:20, 1.90s/it]
14
  0%| | 12/30000 [00:32<15:33:32, 1.87s/it]
15
  0%| | 13/30000 [00:34<15:24:15, 1.85s/it]
16
  0%| | 14/30000 [00:36<15:15:55, 1.83s/it]
17
  0%| | 15/30000 [00:37<15:12:53, 1.83s/it]
18
  0%| | 16/30000 [00:39<15:09:48, 1.82s/it]
19
  0%| | 17/30000 [00:41<15:07:04, 1.82s/it]
20
  0%| | 18/30000 [00:43<15:05:37, 1.81s/it]
21
  0%| | 19/30000 [00:45<15:05:00, 1.81s/it]
22
  0%| | 20/30000 [00:46<15:04:44, 1.81s/it]
23
 
24
  0%| | 20/30000 [00:46<15:04:44, 1.81s/it]
25
  0%| | 21/30000 [00:48<15:07:17, 1.82s/it]
26
  0%| | 22/30000 [00:50<15:10:17, 1.82s/it]
27
  0%| | 23/30000 [00:52<15:06:41, 1.81s/it]
28
  0%| | 24/30000 [00:54<15:05:48, 1.81s/it]
29
  0%| | 25/30000 [00:55<15:03:36, 1.81s/it]
30
  0%| | 26/30000 [00:57<15:04:39, 1.81s/it]
31
  0%| | 27/30000 [00:59<15:05:03, 1.81s/it]
32
  0%| | 28/30000 [01:01<15:05:37, 1.81s/it]
33
  0%| | 29/30000 [01:03<15:05:24, 1.81s/it]
34
  0%| | 30/30000 [01:05<15:05:52, 1.81s/it]
35
 
36
  0%| | 30/30000 [01:05<15:05:52, 1.81s/it]
37
  0%| | 31/30000 [01:06<15:03:54, 1.81s/it]
38
  0%| | 32/30000 [01:08<15:04:22, 1.81s/it]
39
  0%| | 33/30000 [01:10<15:04:56, 1.81s/it]
40
  0%| | 34/30000 [01:12<15:05:47, 1.81s/it]
41
  0%| | 35/30000 [01:14<15:06:06, 1.81s/it]
42
  0%| | 36/30000 [01:15<15:06:55, 1.82s/it]
43
  0%| | 37/30000 [01:17<15:07:31, 1.82s/it]
44
  0%| | 38/30000 [01:19<15:10:34, 1.82s/it]
45
  0%| | 39/30000 [01:21<15:09:30, 1.82s/it]
46
  0%| | 40/30000 [01:23<15:10:54, 1.82s/it]
47
 
48
  0%| | 40/30000 [01:23<15:10:54, 1.82s/it]
49
  0%| | 41/30000 [01:25<15:14:52, 1.83s/it]
50
  0%| | 42/30000 [01:26<15:13:02, 1.83s/it]
51
  0%| | 43/30000 [01:28<15:14:17, 1.83s/it]
52
  0%| | 44/30000 [01:30<15:19:36, 1.84s/it]
53
  0%| | 45/30000 [01:32<15:18:50, 1.84s/it]
54
  0%| | 46/30000 [01:34<15:19:09, 1.84s/it]
55
  0%| | 47/30000 [01:36<15:17:18, 1.84s/it]
56
  0%| | 48/30000 [01:37<15:16:58, 1.84s/it]
57
  0%| | 49/30000 [01:39<15:15:04, 1.83s/it]
58
  0%| | 50/30000 [01:41<15:16:32, 1.84s/it]
59
 
60
  0%| | 50/30000 [01:41<15:16:32, 1.84s/it]
61
  0%| | 51/30000 [01:43<15:18:53, 1.84s/it]
62
  0%| | 52/30000 [01:45<15:22:58, 1.85s/it]
63
  0%| | 53/30000 [01:47<15:19:36, 1.84s/it]
64
  0%| | 54/30000 [01:48<15:17:48, 1.84s/it]
65
  0%| | 55/30000 [01:50<15:17:19, 1.84s/it]
66
  0%| | 56/30000 [01:52<15:16:55, 1.84s/it]
67
  0%| | 57/30000 [01:54<15:16:19, 1.84s/it]
68
  0%| | 58/30000 [01:56<15:16:38, 1.84s/it]
69
  0%| | 59/30000 [01:58<15:16:30, 1.84s/it]
70
  0%| | 60/30000 [02:00<15:18:29, 1.84s/it]
71
 
72
  0%| | 60/30000 [02:00<15:18:29, 1.84s/it]
73
  0%| | 61/30000 [02:01<15:18:43, 1.84s/it]
74
  0%| | 62/30000 [02:03<15:19:39, 1.84s/it]
75
  0%| | 63/30000 [02:05<15:18:38, 1.84s/it]
 
1
+ [robosuite WARNING] No private macro file found! (macros.py:57)
2
+ [robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
3
+ [robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
4
+ [robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
5
+ [robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
6
+ /home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
7
+ check_for_updates()
8
+ `use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
9
+ WARNING: mimicgen environments not imported since mimicgen is not installed!
10
+
11
+ ==================================================
12
+ GR00T FINE-TUNING CONFIGURATION:
13
+ ==================================================
14
+ config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse_phase3_only.yaml
15
+ dataset_soup: None
16
+ output_dir: /tmp/gr00t
17
+ output_root: None
18
+ data_config: panda_omron
19
+ batch_size: 64
20
+ max_steps: 300000
21
+ num_gpus: 1
22
+ save_steps: 20000
23
+ run_name: None
24
+ save_total_limit: 100
25
+ seed: 42
26
+ base_model_path: nvidia/GR00T-N1.5-3B
27
+ tune_llm: False
28
+ tune_visual: False
29
+ tune_projector: True
30
+ tune_diffusion_model: True
31
+ resume: False
32
+ learning_rate: 3e-05
33
+ weight_decay: 1e-05
34
+ warmup_ratio: 0.05
35
+ lora_rank: 0
36
+ lora_alpha: 16
37
+ lora_dropout: 0.1
38
+ lora_full_model: False
39
+ dataloader_num_workers: 8
40
+ report_to: wandb
41
+ embodiment_tag: new_embodiment
42
+ video_backend: opencv
43
+ balance_dataset_weights: True
44
+ balance_trajectory_weights: True
45
+ ds_weights_alpha: 0.4
46
+ ==================================================
47
+
48
+ Using 1 GPUs
49
+
50
+ ================================================================================
51
+ Starting sweep branch: mgd_token_mask_ratio_0.3
52
+ Sweep vars: {'mgd_token_mask_ratio': 0.3}
53
+ ================================================================================
54
+
55
+ --------------------------------------------------------------------------------
56
+ Running phase 1: phase3_fm_mgd_fixed_0p5
57
+ Policy type: groot_mgd
58
+ Base model path: /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
59
+ Output dir: /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3
60
+ Trainable preset: None
61
+ Policy overrides: {'mgd_enabled': True, 'mgd_fm_loss_weight': 1.0, 'mgd_loss_weight': 0.5, 'mgd_token_mask_ratio': 0.3, 'mgd_use_cosine_loss': True, 'mgd_use_mse_loss': True}
62
+ --------------------------------------------------------------------------------
63
+
64
+ [{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
65
+ self.statistics[key] = torch.tensor(value)
66
+
67
+ Using 100 subset demos for filter_key: 100_demos
68
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
69
+ Using 100 subset demos for filter_key: 100_demos
70
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
71
+ Using 100 subset demos for filter_key: 100_demos
72
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
73
+ Using 100 subset demos for filter_key: 100_demos
74
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
75
+ Using 100 subset demos for filter_key: 100_demos
76
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
77
+ Using 100 subset demos for filter_key: 100_demos
78
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
79
+ Using 100 subset demos for filter_key: 100_demos
80
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
81
+ Using 100 subset demos for filter_key: 100_demos
82
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
83
+ Using 100 subset demos for filter_key: 100_demos
84
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
85
+ Using 100 subset demos for filter_key: 100_demos
86
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
87
+ Using 100 subset demos for filter_key: 100_demos
88
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
89
+ Using 100 subset demos for filter_key: 100_demos
90
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
91
+ Using 100 subset demos for filter_key: 100_demos
92
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
93
+ Using 100 subset demos for filter_key: 100_demos
94
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
95
+ Using 100 subset demos for filter_key: 100_demos
96
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
97
+ Using 100 subset demos for filter_key: 100_demos
98
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
99
+ Using 100 subset demos for filter_key: 100_demos
100
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
101
+ Using 100 subset demos for filter_key: 100_demos
102
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
103
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
104
+ Using 100 subset demos for filter_key: 100_demos
105
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
106
+ Using 100 subset demos for filter_key: 100_demos
107
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
108
+ Using 100 subset demos for filter_key: 100_demos
109
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
110
+ Using 100 subset demos for filter_key: 100_demos
111
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
112
+ Using 100 subset demos for filter_key: 100_demos
113
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
114
+ Using 100 subset demos for filter_key: 100_demos
115
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
116
+ Using 100 subset demos for filter_key: 100_demos
117
+ Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
118
+ dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
119
+ 0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
120
+ 0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
121
+ 0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
122
+ 0.75517122 0.7973985 ]
123
+ Loaded 26 datasets
124
+ Loading pretrained dual brain from /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
125
+ Tune backbone vision tower: False
126
+ Tune backbone LLM: False
127
+ Tune action head projector: True
128
+ Tune action head DiT: True
129
+ Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
130
+ Tune backbone llm: False
131
+ Tune backbone visual: True
132
+ Total number of DiT parameters: 550386688
133
+ Total number of SelfAttentionTransformer parameters: 201433088
134
+ Tune action head projector: True
135
+ Tune action head diffusion model: True
136
+
137
+ Tune backbone llm: False
138
+ Tune backbone visual: False
139
+ Warning: No backbone trainable parameters found.
140
+ Tune action head projector: True
141
+ Tune action head diffusion model: True
142
+ Trainable summary: {'trainable_preset': 'default', 'trainable_modules': 'action_head.future_tokens(49,152), action_head.vlln(4,096), action_head.vl_self_attention(201,433,088), action_head.state_encoder(52,510,720), action_head.action_encoder(228,212,736), action_head.action_decoder(34,636,800), action_head.position_embedding(1,572,864), action_head.model(550,386,688), sequence_mgd_head(1,574,913)', 'trainable_module_names': 'action_head.future_tokens, action_head.vlln, action_head.vl_self_attention, action_head.state_encoder, action_head.action_encoder, action_head.action_decoder, action_head.position_embedding, action_head.model, sequence_mgd_head', 'trainable_module_param_counts': {'action_head.future_tokens': 49152, 'action_head.vlln': 4096, 'action_head.vl_self_attention': 201433088, 'action_head.state_encoder': 52510720, 'action_head.action_encoder': 228212736, 'action_head.action_decoder': 34636800, 'action_head.position_embedding': 1572864, 'action_head.model': 550386688, 'sequence_mgd_head': 1574913}, 'trainable_param_count': 1070381057, 'total_param_count': 2738321857, 'trainable_param_ratio': 0.3908894253112628}
143
+ Run name: mgd_token_mask_ratio_0.3_phase3
144
+ train dataloader length: 6873
145
+ train dataset length: 439854
146
+ GPU memory before training: 7.117771625518799 GB
147
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
148
+ wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
149
+ wandb: setting up run 4hsojbcy
150
+ wandb: Tracking run with wandb version 0.25.0
151
+ wandb: Run data is saved locally in /home/seonho/wandb/run-20260614_013832-4hsojbcy
152
+ wandb: Run `wandb offline` to turn off syncing.
153
+ wandb: Syncing run mgd_token_mask_ratio_0.3_phase3
154
+ wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
155
+ wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/4hsojbcy
156
+ TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/runs
157
+
158
  0%| | 0/30000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
159
+
160
  0%| | 1/30000 [00:12<105:19:47, 12.64s/it]
161
  0%| | 2/30000 [00:14<52:20:08, 6.28s/it]
162
  0%| | 3/30000 [00:16<35:11:45, 4.22s/it]
163
  0%| | 4/30000 [00:18<27:10:39, 3.26s/it]
164
  0%| | 5/30000 [00:19<22:43:51, 2.73s/it]
165
  0%| | 6/30000 [00:21<20:05:04, 2.41s/it]
166
  0%| | 7/30000 [00:23<18:23:19, 2.21s/it]
167
  0%| | 8/30000 [00:25<17:24:08, 2.09s/it]
168
  0%| | 9/30000 [00:27<16:37:23, 2.00s/it]
169
  0%| | 10/30000 [00:28<16:10:55, 1.94s/it]
170
 
171
  0%| | 10/30000 [00:28<16:10:55, 1.94s/it]
172
  0%| | 11/30000 [00:30<15:49:20, 1.90s/it]
173
  0%| | 12/30000 [00:32<15:33:32, 1.87s/it]
174
  0%| | 13/30000 [00:34<15:24:15, 1.85s/it]
175
  0%| | 14/30000 [00:36<15:15:55, 1.83s/it]
176
  0%| | 15/30000 [00:37<15:12:53, 1.83s/it]
177
  0%| | 16/30000 [00:39<15:09:48, 1.82s/it]
178
  0%| | 17/30000 [00:41<15:07:04, 1.82s/it]
179
  0%| | 18/30000 [00:43<15:05:37, 1.81s/it]
180
  0%| | 19/30000 [00:45<15:05:00, 1.81s/it]
181
  0%| | 20/30000 [00:46<15:04:44, 1.81s/it]
182
 
183
  0%| | 20/30000 [00:46<15:04:44, 1.81s/it]
184
  0%| | 21/30000 [00:48<15:07:17, 1.82s/it]
185
  0%| | 22/30000 [00:50<15:10:17, 1.82s/it]
186
  0%| | 23/30000 [00:52<15:06:41, 1.81s/it]
187
  0%| | 24/30000 [00:54<15:05:48, 1.81s/it]
188
  0%| | 25/30000 [00:55<15:03:36, 1.81s/it]
189
  0%| | 26/30000 [00:57<15:04:39, 1.81s/it]
190
  0%| | 27/30000 [00:59<15:05:03, 1.81s/it]
191
  0%| | 28/30000 [01:01<15:05:37, 1.81s/it]
192
  0%| | 29/30000 [01:03<15:05:24, 1.81s/it]
193
  0%| | 30/30000 [01:05<15:05:52, 1.81s/it]
194
 
195
  0%| | 30/30000 [01:05<15:05:52, 1.81s/it]
196
  0%| | 31/30000 [01:06<15:03:54, 1.81s/it]
197
  0%| | 32/30000 [01:08<15:04:22, 1.81s/it]
198
  0%| | 33/30000 [01:10<15:04:56, 1.81s/it]
199
  0%| | 34/30000 [01:12<15:05:47, 1.81s/it]
200
  0%| | 35/30000 [01:14<15:06:06, 1.81s/it]
201
  0%| | 36/30000 [01:15<15:06:55, 1.82s/it]
202
  0%| | 37/30000 [01:17<15:07:31, 1.82s/it]
203
  0%| | 38/30000 [01:19<15:10:34, 1.82s/it]
204
  0%| | 39/30000 [01:21<15:09:30, 1.82s/it]
205
  0%| | 40/30000 [01:23<15:10:54, 1.82s/it]
206
 
207
  0%| | 40/30000 [01:23<15:10:54, 1.82s/it]
208
  0%| | 41/30000 [01:25<15:14:52, 1.83s/it]
209
  0%| | 42/30000 [01:26<15:13:02, 1.83s/it]
210
  0%| | 43/30000 [01:28<15:14:17, 1.83s/it]
211
  0%| | 44/30000 [01:30<15:19:36, 1.84s/it]
212
  0%| | 45/30000 [01:32<15:18:50, 1.84s/it]
213
  0%| | 46/30000 [01:34<15:19:09, 1.84s/it]
214
  0%| | 47/30000 [01:36<15:17:18, 1.84s/it]
215
  0%| | 48/30000 [01:37<15:16:58, 1.84s/it]
216
  0%| | 49/30000 [01:39<15:15:04, 1.83s/it]
217
  0%| | 50/30000 [01:41<15:16:32, 1.84s/it]
218
 
219
  0%| | 50/30000 [01:41<15:16:32, 1.84s/it]
220
  0%| | 51/30000 [01:43<15:18:53, 1.84s/it]
221
  0%| | 52/30000 [01:45<15:22:58, 1.85s/it]
222
  0%| | 53/30000 [01:47<15:19:36, 1.84s/it]
223
  0%| | 54/30000 [01:48<15:17:48, 1.84s/it]
224
  0%| | 55/30000 [01:50<15:17:19, 1.84s/it]
225
  0%| | 56/30000 [01:52<15:16:55, 1.84s/it]
226
  0%| | 57/30000 [01:54<15:16:19, 1.84s/it]
227
  0%| | 58/30000 [01:56<15:16:38, 1.84s/it]
228
  0%| | 59/30000 [01:58<15:16:30, 1.84s/it]
229
  0%| | 60/30000 [02:00<15:18:29, 1.84s/it]
230
 
231
  0%| | 60/30000 [02:00<15:18:29, 1.84s/it]
232
  0%| | 61/30000 [02:01<15:18:43, 1.84s/it]
233
  0%| | 62/30000 [02:03<15:19:39, 1.84s/it]
234
  0%| | 63/30000 [02:05<15:18:38, 1.84s/it]
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3_only_train_gpu01.log ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/resolved_config.yaml ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_loss_cosine_mse_phase3
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.3
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase3_fm_mgd_fixed_0p5
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ tune_llm: false
21
+ tune_visual: false
22
+ tune_projector: true
23
+ tune_diffusion_model: true
24
+ losses:
25
+ mgd_enabled: true
26
+ mgd_fm_loss_weight: 1.0
27
+ mgd_loss_weight: 0.5
28
+ mgd_token_mask_ratio: 0.3
29
+ mgd_use_cosine_loss: true
30
+ mgd_use_mse_loss: true
31
+ resolved_sweep:
32
+ mgd_token_mask_ratio: 0.3
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/mgd_v2_mask_ratio_ablation_0p1.yaml ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_mask_ratio_ablation
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.1
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
31
+ mgd_use_cosine_loss: true
32
+ mgd_use_mse_loss: false
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
46
+ mgd_use_cosine_loss: true
47
+ mgd_use_mse_loss: false
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/config.json ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_MGD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "mgd_enabled": true,
61
+ "mgd_fm_loss_weight": 0.0,
62
+ "mgd_loss_weight": 1.0,
63
+ "mgd_loss_weight_end": 0.0,
64
+ "mgd_loss_weight_schedule": null,
65
+ "mgd_loss_weight_start": 0.05,
66
+ "mgd_pretrained_projector_path": null,
67
+ "mgd_sequence_hidden_dim": 512,
68
+ "mgd_target_dim": 512,
69
+ "mgd_target_pooling": "flatten",
70
+ "mgd_target_projection": "frozen_random",
71
+ "mgd_token_mask_ratio": 0.1,
72
+ "mgd_use_cosine_loss": true,
73
+ "mgd_use_mse_loss": false,
74
+ "model_dtype": "float32",
75
+ "model_type": "gr00t_n1_5",
76
+ "torch_dtype": "bfloat16",
77
+ "transformers_version": "4.51.3"
78
+ }
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/trainer_state.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/config.json ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_MGD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "mgd_enabled": true,
61
+ "mgd_fm_loss_weight": 1.0,
62
+ "mgd_loss_weight": 0.5,
63
+ "mgd_loss_weight_end": 0.0,
64
+ "mgd_loss_weight_schedule": null,
65
+ "mgd_loss_weight_start": 0.05,
66
+ "mgd_pretrained_projector_path": null,
67
+ "mgd_sequence_hidden_dim": 512,
68
+ "mgd_target_dim": 512,
69
+ "mgd_target_pooling": "flatten",
70
+ "mgd_target_projection": "frozen_random",
71
+ "mgd_token_mask_ratio": 0.1,
72
+ "mgd_use_cosine_loss": true,
73
+ "mgd_use_mse_loss": false,
74
+ "model_dtype": "float32",
75
+ "model_type": "gr00t_n1_5",
76
+ "torch_dtype": "bfloat16",
77
+ "transformers_version": "4.51.3"
78
+ }
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/resolved_config.yaml ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_mask_ratio_ablation
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.1
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: 0.1
31
+ mgd_use_cosine_loss: true
32
+ mgd_use_mse_loss: false
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: 0.1
46
+ mgd_use_cosine_loss: true
47
+ mgd_use_mse_loss: false
48
+ resolved_sweep:
49
+ mgd_token_mask_ratio: 0.1
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/mgd_v2_mask_ratio_ablation_0p2.yaml ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_mask_ratio_ablation
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.2
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
31
+ mgd_use_cosine_loss: true
32
+ mgd_use_mse_loss: false
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
46
+ mgd_use_cosine_loss: true
47
+ mgd_use_mse_loss: false
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase3/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/resolved_config.yaml ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_mask_ratio_ablation
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.2
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: 0.2
31
+ mgd_use_cosine_loss: true
32
+ mgd_use_mse_loss: false
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: 0.2
46
+ mgd_use_cosine_loss: true
47
+ mgd_use_mse_loss: false
48
+ resolved_sweep:
49
+ mgd_token_mask_ratio: 0.2
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/mgd_v2_mask_ratio_ablation_0p4.yaml ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_mask_ratio_ablation
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.4
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
31
+ mgd_use_cosine_loss: true
32
+ mgd_use_mse_loss: false
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
46
+ mgd_use_cosine_loss: true
47
+ mgd_use_mse_loss: false
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/phase2/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/resolved_config.yaml ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_mask_ratio_ablation
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.4
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: 0.4
31
+ mgd_use_cosine_loss: true
32
+ mgd_use_mse_loss: false
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: 0.4
46
+ mgd_use_cosine_loss: true
47
+ mgd_use_mse_loss: false
48
+ resolved_sweep:
49
+ mgd_token_mask_ratio: 0.4
rkd_v2_1/rkd_v2_1_angle/default/phase2/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
rkd_v2_1/rkd_v2_1_angle/default/resolved_config.yaml ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: rkd_v2_1_angle_phase2_phase3_aux_fixed_0p5
2
+ policy_type: groot_rkd_v2
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/rkd_v2_1/rkd_v2_1_angle
5
+ dataset:
6
+ dataset_soup: my_atomic26_human
7
+ training:
8
+ num_gpus: 1
9
+ batch_size: 128
10
+ seed: 42
11
+ model: null
12
+ phases:
13
+ - name: phase2_rkd_a_only
14
+ max_steps: 60000
15
+ save_steps: 30000
16
+ trainable:
17
+ preset: processing_line_only
18
+ tune_llm: false
19
+ tune_visual: false
20
+ tune_projector: false
21
+ tune_diffusion_model: false
22
+ losses:
23
+ rkd_enabled: true
24
+ rkd_fm_loss_weight: 0.0
25
+ rkd_loss_weight: 1.0
26
+ rkd_relation_mode: token_pair_mean
27
+ rkd_loss_type: angle
28
+ rkd_exclude_diagonal: true
29
+ - name: phase3_fm_rkd_a_fixed_0p5
30
+ max_steps: 80000
31
+ save_steps: 20000
32
+ trainable:
33
+ tune_llm: false
34
+ tune_visual: false
35
+ tune_projector: true
36
+ tune_diffusion_model: true
37
+ losses:
38
+ rkd_enabled: true
39
+ rkd_fm_loss_weight: 1.0
40
+ rkd_loss_weight: 0.5
41
+ rkd_relation_mode: token_pair_mean
42
+ rkd_loss_type: angle
43
+ rkd_exclude_diagonal: true
44
+ resolved_sweep: {}
rkd_v2_1/rkd_v2_1_angle/default/rkd_v2_1_a_stepup.yaml ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: rkd_v2_1_angle_phase2_phase3_aux_fixed_0p5
2
+ policy_type: groot_rkd_v2
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/rkd_v2_1/rkd_v2_1_angle
5
+ dataset:
6
+ dataset_soup: my_atomic26_human
7
+ training:
8
+ num_gpus: 1
9
+ batch_size: 128
10
+ seed: 42
11
+ model: null
12
+ phases:
13
+ - name: phase2_rkd_a_only
14
+ max_steps: 60000
15
+ save_steps: 30000
16
+ trainable:
17
+ preset: processing_line_only
18
+ tune_llm: false
19
+ tune_visual: false
20
+ tune_projector: false
21
+ tune_diffusion_model: false
22
+ losses:
23
+ rkd_enabled: true
24
+ rkd_fm_loss_weight: 0.0
25
+ rkd_loss_weight: 1.0
26
+ rkd_relation_mode: token_pair_mean
27
+ rkd_loss_type: angle
28
+ rkd_exclude_diagonal: true
29
+ - name: phase3_fm_rkd_a_fixed_0p5
30
+ max_steps: 80000
31
+ save_steps: 20000
32
+ trainable:
33
+ tune_llm: false
34
+ tune_visual: false
35
+ tune_projector: true
36
+ tune_diffusion_model: true
37
+ losses:
38
+ rkd_enabled: true
39
+ rkd_fm_loss_weight: 1.0
40
+ rkd_loss_weight: 0.5
41
+ rkd_relation_mode: token_pair_mean
42
+ rkd_loss_type: angle
43
+ rkd_exclude_diagonal: true
rkd_v2_1/rkd_v2_1_distance_angle/default/phase2/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
rkd_v2_1/rkd_v2_1_distance_angle/default/resolved_config.yaml ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: rkd_v2_1_distance_angle_phase2_phase3_aux_fixed_0p5
2
+ policy_type: groot_rkd_v2
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/rkd_v2_1/rkd_v2_1_distance_angle
5
+ dataset:
6
+ dataset_soup: my_atomic26_human
7
+ training:
8
+ num_gpus: 1
9
+ batch_size: 128
10
+ seed: 42
11
+ model: null
12
+ phases:
13
+ - name: phase2_rkd_da_only
14
+ max_steps: 60000
15
+ save_steps: 30000
16
+ trainable:
17
+ preset: processing_line_only
18
+ tune_llm: false
19
+ tune_visual: false
20
+ tune_projector: false
21
+ tune_diffusion_model: false
22
+ losses:
23
+ rkd_enabled: true
24
+ rkd_fm_loss_weight: 0.0
25
+ rkd_loss_weight: 1.0
26
+ rkd_relation_mode: token_pair_mean
27
+ rkd_loss_type: distance_angle
28
+ rkd_distance_loss_weight: 1.0
29
+ rkd_angle_loss_weight: 2.0
30
+ rkd_exclude_diagonal: true
31
+ - name: phase3_fm_rkd_da_fixed_0p5
32
+ max_steps: 80000
33
+ save_steps: 20000
34
+ trainable:
35
+ tune_llm: false
36
+ tune_visual: false
37
+ tune_projector: true
38
+ tune_diffusion_model: true
39
+ losses:
40
+ rkd_enabled: true
41
+ rkd_fm_loss_weight: 1.0
42
+ rkd_loss_weight: 0.5
43
+ rkd_relation_mode: token_pair_mean
44
+ rkd_loss_type: distance_angle
45
+ rkd_distance_loss_weight: 1.0
46
+ rkd_angle_loss_weight: 2.0
47
+ rkd_exclude_diagonal: true
48
+ resolved_sweep: {}
rkd_v2_1/rkd_v2_1_distance_angle/default/rkd_v2_1_da_stepup.yaml ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: rkd_v2_1_distance_angle_phase2_phase3_aux_fixed_0p5
2
+ policy_type: groot_rkd_v2
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/rkd_v2_1/rkd_v2_1_distance_angle
5
+ dataset:
6
+ dataset_soup: my_atomic26_human
7
+ training:
8
+ num_gpus: 1
9
+ batch_size: 128
10
+ seed: 42
11
+ model: null
12
+ phases:
13
+ - name: phase2_rkd_da_only
14
+ max_steps: 60000
15
+ save_steps: 30000
16
+ trainable:
17
+ preset: processing_line_only
18
+ tune_llm: false
19
+ tune_visual: false
20
+ tune_projector: false
21
+ tune_diffusion_model: false
22
+ losses:
23
+ rkd_enabled: true
24
+ rkd_fm_loss_weight: 0.0
25
+ rkd_loss_weight: 1.0
26
+ rkd_relation_mode: token_pair_mean
27
+ rkd_loss_type: distance_angle
28
+ rkd_distance_loss_weight: 1.0
29
+ rkd_angle_loss_weight: 2.0
30
+ rkd_exclude_diagonal: true
31
+ - name: phase3_fm_rkd_da_fixed_0p5
32
+ max_steps: 80000
33
+ save_steps: 20000
34
+ trainable:
35
+ tune_llm: false
36
+ tune_visual: false
37
+ tune_projector: true
38
+ tune_diffusion_model: true
39
+ losses:
40
+ rkd_enabled: true
41
+ rkd_fm_loss_weight: 1.0
42
+ rkd_loss_weight: 0.5
43
+ rkd_relation_mode: token_pair_mean
44
+ rkd_loss_type: distance_angle
45
+ rkd_distance_loss_weight: 1.0
46
+ rkd_angle_loss_weight: 2.0
47
+ rkd_exclude_diagonal: true