Add files using upload-large-folder tool
Browse files- VLM_Only_v2/RKD_A_VLM_Only/default/phase2/experiment_cfg/metadata.json +431 -0
- VLM_Only_v2/RKD_A_VLM_Only/default/resolved_config.yaml +44 -0
- VLM_Only_v2/RKD_A_VLM_Only/default/rkd_v2_a_vlm_only.yaml +43 -0
- VLM_Only_v2/RKD_A_VLM_Only/default/train_bs16_gpu2_3_probe.log +822 -0
- VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_3_60k.log +1088 -0
- VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_3_probe.log +997 -0
- VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_single_probe.log +401 -0
- VLM_Only_v2/RKD_A_VLM_Only/default/train_bs64_gpu2_3.log +1095 -0
- VLM_Only_v2/RKD_A_VLM_Only/default/train_bs64_gpu2_3_lmheadfreeze.log +1087 -0
- experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p1_train.log +0 -0
- experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p2_train.log +0 -0
- experiment_cfg/processing_line_only_v2/MGD_v2/train.log +0 -0
- experiment_cfg/processing_line_only_v2/MGD_v2/train_mse.log +0 -0
- processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/trainer_state.json +0 -0
- processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/trainer_state.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_cosine_mse.yaml +47 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_cosine_mse_phase3_only.yaml +30 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/config.json +78 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/experiment_cfg/metadata.json +431 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/model.safetensors.index.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/trainer_state.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/config.json +78 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/experiment_cfg/metadata.json +431 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/model.safetensors.index.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/trainer_state.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3_only_train.log +158 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3_only_train_gpu01.log +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/resolved_config.yaml +32 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/mgd_v2_mask_ratio_ablation_0p1.yaml +47 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/config.json +78 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/experiment_cfg/metadata.json +431 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/model.safetensors.index.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/trainer_state.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/config.json +78 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/experiment_cfg/metadata.json +431 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/model.safetensors.index.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/resolved_config.yaml +49 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/mgd_v2_mask_ratio_ablation_0p2.yaml +47 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/model.safetensors.index.json +0 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase3/experiment_cfg/metadata.json +431 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/resolved_config.yaml +49 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/mgd_v2_mask_ratio_ablation_0p4.yaml +47 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/phase2/experiment_cfg/metadata.json +431 -0
- processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/resolved_config.yaml +49 -0
- rkd_v2_1/rkd_v2_1_angle/default/phase2/experiment_cfg/metadata.json +431 -0
- rkd_v2_1/rkd_v2_1_angle/default/resolved_config.yaml +44 -0
- rkd_v2_1/rkd_v2_1_angle/default/rkd_v2_1_a_stepup.yaml +43 -0
- rkd_v2_1/rkd_v2_1_distance_angle/default/phase2/experiment_cfg/metadata.json +431 -0
- rkd_v2_1/rkd_v2_1_distance_angle/default/resolved_config.yaml +48 -0
- rkd_v2_1/rkd_v2_1_distance_angle/default/rkd_v2_1_da_stepup.yaml +47 -0
VLM_Only_v2/RKD_A_VLM_Only/default/phase2/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
VLM_Only_v2/RKD_A_VLM_Only/default/resolved_config.yaml
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: rkd_v2_angle_vlm_only_phase2
|
| 2 |
+
policy_type: groot_rkd_v2_raw
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only
|
| 5 |
+
dataset:
|
| 6 |
+
dataset_soup: my_atomic26_human
|
| 7 |
+
training:
|
| 8 |
+
num_gpus: 2
|
| 9 |
+
batch_size: 32
|
| 10 |
+
seed: 42
|
| 11 |
+
model: null
|
| 12 |
+
phases:
|
| 13 |
+
- name: phase2_rkd_a_vlm_only
|
| 14 |
+
max_steps: 60000
|
| 15 |
+
save_steps: 0
|
| 16 |
+
trainable:
|
| 17 |
+
preset: freeze_processing_line
|
| 18 |
+
tune_llm: true
|
| 19 |
+
tune_visual: true
|
| 20 |
+
tune_projector: false
|
| 21 |
+
tune_diffusion_model: false
|
| 22 |
+
losses:
|
| 23 |
+
rkd_enabled: true
|
| 24 |
+
rkd_fm_loss_weight: 0.0
|
| 25 |
+
rkd_loss_weight: 1.0
|
| 26 |
+
rkd_relation_mode: token_pair_mean
|
| 27 |
+
rkd_loss_type: angle
|
| 28 |
+
rkd_exclude_diagonal: true
|
| 29 |
+
- name: phase3_fm_rkd_a_fixed_0p5
|
| 30 |
+
max_steps: 60000
|
| 31 |
+
save_steps: 0
|
| 32 |
+
trainable:
|
| 33 |
+
tune_llm: false
|
| 34 |
+
tune_visual: false
|
| 35 |
+
tune_projector: true
|
| 36 |
+
tune_diffusion_model: true
|
| 37 |
+
losses:
|
| 38 |
+
rkd_enabled: true
|
| 39 |
+
rkd_fm_loss_weight: 1.0
|
| 40 |
+
rkd_loss_weight: 0.5
|
| 41 |
+
rkd_relation_mode: token_pair_mean
|
| 42 |
+
rkd_loss_type: angle
|
| 43 |
+
rkd_exclude_diagonal: true
|
| 44 |
+
resolved_sweep: {}
|
VLM_Only_v2/RKD_A_VLM_Only/default/rkd_v2_a_vlm_only.yaml
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: rkd_v2_angle_vlm_only_phase2
|
| 2 |
+
policy_type: groot_rkd_v2_raw
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only
|
| 5 |
+
dataset:
|
| 6 |
+
dataset_soup: my_atomic26_human
|
| 7 |
+
training:
|
| 8 |
+
num_gpus: 2
|
| 9 |
+
batch_size: 32
|
| 10 |
+
seed: 42
|
| 11 |
+
model: null
|
| 12 |
+
phases:
|
| 13 |
+
- name: phase2_rkd_a_vlm_only
|
| 14 |
+
max_steps: 60000
|
| 15 |
+
save_steps: 0
|
| 16 |
+
trainable:
|
| 17 |
+
preset: freeze_processing_line
|
| 18 |
+
tune_llm: true
|
| 19 |
+
tune_visual: true
|
| 20 |
+
tune_projector: false
|
| 21 |
+
tune_diffusion_model: false
|
| 22 |
+
losses:
|
| 23 |
+
rkd_enabled: true
|
| 24 |
+
rkd_fm_loss_weight: 0.0
|
| 25 |
+
rkd_loss_weight: 1.0
|
| 26 |
+
rkd_relation_mode: token_pair_mean
|
| 27 |
+
rkd_loss_type: angle
|
| 28 |
+
rkd_exclude_diagonal: true
|
| 29 |
+
- name: phase3_fm_rkd_a_fixed_0p5
|
| 30 |
+
max_steps: 60000
|
| 31 |
+
save_steps: 0
|
| 32 |
+
trainable:
|
| 33 |
+
tune_llm: false
|
| 34 |
+
tune_visual: false
|
| 35 |
+
tune_projector: true
|
| 36 |
+
tune_diffusion_model: true
|
| 37 |
+
losses:
|
| 38 |
+
rkd_enabled: true
|
| 39 |
+
rkd_fm_loss_weight: 1.0
|
| 40 |
+
rkd_loss_weight: 0.5
|
| 41 |
+
rkd_relation_mode: token_pair_mean
|
| 42 |
+
rkd_loss_type: angle
|
| 43 |
+
rkd_exclude_diagonal: true
|
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs16_gpu2_3_probe.log
ADDED
|
@@ -0,0 +1,822 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/30000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
|
|
|
|
|
|
|
| 1 |
0%| | 1/30000 [00:05<43:01:49, 5.16s/it][rank1]: Traceback (most recent call last):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 2 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 3 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 4 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 5 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 6 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 7 |
+
check_for_updates()
|
| 8 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 9 |
+
|
| 10 |
+
*****************************************
|
| 11 |
+
Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
|
| 12 |
+
*****************************************
|
| 13 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 14 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 15 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 16 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 17 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 18 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 19 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 20 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 21 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 22 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 23 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 24 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 25 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 26 |
+
check_for_updates()
|
| 27 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 28 |
+
check_for_updates()
|
| 29 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 30 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 31 |
+
|
| 32 |
+
==================================================
|
| 33 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 34 |
+
==================================================
|
| 35 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 36 |
+
dataset_soup: None
|
| 37 |
+
output_dir: /tmp/gr00t
|
| 38 |
+
output_root: None
|
| 39 |
+
data_config: panda_omron
|
| 40 |
+
batch_size: 16
|
| 41 |
+
max_steps: 300000
|
| 42 |
+
num_gpus: 2
|
| 43 |
+
save_steps: 20000
|
| 44 |
+
run_name: None
|
| 45 |
+
save_total_limit: 100
|
| 46 |
+
seed: 42
|
| 47 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 48 |
+
tune_llm: False
|
| 49 |
+
tune_visual: False
|
| 50 |
+
tune_projector: True
|
| 51 |
+
tune_diffusion_model: True
|
| 52 |
+
resume: False
|
| 53 |
+
learning_rate: 3e-05
|
| 54 |
+
weight_decay: 1e-05
|
| 55 |
+
warmup_ratio: 0.05
|
| 56 |
+
lora_rank: 0
|
| 57 |
+
lora_alpha: 16
|
| 58 |
+
lora_dropout: 0.1
|
| 59 |
+
lora_full_model: False
|
| 60 |
+
dataloader_num_workers: 8
|
| 61 |
+
report_to: wandb
|
| 62 |
+
embodiment_tag: new_embodiment
|
| 63 |
+
video_backend: opencv
|
| 64 |
+
balance_dataset_weights: True
|
| 65 |
+
balance_trajectory_weights: True
|
| 66 |
+
ds_weights_alpha: 0.4
|
| 67 |
+
==================================================
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
==================================================
|
| 71 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 72 |
+
==================================================
|
| 73 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 74 |
+
dataset_soup: None
|
| 75 |
+
output_dir: /tmp/gr00t
|
| 76 |
+
output_root: None
|
| 77 |
+
data_config: panda_omron
|
| 78 |
+
batch_size: 16
|
| 79 |
+
max_steps: 300000
|
| 80 |
+
num_gpus: 2
|
| 81 |
+
save_steps: 20000
|
| 82 |
+
run_name: None
|
| 83 |
+
save_total_limit: 100
|
| 84 |
+
seed: 42
|
| 85 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 86 |
+
tune_llm: False
|
| 87 |
+
tune_visual: False
|
| 88 |
+
tune_projector: True
|
| 89 |
+
tune_diffusion_model: True
|
| 90 |
+
resume: False
|
| 91 |
+
learning_rate: 3e-05
|
| 92 |
+
weight_decay: 1e-05
|
| 93 |
+
warmup_ratio: 0.05
|
| 94 |
+
lora_rank: 0
|
| 95 |
+
lora_alpha: 16
|
| 96 |
+
lora_dropout: 0.1
|
| 97 |
+
lora_full_model: False
|
| 98 |
+
dataloader_num_workers: 8
|
| 99 |
+
report_to: wandb
|
| 100 |
+
embodiment_tag: new_embodiment
|
| 101 |
+
video_backend: opencv
|
| 102 |
+
balance_dataset_weights: True
|
| 103 |
+
balance_trajectory_weights: True
|
| 104 |
+
ds_weights_alpha: 0.4
|
| 105 |
+
==================================================
|
| 106 |
+
|
| 107 |
+
Using 2 GPUs
|
| 108 |
+
|
| 109 |
+
================================================================================
|
| 110 |
+
Starting sweep branch: default
|
| 111 |
+
Sweep vars: {}
|
| 112 |
+
================================================================================
|
| 113 |
+
|
| 114 |
+
--------------------------------------------------------------------------------
|
| 115 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 116 |
+
Policy type: groot_rkd_v2_raw
|
| 117 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 118 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 119 |
+
Trainable preset: freeze_processing_line
|
| 120 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 121 |
+
--------------------------------------------------------------------------------
|
| 122 |
+
|
| 123 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 124 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 125 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 126 |
+
self.statistics[key] = torch.tensor(value)
|
| 127 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 128 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 129 |
+
Using 2 GPUs
|
| 130 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 131 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 132 |
+
|
| 133 |
+
================================================================================
|
| 134 |
+
Starting sweep branch: default
|
| 135 |
+
Sweep vars: {}
|
| 136 |
+
================================================================================
|
| 137 |
+
|
| 138 |
+
--------------------------------------------------------------------------------
|
| 139 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 140 |
+
Policy type: groot_rkd_v2_raw
|
| 141 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 142 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 143 |
+
Trainable preset: freeze_processing_line
|
| 144 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 145 |
+
--------------------------------------------------------------------------------
|
| 146 |
+
|
| 147 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 148 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 149 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 150 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 151 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 152 |
+
self.statistics[key] = torch.tensor(value)
|
| 153 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 154 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 155 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 156 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 157 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 158 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 159 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 160 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 161 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 162 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 163 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 164 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 165 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 166 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 167 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 168 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 169 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 170 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 171 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 172 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 173 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 174 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 175 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 176 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 177 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 178 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 179 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 180 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 181 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 182 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 183 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 184 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 185 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 186 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 187 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 188 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 189 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 190 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 191 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 192 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 193 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 194 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 195 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 196 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 197 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 198 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 199 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 200 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 201 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 202 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 203 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 204 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 205 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 206 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 207 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 208 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 209 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 210 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 211 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 212 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 213 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 214 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 215 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 216 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 217 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 218 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 219 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 220 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 221 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 222 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 223 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 224 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 225 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 226 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 227 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 228 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 229 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 230 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 231 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 232 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 233 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 234 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 235 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 236 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 237 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 238 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 239 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 240 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 241 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 242 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 243 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 244 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 245 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 246 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 247 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 248 |
+
0.75517122 0.7973985 ]
|
| 249 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 250 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 251 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 252 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 253 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 254 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 255 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 256 |
+
0.75517122 0.7973985 ]
|
| 257 |
+
Loaded 26 datasets
|
| 258 |
+
Loaded 26 datasets
|
| 259 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 260 |
+
Tune backbone vision tower: True
|
| 261 |
+
Tune backbone LLM: True
|
| 262 |
+
Tune action head projector: False
|
| 263 |
+
Tune action head DiT: False
|
| 264 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 265 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 266 |
+
Tune backbone vision tower: True
|
| 267 |
+
Tune backbone LLM: True
|
| 268 |
+
Tune action head projector: False
|
| 269 |
+
Tune action head DiT: False
|
| 270 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 271 |
+
Tune backbone llm: False
|
| 272 |
+
Tune backbone visual: True
|
| 273 |
+
Tune backbone llm: False
|
| 274 |
+
Tune backbone visual: True
|
| 275 |
+
Total number of DiT parameters: 550386688
|
| 276 |
+
Total number of DiT parameters: 550386688
|
| 277 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 278 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 279 |
+
Tune action head projector: True
|
| 280 |
+
Tune action head diffusion model: True
|
| 281 |
+
Tune action head projector: True
|
| 282 |
+
Tune action head diffusion model: True
|
| 283 |
+
|
| 284 |
+
|
| 285 |
+
Tune backbone llm: True
|
| 286 |
+
Tune backbone visual: True
|
| 287 |
+
Tune backbone llm: True
|
| 288 |
+
Tune backbone visual: True
|
| 289 |
+
Tune action head projector: False
|
| 290 |
+
Tune action head diffusion model: False
|
| 291 |
+
Tune action head projector: False
|
| 292 |
+
Tune action head diffusion model: False
|
| 293 |
+
Action head trainable parameter: future_tokens.weight
|
| 294 |
+
Action head trainable parameter: vlln.weight
|
| 295 |
+
Action head trainable parameter: vlln.bias
|
| 296 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 297 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 298 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 299 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 300 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 301 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 302 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 303 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 304 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weightAction head trainable parameter: future_tokens.weight
|
| 305 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 306 |
+
|
| 307 |
+
Action head trainable parameter: vlln.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 308 |
+
|
| 309 |
+
Action head trainable parameter: vlln.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 310 |
+
|
| 311 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 312 |
+
|
| 313 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 314 |
+
|
| 315 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 316 |
+
|
| 317 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 318 |
+
|
| 319 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 320 |
+
|
| 321 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 322 |
+
|
| 323 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 324 |
+
|
| 325 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 326 |
+
|
| 327 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 328 |
+
|
| 329 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 330 |
+
|
| 331 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 332 |
+
|
| 333 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 334 |
+
|
| 335 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 336 |
+
|
| 337 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 338 |
+
|
| 339 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weightAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 340 |
+
|
| 341 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.biasAction head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 342 |
+
|
| 343 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 344 |
+
|
| 345 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 346 |
+
|
| 347 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 348 |
+
|
| 349 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 350 |
+
|
| 351 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 352 |
+
|
| 353 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 354 |
+
|
| 355 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 356 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 357 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 358 |
+
|
| 359 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 360 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 361 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 362 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 363 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weightAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 364 |
+
|
| 365 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.biasAction head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 366 |
+
|
| 367 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 368 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 369 |
+
|
| 370 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 371 |
+
|
| 372 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 373 |
+
|
| 374 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 375 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 376 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 377 |
+
|
| 378 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 379 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 380 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 381 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 382 |
+
|
| 383 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 384 |
+
|
| 385 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 386 |
+
|
| 387 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 388 |
+
|
| 389 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 390 |
+
|
| 391 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 392 |
+
|
| 393 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 394 |
+
|
| 395 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 396 |
+
|
| 397 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 398 |
+
|
| 399 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 400 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 401 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.biasAction head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 402 |
+
|
| 403 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weightAction head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 404 |
+
|
| 405 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 406 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 407 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weightAction head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 408 |
+
|
| 409 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.biasAction head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 410 |
+
|
| 411 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weightAction head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 412 |
+
|
| 413 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.biasAction head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 414 |
+
|
| 415 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 416 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 417 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 418 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 419 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 420 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 421 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 422 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 423 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 424 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 425 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 426 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 427 |
+
Applied trainable preset: freeze_processing_line
|
| 428 |
+
Trainable parameter tensors after preset: 585
|
| 429 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 430 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 431 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 432 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 433 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 434 |
+
Applied trainable preset: freeze_processing_line trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 435 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 436 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 437 |
+
|
| 438 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.biasTrainable parameter tensors after preset: 585
|
| 439 |
+
|
| 440 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 441 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 442 |
+
|
| 443 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 444 |
+
|
| 445 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 446 |
+
|
| 447 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 448 |
+
|
| 449 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 450 |
+
|
| 451 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 452 |
+
|
| 453 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 454 |
+
|
| 455 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 456 |
+
|
| 457 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 458 |
+
|
| 459 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 460 |
+
|
| 461 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 462 |
+
|
| 463 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 464 |
+
|
| 465 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 466 |
+
|
| 467 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 468 |
+
|
| 469 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 470 |
+
|
| 471 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 472 |
+
|
| 473 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 474 |
+
|
| 475 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 476 |
+
|
| 477 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 478 |
+
|
| 479 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 480 |
+
|
| 481 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 482 |
+
|
| 483 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 484 |
+
|
| 485 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 486 |
+
|
| 487 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 488 |
+
|
| 489 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 490 |
+
|
| 491 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 492 |
+
|
| 493 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 494 |
+
|
| 495 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 496 |
+
|
| 497 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 498 |
+
|
| 499 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 500 |
+
|
| 501 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 502 |
+
|
| 503 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 504 |
+
|
| 505 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 506 |
+
|
| 507 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 508 |
+
|
| 509 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 510 |
+
|
| 511 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 512 |
+
|
| 513 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 514 |
+
|
| 515 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 516 |
+
|
| 517 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 518 |
+
|
| 519 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 520 |
+
|
| 521 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 522 |
+
|
| 523 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 524 |
+
|
| 525 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 526 |
+
|
| 527 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 528 |
+
|
| 529 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 530 |
+
|
| 531 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 532 |
+
|
| 533 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 534 |
+
|
| 535 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 536 |
+
|
| 537 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 538 |
+
|
| 539 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 540 |
+
|
| 541 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 542 |
+
|
| 543 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 544 |
+
|
| 545 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 546 |
+
|
| 547 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 548 |
+
|
| 549 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 550 |
+
|
| 551 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 552 |
+
|
| 553 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 554 |
+
|
| 555 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 556 |
+
|
| 557 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 558 |
+
|
| 559 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 560 |
+
|
| 561 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 562 |
+
|
| 563 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 564 |
+
|
| 565 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 566 |
+
|
| 567 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 568 |
+
|
| 569 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 570 |
+
|
| 571 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 572 |
+
|
| 573 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 574 |
+
|
| 575 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 576 |
+
|
| 577 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 578 |
+
|
| 579 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 580 |
+
|
| 581 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias ... 505 more
|
| 582 |
+
|
| 583 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 584 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 585 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 586 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 587 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 588 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 589 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 590 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 591 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 592 |
+
... 505 more
|
| 593 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
|
| 594 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
|
| 595 |
+
Run name: default_phase2
|
| 596 |
+
Run name: default_phase2
|
| 597 |
+
train dataloader length: 13746
|
| 598 |
+
train dataset length: 439854
|
| 599 |
+
GPU memory before training: 7.076685905456543 GBtrain dataloader length: 13746
|
| 600 |
+
train dataset length: 439854
|
| 601 |
+
GPU memory before training: 7.076685905456543 GB
|
| 602 |
+
|
| 603 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 604 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 605 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
|
| 606 |
+
wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 607 |
+
wandb: setting up run h4cbkppl
|
| 608 |
+
wandb: Tracking run with wandb version 0.25.0
|
| 609 |
+
wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_142106-h4cbkppl
|
| 610 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 611 |
+
wandb: Syncing run default_phase2
|
| 612 |
+
wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 613 |
+
wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/h4cbkppl
|
| 614 |
+
|
| 615 |
0%| | 0/30000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
|
| 616 |
+
Could not estimate the number of tokens of the input, floating-point operations will not be computed
|
| 617 |
+
|
| 618 |
0%| | 1/30000 [00:05<43:01:49, 5.16s/it][rank1]: Traceback (most recent call last):
|
| 619 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 620 |
+
[rank1]: run_yaml_experiment(
|
| 621 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 622 |
+
[rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 623 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 624 |
+
[rank1]: experiment.train()
|
| 625 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 626 |
+
[rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 627 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 628 |
+
[rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 629 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 630 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 631 |
+
[rank1]: return inner_training_loop(
|
| 632 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 633 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 634 |
+
[rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 635 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 636 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 637 |
+
[rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 638 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 639 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 640 |
+
[rank1]: outputs = model(inputs)
|
| 641 |
+
[rank1]: ^^^^^^^^^^^^^
|
| 642 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 643 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 644 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 645 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 646 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 647 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 648 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1633, in forward
|
| 649 |
+
[rank1]: inputs, kwargs = self._pre_forward(*inputs, **kwargs)
|
| 650 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 651 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1522, in _pre_forward
|
| 652 |
+
[rank1]: if torch.is_grad_enabled() and self.reducer._rebuild_buckets():
|
| 653 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 654 |
+
[rank1]: RuntimeError: Expected to have finished reduction in the prior iteration before starting a new one. This error indicates that your module has parameters that were not used in producing loss. You can enable unused parameter detection by passing the keyword argument `find_unused_parameters=True` to `torch.nn.parallel.DistributedDataParallel`, and by
|
| 655 |
+
[rank1]: making sure all `forward` function outputs participate in calculating loss.
|
| 656 |
+
[rank1]: If you already have done the above, then the distributed data parallel module wasn't able to locate the output tensors in the return value of your module's `forward` function. Please include the loss function and the structure of the return value of `forward` of your module when reporting this issue (e.g. list, dict, iterable).
|
| 657 |
+
[rank1]: Parameter indices which did not receive grad for rank 1: 582
|
| 658 |
+
[rank1]: In addition, you can set the environment variable TORCH_DISTRIBUTED_DEBUG to either INFO or DETAIL to print out information about which particular parameters did not receive gradient on this rank as part of this error
|
| 659 |
+
wandb: uploading wandb-metadata.json; uploading requirements.txt; updating run metadata
|
| 660 |
+
wandb: uploading wandb-metadata.json; uploading requirements.txt; uploading wandb-summary.json; uploading config.yaml; uploading output.log
|
| 661 |
+
wandb: uploading summary, console lines 0-1
|
| 662 |
+
wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/h4cbkppl
|
| 663 |
+
wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 664 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 665 |
+
wandb: Find logs at: ./wandb/run-20260615_142106-h4cbkppl/logs
|
| 666 |
+
Traceback (most recent call last):
|
| 667 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 668 |
+
run_yaml_experiment(
|
| 669 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 670 |
+
main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 671 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 672 |
+
experiment.train()
|
| 673 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 674 |
+
self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 675 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 676 |
+
return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 677 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 678 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 679 |
+
return inner_training_loop(
|
| 680 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 681 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 682 |
+
tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 683 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 684 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 685 |
+
loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 686 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 687 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 688 |
+
outputs = model(inputs)
|
| 689 |
+
^^^^^^^^^^^^^
|
| 690 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 691 |
+
return self._call_impl(*args, **kwargs)
|
| 692 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 693 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 694 |
+
return forward_call(*args, **kwargs)
|
| 695 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 696 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1633, in forward
|
| 697 |
+
inputs, kwargs = self._pre_forward(*inputs, **kwargs)
|
| 698 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 699 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1522, in _pre_forward
|
| 700 |
+
if torch.is_grad_enabled() and self.reducer._rebuild_buckets():
|
| 701 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 702 |
+
RuntimeError: Expected to have finished reduction in the prior iteration before starting a new one. This error indicates that your module has parameters that were not used in producing loss. You can enable unused parameter detection by passing the keyword argument `find_unused_parameters=True` to `torch.nn.parallel.DistributedDataParallel`, and by
|
| 703 |
+
making sure all `forward` function outputs participate in calculating loss.
|
| 704 |
+
If you already have done the above, then the distributed data parallel module wasn't able to locate the output tensors in the return value of your module's `forward` function. Please include the loss function and the structure of the return value of `forward` of your module when reporting this issue (e.g. list, dict, iterable).
|
| 705 |
+
Parameter indices which did not receive grad for rank 0: 582
|
| 706 |
+
In addition, you can set the environment variable TORCH_DISTRIBUTED_DEBUG to either INFO or DETAIL to print out information about which particular parameters did not receive gradient on this rank as part of this error
|
| 707 |
+
[rank0]: Traceback (most recent call last):
|
| 708 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 709 |
+
[rank0]: run_yaml_experiment(
|
| 710 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 711 |
+
[rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 712 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 713 |
+
[rank0]: experiment.train()
|
| 714 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 715 |
+
[rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 716 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 717 |
+
[rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 718 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 719 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 720 |
+
[rank0]: return inner_training_loop(
|
| 721 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 722 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 723 |
+
[rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 724 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 725 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 726 |
+
[rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 727 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 728 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 729 |
+
[rank0]: outputs = model(inputs)
|
| 730 |
+
[rank0]: ^^^^^^^^^^^^^
|
| 731 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 732 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 733 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 734 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 735 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 736 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 737 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1633, in forward
|
| 738 |
+
[rank0]: inputs, kwargs = self._pre_forward(*inputs, **kwargs)
|
| 739 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 740 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1522, in _pre_forward
|
| 741 |
+
[rank0]: if torch.is_grad_enabled() and self.reducer._rebuild_buckets():
|
| 742 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 743 |
+
[rank0]: RuntimeError: Expected to have finished reduction in the prior iteration before starting a new one. This error indicates that your module has parameters that were not used in producing loss. You can enable unused parameter detection by passing the keyword argument `find_unused_parameters=True` to `torch.nn.parallel.DistributedDataParallel`, and by
|
| 744 |
+
[rank0]: making sure all `forward` function outputs participate in calculating loss.
|
| 745 |
+
[rank0]: If you already have done the above, then the distributed data parallel module wasn't able to locate the output tensors in the return value of your module's `forward` function. Please include the loss function and the structure of the return value of `forward` of your module when reporting this issue (e.g. list, dict, iterable).
|
| 746 |
+
[rank0]: Parameter indices which did not receive grad for rank 0: 582
|
| 747 |
+
[rank0]: In addition, you can set the environment variable TORCH_DISTRIBUTED_DEBUG to either INFO or DETAIL to print out information about which particular parameters did not receive gradient on this rank as part of this error
|
| 748 |
+
[rank0]:[W615 14:21:15.946760892 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
|
| 749 |
+
W0615 14:21:15.519000 1313589 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 1319773 closing signal SIGTERM
|
| 750 |
+
E0615 14:21:15.883000 1313589 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 1319776) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
|
| 751 |
+
Traceback (most recent call last):
|
| 752 |
+
File "<frozen runpy>", line 198, in _run_module_as_main
|
| 753 |
+
File "<frozen runpy>", line 88, in _run_code
|
| 754 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
|
| 755 |
+
main()
|
| 756 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
|
| 757 |
+
return f(*args, **kwargs)
|
| 758 |
+
^^^^^^^^^^^^^^^^^^
|
| 759 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
|
| 760 |
+
run(args)
|
| 761 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
|
| 762 |
+
elastic_launch(
|
| 763 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
|
| 764 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 765 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 766 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
|
| 767 |
+
raise ChildFailedError(
|
| 768 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 769 |
+
============================================================
|
| 770 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
|
| 771 |
+
------------------------------------------------------------
|
| 772 |
+
Failures:
|
| 773 |
+
<NO_OTHER_FAILURES>
|
| 774 |
+
------------------------------------------------------------
|
| 775 |
+
Root Cause (first observed failure):
|
| 776 |
+
[0]:
|
| 777 |
+
time : 2026-06-15_14:21:15
|
| 778 |
+
host : worker1
|
| 779 |
+
rank : 1 (local_rank: 1)
|
| 780 |
+
exitcode : 1 (pid: 1319776)
|
| 781 |
+
error_file: <N/A>
|
| 782 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 783 |
+
============================================================
|
| 784 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 785 |
+
|
| 786 |
+
==================================================
|
| 787 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 788 |
+
==================================================
|
| 789 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 790 |
+
dataset_soup: None
|
| 791 |
+
output_dir: /tmp/gr00t
|
| 792 |
+
output_root: None
|
| 793 |
+
data_config: panda_omron
|
| 794 |
+
batch_size: 16
|
| 795 |
+
max_steps: 300000
|
| 796 |
+
num_gpus: 2
|
| 797 |
+
save_steps: 20000
|
| 798 |
+
run_name: None
|
| 799 |
+
save_total_limit: 100
|
| 800 |
+
seed: 42
|
| 801 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 802 |
+
tune_llm: False
|
| 803 |
+
tune_visual: False
|
| 804 |
+
tune_projector: True
|
| 805 |
+
tune_diffusion_model: True
|
| 806 |
+
resume: False
|
| 807 |
+
learning_rate: 3e-05
|
| 808 |
+
weight_decay: 1e-05
|
| 809 |
+
warmup_ratio: 0.05
|
| 810 |
+
lora_rank: 0
|
| 811 |
+
lora_alpha: 16
|
| 812 |
+
lora_dropout: 0.1
|
| 813 |
+
lora_full_model: False
|
| 814 |
+
dataloader_num_workers: 8
|
| 815 |
+
report_to: wandb
|
| 816 |
+
embodiment_tag: new_embodiment
|
| 817 |
+
video_backend: opencv
|
| 818 |
+
balance_dataset_weights: True
|
| 819 |
+
balance_trajectory_weights: True
|
| 820 |
+
ds_weights_alpha: 0.4
|
| 821 |
+
==================================================
|
| 822 |
+
|
| 823 |
+
Using 2 GPUs
|
| 824 |
+
Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '16', '--num-gpus', '2']
|
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_3_60k.log
ADDED
|
@@ -0,0 +1,1088 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/60000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
|
|
|
|
|
|
|
| 1 |
0%| | 1/60000 [00:08<139:31:18, 8.37s/it][rank1]: Traceback (most recent call last):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 2 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 3 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 4 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 5 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 6 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 7 |
+
check_for_updates()
|
| 8 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 9 |
+
|
| 10 |
+
*****************************************
|
| 11 |
+
Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
|
| 12 |
+
*****************************************
|
| 13 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 14 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 15 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 16 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 17 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 18 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 19 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 20 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 21 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 22 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 23 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 24 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 25 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 26 |
+
check_for_updates()
|
| 27 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 28 |
+
check_for_updates()
|
| 29 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 30 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 31 |
+
|
| 32 |
+
==================================================
|
| 33 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 34 |
+
==================================================
|
| 35 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 36 |
+
dataset_soup: None
|
| 37 |
+
output_dir: /tmp/gr00t
|
| 38 |
+
output_root: None
|
| 39 |
+
data_config: panda_omron
|
| 40 |
+
batch_size: 32
|
| 41 |
+
max_steps: 300000
|
| 42 |
+
num_gpus: 2
|
| 43 |
+
save_steps: 20000
|
| 44 |
+
run_name: None
|
| 45 |
+
save_total_limit: 100
|
| 46 |
+
seed: 42
|
| 47 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 48 |
+
tune_llm: False
|
| 49 |
+
tune_visual: False
|
| 50 |
+
tune_projector: True
|
| 51 |
+
tune_diffusion_model: True
|
| 52 |
+
resume: False
|
| 53 |
+
learning_rate: 3e-05
|
| 54 |
+
weight_decay: 1e-05
|
| 55 |
+
warmup_ratio: 0.05
|
| 56 |
+
lora_rank: 0
|
| 57 |
+
lora_alpha: 16
|
| 58 |
+
lora_dropout: 0.1
|
| 59 |
+
lora_full_model: False
|
| 60 |
+
dataloader_num_workers: 8
|
| 61 |
+
report_to: wandb
|
| 62 |
+
embodiment_tag: new_embodiment
|
| 63 |
+
video_backend: opencv
|
| 64 |
+
balance_dataset_weights: True
|
| 65 |
+
balance_trajectory_weights: True
|
| 66 |
+
ds_weights_alpha: 0.4
|
| 67 |
+
==================================================
|
| 68 |
+
|
| 69 |
+
Using 2 GPUs
|
| 70 |
+
|
| 71 |
+
================================================================================
|
| 72 |
+
Starting sweep branch: default
|
| 73 |
+
Sweep vars: {}
|
| 74 |
+
================================================================================
|
| 75 |
+
|
| 76 |
+
--------------------------------------------------------------------------------
|
| 77 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 78 |
+
Policy type: groot_rkd_v2_raw
|
| 79 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 80 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 81 |
+
Trainable preset: freeze_processing_line
|
| 82 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 83 |
+
--------------------------------------------------------------------------------
|
| 84 |
+
|
| 85 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 86 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 87 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 88 |
+
self.statistics[key] = torch.tensor(value)
|
| 89 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 90 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 91 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 92 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 93 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 94 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 95 |
+
|
| 96 |
+
==================================================
|
| 97 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 98 |
+
==================================================
|
| 99 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 100 |
+
dataset_soup: None
|
| 101 |
+
output_dir: /tmp/gr00t
|
| 102 |
+
output_root: None
|
| 103 |
+
data_config: panda_omron
|
| 104 |
+
batch_size: 32
|
| 105 |
+
max_steps: 300000
|
| 106 |
+
num_gpus: 2
|
| 107 |
+
save_steps: 20000
|
| 108 |
+
run_name: None
|
| 109 |
+
save_total_limit: 100
|
| 110 |
+
seed: 42
|
| 111 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 112 |
+
tune_llm: False
|
| 113 |
+
tune_visual: False
|
| 114 |
+
tune_projector: True
|
| 115 |
+
tune_diffusion_model: True
|
| 116 |
+
resume: False
|
| 117 |
+
learning_rate: 3e-05
|
| 118 |
+
weight_decay: 1e-05
|
| 119 |
+
warmup_ratio: 0.05
|
| 120 |
+
lora_rank: 0
|
| 121 |
+
lora_alpha: 16
|
| 122 |
+
lora_dropout: 0.1
|
| 123 |
+
lora_full_model: False
|
| 124 |
+
dataloader_num_workers: 8
|
| 125 |
+
report_to: wandb
|
| 126 |
+
embodiment_tag: new_embodiment
|
| 127 |
+
video_backend: opencv
|
| 128 |
+
balance_dataset_weights: True
|
| 129 |
+
balance_trajectory_weights: True
|
| 130 |
+
ds_weights_alpha: 0.4
|
| 131 |
+
==================================================
|
| 132 |
+
|
| 133 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 134 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 135 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 136 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 137 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 138 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 139 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 140 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 141 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 142 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 143 |
+
Using 2 GPUs
|
| 144 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 145 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 146 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 147 |
+
|
| 148 |
+
================================================================================
|
| 149 |
+
Starting sweep branch: default
|
| 150 |
+
Sweep vars: {}
|
| 151 |
+
================================================================================
|
| 152 |
+
|
| 153 |
+
--------------------------------------------------------------------------------
|
| 154 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 155 |
+
Policy type: groot_rkd_v2_raw
|
| 156 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 157 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 158 |
+
Trainable preset: freeze_processing_line
|
| 159 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 160 |
+
--------------------------------------------------------------------------------
|
| 161 |
+
|
| 162 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 163 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 164 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 165 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 166 |
+
self.statistics[key] = torch.tensor(value)
|
| 167 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 168 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 169 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 170 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 171 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 172 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 173 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 174 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 175 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 176 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 177 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 178 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 179 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 180 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 181 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 182 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 183 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 184 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 185 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 186 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 187 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 188 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 189 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 190 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 191 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 192 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 193 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 194 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 195 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 196 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 197 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 198 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 199 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 200 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 201 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 202 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 203 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 204 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 205 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 206 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 207 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 208 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 209 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 210 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 211 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 212 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 213 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 214 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 215 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 216 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 217 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 218 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 219 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 220 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 221 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 222 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 223 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 224 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 225 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 226 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 227 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 228 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 229 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 230 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 231 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 232 |
+
0.75517122 0.7973985 ]
|
| 233 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 234 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 235 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 236 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 237 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 238 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 239 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 240 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 241 |
+
Loaded 26 datasets
|
| 242 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 243 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 244 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 245 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 246 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 247 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 248 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 249 |
+
Tune backbone vision tower: True
|
| 250 |
+
Tune backbone LLM: True
|
| 251 |
+
Tune action head projector: False
|
| 252 |
+
Tune action head DiT: False
|
| 253 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 254 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 255 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 256 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 257 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 258 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 259 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 260 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 261 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 262 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 263 |
+
0.75517122 0.7973985 ]
|
| 264 |
+
Loaded 26 datasets
|
| 265 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 266 |
+
Tune backbone vision tower: True
|
| 267 |
+
Tune backbone LLM: True
|
| 268 |
+
Tune action head projector: False
|
| 269 |
+
Tune action head DiT: False
|
| 270 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 271 |
+
Tune backbone llm: False
|
| 272 |
+
Tune backbone visual: True
|
| 273 |
+
Total number of DiT parameters: 550386688
|
| 274 |
+
Tune backbone llm: False
|
| 275 |
+
Tune backbone visual: True
|
| 276 |
+
Total number of DiT parameters: 550386688
|
| 277 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 278 |
+
Tune action head projector: True
|
| 279 |
+
Tune action head diffusion model: True
|
| 280 |
+
|
| 281 |
+
Tune action head projector: True
|
| 282 |
+
Tune action head diffusion model: True
|
| 283 |
+
|
| 284 |
+
Tune backbone llm: True
|
| 285 |
+
Tune backbone visual: True
|
| 286 |
+
Tune action head projector: False
|
| 287 |
+
Tune action head diffusion model: False
|
| 288 |
+
Action head trainable parameter: future_tokens.weight
|
| 289 |
+
Action head trainable parameter: vlln.weight
|
| 290 |
+
Action head trainable parameter: vlln.bias
|
| 291 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 292 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 293 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 294 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 295 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 296 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 297 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 298 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 299 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 300 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 301 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 302 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 303 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 304 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 305 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 306 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 307 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 308 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 309 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 310 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 311 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 312 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 313 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 314 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 315 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 316 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 317 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 318 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 319 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 320 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 321 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 322 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 323 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 324 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 325 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 326 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 327 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 328 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 329 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 330 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 331 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 332 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 333 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 334 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 335 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 336 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 337 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 338 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 339 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 340 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 341 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 342 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 343 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 344 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 345 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 346 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 347 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 348 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 349 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 350 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 351 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 352 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 353 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 354 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 355 |
+
Applied trainable preset: freeze_processing_line
|
| 356 |
+
Trainable parameter tensors after preset: 584
|
| 357 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 358 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 359 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 360 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 361 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 362 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 363 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 364 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 365 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 366 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 367 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 368 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 369 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 370 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 371 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 372 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 373 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 374 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 375 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 376 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 377 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 378 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 379 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 380 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 381 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 382 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 383 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 384 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 385 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 386 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 387 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 388 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 389 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 390 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 391 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 392 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 393 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 394 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 395 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 396 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 397 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 398 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 399 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 400 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 401 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 402 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 403 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 404 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 405 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 406 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 407 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 408 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 409 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 410 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 411 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 412 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 413 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 414 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 415 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 416 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 417 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 418 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 419 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 420 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 421 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 422 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 423 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 424 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 425 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 426 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 427 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 428 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 429 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 430 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 431 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 432 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 433 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 434 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 435 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 436 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 437 |
+
... 504 more
|
| 438 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
|
| 439 |
+
|
| 440 |
+
Tune backbone llm: True
|
| 441 |
+
Tune backbone visual: True
|
| 442 |
+
Tune action head projector: False
|
| 443 |
+
Tune action head diffusion model: False
|
| 444 |
+
Action head trainable parameter: future_tokens.weight
|
| 445 |
+
Action head trainable parameter: vlln.weight
|
| 446 |
+
Action head trainable parameter: vlln.bias
|
| 447 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 448 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 449 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 450 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 451 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 452 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 453 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 454 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 455 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 456 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 457 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 458 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 459 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 460 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 461 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 462 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 463 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 464 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 465 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 466 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 467 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 468 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 469 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 470 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 471 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 472 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 473 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 474 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 475 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 476 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 477 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 478 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 479 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 480 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 481 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 482 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 483 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 484 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 485 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 486 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 487 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 488 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 489 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 490 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 491 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 492 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 493 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 494 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 495 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 496 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 497 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 498 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 499 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 500 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 501 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 502 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 503 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 504 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 505 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 506 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 507 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 508 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 509 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 510 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 511 |
+
Applied trainable preset: freeze_processing_line
|
| 512 |
+
Trainable parameter tensors after preset: 584
|
| 513 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 514 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 515 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 516 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 517 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 518 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 519 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 520 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 521 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 522 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 523 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 524 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 525 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 526 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 527 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 528 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 529 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 530 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 531 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 532 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 533 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 534 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 535 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 536 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 537 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 538 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 539 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 540 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 541 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 542 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 543 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 544 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 545 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 546 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 547 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 548 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 549 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 550 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 551 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 552 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 553 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 554 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 555 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 556 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 557 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 558 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 559 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 560 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 561 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 562 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 563 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 564 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 565 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 566 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 567 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 568 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 569 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 570 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 571 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 572 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 573 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 574 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 575 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 576 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 577 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 578 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 579 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 580 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 581 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 582 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 583 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 584 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 585 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 586 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 587 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 588 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 589 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 590 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 591 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 592 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 593 |
+
... 504 more
|
| 594 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
|
| 595 |
+
Run name: default_phase2
|
| 596 |
+
Run name: default_phase2
|
| 597 |
+
train dataloader length: 6873
|
| 598 |
+
train dataset length: 439854
|
| 599 |
+
GPU memory before training: 7.076685905456543 GB
|
| 600 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 601 |
+
train dataloader length: 6873
|
| 602 |
+
train dataset length: 439854
|
| 603 |
+
GPU memory before training: 7.076685905456543 GB
|
| 604 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 605 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
|
| 606 |
+
wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 607 |
+
wandb: setting up run tu7a0ujw
|
| 608 |
+
wandb: Tracking run with wandb version 0.25.0
|
| 609 |
+
wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_150511-tu7a0ujw
|
| 610 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 611 |
+
wandb: Syncing run default_phase2
|
| 612 |
+
wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 613 |
+
wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/tu7a0ujw
|
| 614 |
+
|
| 615 |
0%| | 0/60000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
|
| 616 |
+
Could not estimate the number of tokens of the input, floating-point operations will not be computed
|
| 617 |
+
|
| 618 |
0%| | 1/60000 [00:08<139:31:18, 8.37s/it][rank1]: Traceback (most recent call last):
|
| 619 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
|
| 620 |
+
[rank1]: run_yaml_experiment(
|
| 621 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
|
| 622 |
+
[rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 623 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
|
| 624 |
+
[rank1]: experiment.train()
|
| 625 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 626 |
+
[rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 627 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 628 |
+
[rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 629 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 630 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 631 |
+
[rank1]: return inner_training_loop(
|
| 632 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 633 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 634 |
+
[rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 635 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 636 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 637 |
+
[rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 638 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 639 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 640 |
+
[rank1]: outputs = model(inputs)
|
| 641 |
+
[rank1]: ^^^^^^^^^^^^^
|
| 642 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 643 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 644 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 645 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 646 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 647 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 648 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 649 |
+
[rank1]: else self._run_ddp_forward(*inputs, **kwargs)
|
| 650 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 651 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 652 |
+
[rank1]: return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 653 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 654 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 655 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 656 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 657 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 658 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 659 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 660 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 661 |
+
[rank1]: return model_forward(*args, **kwargs)
|
| 662 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 663 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 664 |
+
[rank1]: return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 665 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 666 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 667 |
+
[rank1]: return func(*args, **kwargs)
|
| 668 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^
|
| 669 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 670 |
+
[rank1]: backbone_outputs = self.backbone(backbone_inputs)
|
| 671 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 672 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 673 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 674 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 675 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 676 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 677 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 678 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 123, in forward
|
| 679 |
+
[rank1]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 680 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 681 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 682 |
+
[rank1]: eagle_output = self.eagle_model(
|
| 683 |
+
[rank1]: ^^^^^^^^^^^^^^^^^
|
| 684 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 685 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 686 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 687 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 688 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 689 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 690 |
+
[rank1]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 262, in forward
|
| 691 |
+
[rank1]: outputs = self.language_model(
|
| 692 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 693 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 694 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 695 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 696 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 697 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 698 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 699 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 700 |
+
[rank1]: output = func(self, *args, **kwargs)
|
| 701 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 702 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 703 |
+
[rank1]: return func(*args, **kwargs)
|
| 704 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^
|
| 705 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 706 |
+
[rank1]: outputs: BaseModelOutputWithPast = self.model(
|
| 707 |
+
[rank1]: ^^^^^^^^^^^
|
| 708 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 709 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 710 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 711 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 712 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 713 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 714 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 715 |
+
[rank1]: output = func(self, *args, **kwargs)
|
| 716 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 717 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 718 |
+
[rank1]: layer_outputs = decoder_layer(
|
| 719 |
+
[rank1]: ^^^^^^^^^^^^^^
|
| 720 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 721 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 722 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 723 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 724 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 725 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 726 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 289, in forward
|
| 727 |
+
[rank1]: hidden_states, self_attn_weights = self.self_attn(
|
| 728 |
+
[rank1]: ^^^^^^^^^^^^^^^
|
| 729 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 730 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 731 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 732 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 733 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 734 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 735 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 217, in forward
|
| 736 |
+
[rank1]: query_states = self.q_norm(self.q_proj(hidden_states).view(hidden_shape)).transpose(1, 2)
|
| 737 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 738 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 739 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 740 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 741 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 742 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 743 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 744 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 74, in forward
|
| 745 |
+
[rank1]: variance = hidden_states.pow(2).mean(-1, keepdim=True)
|
| 746 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 747 |
+
[rank1]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 210.00 MiB. GPU 1 has a total capacity of 79.25 GiB of which 129.94 MiB is free. Including non-PyTorch memory, this process has 79.10 GiB memory in use. Of the allocated memory 73.70 GiB is allocated by PyTorch, and 4.76 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 748 |
+
wandb: updating run metadata
|
| 749 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 750 |
+
wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/tu7a0ujw
|
| 751 |
+
wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 752 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 753 |
+
wandb: Find logs at: ./wandb/run-20260615_150511-tu7a0ujw/logs
|
| 754 |
+
Traceback (most recent call last):
|
| 755 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
|
| 756 |
+
run_yaml_experiment(
|
| 757 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
|
| 758 |
+
main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 759 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
|
| 760 |
+
experiment.train()
|
| 761 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 762 |
+
self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 763 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 764 |
+
return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 765 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 766 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 767 |
+
return inner_training_loop(
|
| 768 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 769 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 770 |
+
tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 771 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 772 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 773 |
+
loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 774 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 775 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 776 |
+
outputs = model(inputs)
|
| 777 |
+
^^^^^^^^^^^^^
|
| 778 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 779 |
+
return self._call_impl(*args, **kwargs)
|
| 780 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 781 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 782 |
+
return forward_call(*args, **kwargs)
|
| 783 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 784 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 785 |
+
else self._run_ddp_forward(*inputs, **kwargs)
|
| 786 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 787 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 788 |
+
return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 789 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 790 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 791 |
+
return self._call_impl(*args, **kwargs)
|
| 792 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 793 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 794 |
+
return forward_call(*args, **kwargs)
|
| 795 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 796 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 797 |
+
return model_forward(*args, **kwargs)
|
| 798 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 799 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 800 |
+
return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 801 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 802 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 803 |
+
return func(*args, **kwargs)
|
| 804 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 805 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 806 |
+
backbone_outputs = self.backbone(backbone_inputs)
|
| 807 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 808 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 809 |
+
return self._call_impl(*args, **kwargs)
|
| 810 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 811 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 812 |
+
return forward_call(*args, **kwargs)
|
| 813 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 814 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 123, in forward
|
| 815 |
+
eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 816 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 817 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 818 |
+
eagle_output = self.eagle_model(
|
| 819 |
+
^^^^^^^^^^^^^^^^^
|
| 820 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 821 |
+
return self._call_impl(*args, **kwargs)
|
| 822 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 823 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 824 |
+
return forward_call(*args, **kwargs)
|
| 825 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 826 |
+
File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 262, in forward
|
| 827 |
+
outputs = self.language_model(
|
| 828 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 829 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 830 |
+
return self._call_impl(*args, **kwargs)
|
| 831 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 832 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 833 |
+
return forward_call(*args, **kwargs)
|
| 834 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 835 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 836 |
+
output = func(self, *args, **kwargs)
|
| 837 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 838 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 839 |
+
return func(*args, **kwargs)
|
| 840 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 841 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 842 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 843 |
+
^^^^^^^^^^^
|
| 844 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 845 |
+
return self._call_impl(*args, **kwargs)
|
| 846 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 847 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 848 |
+
return forward_call(*args, **kwargs)
|
| 849 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 850 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 851 |
+
output = func(self, *args, **kwargs)
|
| 852 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 853 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 854 |
+
layer_outputs = decoder_layer(
|
| 855 |
+
^^^^^^^^^^^^^^
|
| 856 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 857 |
+
return self._call_impl(*args, **kwargs)
|
| 858 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 859 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 860 |
+
return forward_call(*args, **kwargs)
|
| 861 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 862 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 289, in forward
|
| 863 |
+
hidden_states, self_attn_weights = self.self_attn(
|
| 864 |
+
^^^^^^^^^^^^^^^
|
| 865 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 866 |
+
return self._call_impl(*args, **kwargs)
|
| 867 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 868 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 869 |
+
return forward_call(*args, **kwargs)
|
| 870 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 871 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 217, in forward
|
| 872 |
+
query_states = self.q_norm(self.q_proj(hidden_states).view(hidden_shape)).transpose(1, 2)
|
| 873 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 874 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 875 |
+
return self._call_impl(*args, **kwargs)
|
| 876 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 877 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 878 |
+
return forward_call(*args, **kwargs)
|
| 879 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 880 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 74, in forward
|
| 881 |
+
variance = hidden_states.pow(2).mean(-1, keepdim=True)
|
| 882 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 883 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 210.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 129.94 MiB is free. Including non-PyTorch memory, this process has 79.10 GiB memory in use. Of the allocated memory 73.70 GiB is allocated by PyTorch, and 4.76 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 884 |
+
[rank0]: Traceback (most recent call last):
|
| 885 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
|
| 886 |
+
[rank0]: run_yaml_experiment(
|
| 887 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
|
| 888 |
+
[rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 889 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
|
| 890 |
+
[rank0]: experiment.train()
|
| 891 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 892 |
+
[rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 893 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 894 |
+
[rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 895 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 896 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 897 |
+
[rank0]: return inner_training_loop(
|
| 898 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 899 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 900 |
+
[rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 901 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 902 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 903 |
+
[rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 904 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 905 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 906 |
+
[rank0]: outputs = model(inputs)
|
| 907 |
+
[rank0]: ^^^^^^^^^^^^^
|
| 908 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 909 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 910 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 911 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 912 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 913 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 914 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 915 |
+
[rank0]: else self._run_ddp_forward(*inputs, **kwargs)
|
| 916 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 917 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 918 |
+
[rank0]: return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 919 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 920 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 921 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 922 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 923 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 924 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 925 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 926 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 927 |
+
[rank0]: return model_forward(*args, **kwargs)
|
| 928 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 929 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 930 |
+
[rank0]: return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 931 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 932 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 933 |
+
[rank0]: return func(*args, **kwargs)
|
| 934 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^
|
| 935 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 936 |
+
[rank0]: backbone_outputs = self.backbone(backbone_inputs)
|
| 937 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 938 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 939 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 940 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 941 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 942 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 943 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 944 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 123, in forward
|
| 945 |
+
[rank0]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 946 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 947 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 948 |
+
[rank0]: eagle_output = self.eagle_model(
|
| 949 |
+
[rank0]: ^^^^^^^^^^^^^^^^^
|
| 950 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 951 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 952 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 953 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 954 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 955 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 956 |
+
[rank0]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 262, in forward
|
| 957 |
+
[rank0]: outputs = self.language_model(
|
| 958 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 959 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 960 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 961 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 962 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 963 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 964 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 965 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 966 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 967 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 968 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 969 |
+
[rank0]: return func(*args, **kwargs)
|
| 970 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^
|
| 971 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 972 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 973 |
+
[rank0]: ^^^^^^^^^^^
|
| 974 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 975 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 976 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 977 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 978 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 979 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 980 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 981 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 982 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 983 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 984 |
+
[rank0]: layer_outputs = decoder_layer(
|
| 985 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 986 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 987 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 988 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 989 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 990 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 991 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 992 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 289, in forward
|
| 993 |
+
[rank0]: hidden_states, self_attn_weights = self.self_attn(
|
| 994 |
+
[rank0]: ^^^^^^^^^^^^^^^
|
| 995 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 996 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 997 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 998 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 999 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 1000 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1001 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 217, in forward
|
| 1002 |
+
[rank0]: query_states = self.q_norm(self.q_proj(hidden_states).view(hidden_shape)).transpose(1, 2)
|
| 1003 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1004 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 1005 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 1006 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1007 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 1008 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 1009 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1010 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 74, in forward
|
| 1011 |
+
[rank0]: variance = hidden_states.pow(2).mean(-1, keepdim=True)
|
| 1012 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 1013 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 210.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 129.94 MiB is free. Including non-PyTorch memory, this process has 79.10 GiB memory in use. Of the allocated memory 73.70 GiB is allocated by PyTorch, and 4.76 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 1014 |
+
[rank0]:[W615 15:05:24.539442388 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
|
| 1015 |
+
W0615 15:05:24.803000 2931837 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 2938608 closing signal SIGTERM
|
| 1016 |
+
E0615 15:05:25.318000 2931837 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 2938609) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
|
| 1017 |
+
Traceback (most recent call last):
|
| 1018 |
+
File "<frozen runpy>", line 198, in _run_module_as_main
|
| 1019 |
+
File "<frozen runpy>", line 88, in _run_code
|
| 1020 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
|
| 1021 |
+
main()
|
| 1022 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
|
| 1023 |
+
return f(*args, **kwargs)
|
| 1024 |
+
^^^^^^^^^^^^^^^^^^
|
| 1025 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
|
| 1026 |
+
run(args)
|
| 1027 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
|
| 1028 |
+
elastic_launch(
|
| 1029 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
|
| 1030 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 1031 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1032 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
|
| 1033 |
+
raise ChildFailedError(
|
| 1034 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 1035 |
+
============================================================
|
| 1036 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
|
| 1037 |
+
------------------------------------------------------------
|
| 1038 |
+
Failures:
|
| 1039 |
+
<NO_OTHER_FAILURES>
|
| 1040 |
+
------------------------------------------------------------
|
| 1041 |
+
Root Cause (first observed failure):
|
| 1042 |
+
[0]:
|
| 1043 |
+
time : 2026-06-15_15:05:24
|
| 1044 |
+
host : worker1
|
| 1045 |
+
rank : 1 (local_rank: 1)
|
| 1046 |
+
exitcode : 1 (pid: 2938609)
|
| 1047 |
+
error_file: <N/A>
|
| 1048 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 1049 |
+
============================================================
|
| 1050 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 1051 |
+
|
| 1052 |
+
==================================================
|
| 1053 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 1054 |
+
==================================================
|
| 1055 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 1056 |
+
dataset_soup: None
|
| 1057 |
+
output_dir: /tmp/gr00t
|
| 1058 |
+
output_root: None
|
| 1059 |
+
data_config: panda_omron
|
| 1060 |
+
batch_size: 32
|
| 1061 |
+
max_steps: 300000
|
| 1062 |
+
num_gpus: 2
|
| 1063 |
+
save_steps: 20000
|
| 1064 |
+
run_name: None
|
| 1065 |
+
save_total_limit: 100
|
| 1066 |
+
seed: 42
|
| 1067 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 1068 |
+
tune_llm: False
|
| 1069 |
+
tune_visual: False
|
| 1070 |
+
tune_projector: True
|
| 1071 |
+
tune_diffusion_model: True
|
| 1072 |
+
resume: False
|
| 1073 |
+
learning_rate: 3e-05
|
| 1074 |
+
weight_decay: 1e-05
|
| 1075 |
+
warmup_ratio: 0.05
|
| 1076 |
+
lora_rank: 0
|
| 1077 |
+
lora_alpha: 16
|
| 1078 |
+
lora_dropout: 0.1
|
| 1079 |
+
lora_full_model: False
|
| 1080 |
+
dataloader_num_workers: 8
|
| 1081 |
+
report_to: wandb
|
| 1082 |
+
embodiment_tag: new_embodiment
|
| 1083 |
+
video_backend: opencv
|
| 1084 |
+
balance_dataset_weights: True
|
| 1085 |
+
balance_trajectory_weights: True
|
| 1086 |
+
ds_weights_alpha: 0.4
|
| 1087 |
+
==================================================
|
| 1088 |
+
|
| 1089 |
+
Using 2 GPUs
|
| 1090 |
+
Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '32', '--num-gpus', '2']
|
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_3_probe.log
ADDED
|
@@ -0,0 +1,997 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 2 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 3 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 4 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 5 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 6 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 7 |
+
check_for_updates()
|
| 8 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 9 |
+
|
| 10 |
+
*****************************************
|
| 11 |
+
Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
|
| 12 |
+
*****************************************
|
| 13 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 14 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 15 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 16 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 17 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 18 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 19 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 20 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 21 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 22 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 23 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 24 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 25 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 26 |
+
check_for_updates()
|
| 27 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 28 |
+
check_for_updates()
|
| 29 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 30 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 31 |
+
|
| 32 |
+
==================================================
|
| 33 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 34 |
+
==================================================
|
| 35 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 36 |
+
dataset_soup: None
|
| 37 |
+
output_dir: /tmp/gr00t
|
| 38 |
+
output_root: None
|
| 39 |
+
data_config: panda_omron
|
| 40 |
+
batch_size: 32
|
| 41 |
+
max_steps: 300000
|
| 42 |
+
num_gpus: 2
|
| 43 |
+
save_steps: 20000
|
| 44 |
+
run_name: None
|
| 45 |
+
save_total_limit: 100
|
| 46 |
+
seed: 42
|
| 47 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 48 |
+
tune_llm: False
|
| 49 |
+
tune_visual: False
|
| 50 |
+
tune_projector: True
|
| 51 |
+
tune_diffusion_model: True
|
| 52 |
+
resume: False
|
| 53 |
+
learning_rate: 3e-05
|
| 54 |
+
weight_decay: 1e-05
|
| 55 |
+
warmup_ratio: 0.05
|
| 56 |
+
lora_rank: 0
|
| 57 |
+
lora_alpha: 16
|
| 58 |
+
lora_dropout: 0.1
|
| 59 |
+
lora_full_model: False
|
| 60 |
+
dataloader_num_workers: 8
|
| 61 |
+
report_to: wandb
|
| 62 |
+
embodiment_tag: new_embodiment
|
| 63 |
+
video_backend: opencv
|
| 64 |
+
balance_dataset_weights: True
|
| 65 |
+
balance_trajectory_weights: True
|
| 66 |
+
ds_weights_alpha: 0.4
|
| 67 |
+
==================================================
|
| 68 |
+
|
| 69 |
+
Using 2 GPUs
|
| 70 |
+
|
| 71 |
+
================================================================================
|
| 72 |
+
Starting sweep branch: default
|
| 73 |
+
Sweep vars: {}
|
| 74 |
+
================================================================================
|
| 75 |
+
|
| 76 |
+
--------------------------------------------------------------------------------
|
| 77 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 78 |
+
Policy type: groot_rkd_v2_raw
|
| 79 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 80 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 81 |
+
Trainable preset: freeze_processing_line
|
| 82 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 83 |
+
--------------------------------------------------------------------------------
|
| 84 |
+
|
| 85 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 86 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 87 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 88 |
+
self.statistics[key] = torch.tensor(value)
|
| 89 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 90 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 91 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 92 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 93 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 94 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 95 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 96 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 97 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 98 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 99 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 100 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 101 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 102 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 103 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 104 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 105 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 106 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 107 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 108 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 109 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 110 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 111 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 112 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 113 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 114 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 115 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 116 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 117 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 118 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 119 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 120 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 121 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 122 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 123 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 124 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 125 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 126 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 127 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 128 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 129 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 130 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 131 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 132 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 133 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 134 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 135 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 136 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 137 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 138 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 139 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 140 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 141 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 142 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 143 |
+
0.75517122 0.7973985 ]
|
| 144 |
+
Loaded 26 datasets
|
| 145 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 146 |
+
Tune backbone vision tower: True
|
| 147 |
+
Tune backbone LLM: True
|
| 148 |
+
Tune action head projector: False
|
| 149 |
+
Tune action head DiT: False
|
| 150 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 151 |
+
|
| 152 |
+
==================================================
|
| 153 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 154 |
+
==================================================
|
| 155 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 156 |
+
dataset_soup: None
|
| 157 |
+
output_dir: /tmp/gr00t
|
| 158 |
+
output_root: None
|
| 159 |
+
data_config: panda_omron
|
| 160 |
+
batch_size: 32
|
| 161 |
+
max_steps: 300000
|
| 162 |
+
num_gpus: 2
|
| 163 |
+
save_steps: 20000
|
| 164 |
+
run_name: None
|
| 165 |
+
save_total_limit: 100
|
| 166 |
+
seed: 42
|
| 167 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 168 |
+
tune_llm: False
|
| 169 |
+
tune_visual: False
|
| 170 |
+
tune_projector: True
|
| 171 |
+
tune_diffusion_model: True
|
| 172 |
+
resume: False
|
| 173 |
+
learning_rate: 3e-05
|
| 174 |
+
weight_decay: 1e-05
|
| 175 |
+
warmup_ratio: 0.05
|
| 176 |
+
lora_rank: 0
|
| 177 |
+
lora_alpha: 16
|
| 178 |
+
lora_dropout: 0.1
|
| 179 |
+
lora_full_model: False
|
| 180 |
+
dataloader_num_workers: 8
|
| 181 |
+
report_to: wandb
|
| 182 |
+
embodiment_tag: new_embodiment
|
| 183 |
+
video_backend: opencv
|
| 184 |
+
balance_dataset_weights: True
|
| 185 |
+
balance_trajectory_weights: True
|
| 186 |
+
ds_weights_alpha: 0.4
|
| 187 |
+
==================================================
|
| 188 |
+
|
| 189 |
+
Using 2 GPUs
|
| 190 |
+
|
| 191 |
+
================================================================================
|
| 192 |
+
Starting sweep branch: default
|
| 193 |
+
Sweep vars: {}
|
| 194 |
+
================================================================================
|
| 195 |
+
|
| 196 |
+
--------------------------------------------------------------------------------
|
| 197 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 198 |
+
Policy type: groot_rkd_v2_raw
|
| 199 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 200 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 201 |
+
Trainable preset: freeze_processing_line
|
| 202 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 203 |
+
--------------------------------------------------------------------------------
|
| 204 |
+
|
| 205 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 206 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 207 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 208 |
+
self.statistics[key] = torch.tensor(value)
|
| 209 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 210 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 211 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 212 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 213 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 214 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 215 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 216 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 217 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 218 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 219 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 220 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 221 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 222 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 223 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 224 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 225 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 226 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 227 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 228 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 229 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 230 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 231 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 232 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 233 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 234 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 235 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 236 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 237 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 238 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 239 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 240 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 241 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 242 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 243 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 244 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 245 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 246 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 247 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 248 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 249 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 250 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 251 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 252 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 253 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 254 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 255 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 256 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 257 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 258 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 259 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 260 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 261 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 262 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 263 |
+
0.75517122 0.7973985 ]
|
| 264 |
+
Loaded 26 datasets
|
| 265 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 266 |
+
Tune backbone vision tower: True
|
| 267 |
+
Tune backbone LLM: True
|
| 268 |
+
Tune action head projector: False
|
| 269 |
+
Tune action head DiT: False
|
| 270 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 271 |
+
Tune backbone llm: False
|
| 272 |
+
Tune backbone visual: True
|
| 273 |
+
Total number of DiT parameters: 550386688
|
| 274 |
+
Tune backbone llm: False
|
| 275 |
+
Tune backbone visual: True
|
| 276 |
+
Total number of DiT parameters: 550386688
|
| 277 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 278 |
+
Tune action head projector: True
|
| 279 |
+
Tune action head diffusion model: True
|
| 280 |
+
|
| 281 |
+
Tune backbone llm: True
|
| 282 |
+
Tune backbone visual: True
|
| 283 |
+
Tune action head projector: False
|
| 284 |
+
Tune action head diffusion model: False
|
| 285 |
+
Action head trainable parameter: future_tokens.weight
|
| 286 |
+
Action head trainable parameter: vlln.weight
|
| 287 |
+
Action head trainable parameter: vlln.bias
|
| 288 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 289 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 290 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 291 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 292 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 293 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 294 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 295 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 296 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 297 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 298 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 299 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 300 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 301 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 302 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 303 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 304 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 305 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 306 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 307 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 308 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 309 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 310 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 311 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 312 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 313 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 314 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 315 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 316 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 317 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 318 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 319 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 320 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 321 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 322 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 323 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 324 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 325 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 326 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 327 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 328 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 329 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 330 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 331 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 332 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 333 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 334 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 335 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 336 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 337 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 338 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 339 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 340 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 341 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 342 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 343 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 344 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 345 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 346 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 347 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 348 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 349 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 350 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 351 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 352 |
+
Applied trainable preset: freeze_processing_line
|
| 353 |
+
Trainable parameter tensors after preset: 585
|
| 354 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 355 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 356 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 357 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 358 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 359 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 360 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 361 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 362 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 363 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 364 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 365 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 366 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 367 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 368 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 369 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 370 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 371 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 372 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 373 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 374 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 375 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 376 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 377 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 378 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 379 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 380 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 381 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 382 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 383 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 384 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 385 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 386 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 387 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 388 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 389 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 390 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 391 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 392 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 393 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 394 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 395 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 396 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 397 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 398 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 399 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 400 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 401 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 402 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 403 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 404 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 405 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 406 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 407 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 408 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 409 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 410 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 411 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 412 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 413 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 414 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 415 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 416 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 417 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 418 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 419 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 420 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 421 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 422 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 423 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 424 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 425 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 426 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 427 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 428 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 429 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 430 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 431 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 432 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 433 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 434 |
+
... 505 more
|
| 435 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
|
| 436 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 437 |
+
Tune action head projector: True
|
| 438 |
+
Tune action head diffusion model: True
|
| 439 |
+
|
| 440 |
+
Tune backbone llm: True
|
| 441 |
+
Tune backbone visual: True
|
| 442 |
+
Tune action head projector: False
|
| 443 |
+
Tune action head diffusion model: False
|
| 444 |
+
Action head trainable parameter: future_tokens.weight
|
| 445 |
+
Action head trainable parameter: vlln.weight
|
| 446 |
+
Action head trainable parameter: vlln.bias
|
| 447 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 448 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 449 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 450 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 451 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 452 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 453 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 454 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 455 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 456 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 457 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 458 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 459 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 460 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 461 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 462 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 463 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 464 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 465 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 466 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 467 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 468 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 469 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 470 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 471 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 472 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 473 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 474 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 475 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 476 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 477 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 478 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 479 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 480 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 481 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 482 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 483 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 484 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 485 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 486 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 487 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 488 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 489 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 490 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 491 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 492 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 493 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 494 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 495 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 496 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 497 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 498 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 499 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 500 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 501 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 502 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 503 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 504 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 505 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 506 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 507 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 508 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 509 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 510 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 511 |
+
Applied trainable preset: freeze_processing_line
|
| 512 |
+
Trainable parameter tensors after preset: 585
|
| 513 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 514 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 515 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 516 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 517 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 518 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 519 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 520 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 521 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 522 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 523 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 524 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 525 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 526 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 527 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 528 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 529 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 530 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 531 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 532 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 533 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 534 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 535 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 536 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 537 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 538 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 539 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 540 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 541 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 542 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 543 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 544 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 545 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 546 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 547 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 548 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 549 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 550 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 551 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 552 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 553 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 554 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 555 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 556 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 557 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 558 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 559 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 560 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 561 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 562 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 563 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 564 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 565 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 566 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 567 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 568 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 569 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 570 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 571 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 572 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 573 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 574 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 575 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 576 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 577 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 578 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 579 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 580 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 581 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 582 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 583 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 584 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 585 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 586 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 587 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 588 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 589 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 590 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 591 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 592 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 593 |
+
... 505 more
|
| 594 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
|
| 595 |
+
Run name: default_phase2
|
| 596 |
+
Run name: default_phase2
|
| 597 |
+
train dataloader length: 6873
|
| 598 |
+
train dataset length: 439854
|
| 599 |
+
GPU memory before training: 7.076685905456543 GB
|
| 600 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 601 |
+
train dataloader length: 6873
|
| 602 |
+
train dataset length: 439854
|
| 603 |
+
GPU memory before training: 7.076685905456543 GB
|
| 604 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 605 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
|
| 606 |
+
wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 607 |
+
wandb: setting up run e2phs2yb
|
| 608 |
+
wandb: Tracking run with wandb version 0.25.0
|
| 609 |
+
wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_141733-e2phs2yb
|
| 610 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 611 |
+
wandb: Syncing run default_phase2
|
| 612 |
+
wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 613 |
+
wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/e2phs2yb
|
| 614 |
+
|
| 615 |
0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
|
| 616 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 617 |
+
[rank1]: run_yaml_experiment(
|
| 618 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 619 |
+
[rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 620 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 621 |
+
[rank1]: experiment.train()
|
| 622 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 623 |
+
[rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 624 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 625 |
+
[rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 626 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 627 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 628 |
+
[rank1]: return inner_training_loop(
|
| 629 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 630 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 631 |
+
[rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 632 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 633 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 634 |
+
[rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 635 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 636 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 637 |
+
[rank1]: outputs = model(inputs)
|
| 638 |
+
[rank1]: ^^^^^^^^^^^^^
|
| 639 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 640 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 641 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 642 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 643 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 644 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 645 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 646 |
+
[rank1]: else self._run_ddp_forward(*inputs, **kwargs)
|
| 647 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 648 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 649 |
+
[rank1]: return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 650 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 651 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 652 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 653 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 654 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 655 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 656 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 657 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 658 |
+
[rank1]: return model_forward(*args, **kwargs)
|
| 659 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 660 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 661 |
+
[rank1]: return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 662 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 663 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 664 |
+
[rank1]: return func(*args, **kwargs)
|
| 665 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^
|
| 666 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 667 |
+
[rank1]: backbone_outputs = self.backbone(backbone_inputs)
|
| 668 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 669 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 670 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 671 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 672 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 673 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 674 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 675 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 676 |
+
[rank1]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 677 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 678 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 679 |
+
[rank1]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 680 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 681 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 682 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 683 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 684 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 685 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 686 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 687 |
+
[rank1]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 688 |
+
[rank1]: outputs = self.language_model(
|
| 689 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 690 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 691 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 692 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 693 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 694 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 695 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 696 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 697 |
+
[rank1]: output = func(self, *args, **kwargs)
|
| 698 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 699 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 700 |
+
[rank1]: return func(*args, **kwargs)
|
| 701 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^
|
| 702 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 866, in forward
|
| 703 |
+
[rank1]: logits = self.lm_head(hidden_states[:, slice_indices, :])
|
| 704 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 705 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 706 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 707 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 708 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 709 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 710 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 711 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
|
| 712 |
+
[rank1]: return F.linear(input, self.weight, self.bias)
|
| 713 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 714 |
+
[rank1]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 7.55 GiB. GPU 1 has a total capacity of 79.25 GiB of which 2.73 GiB is free. Including non-PyTorch memory, this process has 76.50 GiB memory in use. Of the allocated memory 74.82 GiB is allocated by PyTorch, and 1.05 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 715 |
+
wandb: updating run metadata
|
| 716 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 717 |
+
wandb: uploading config.yaml
|
| 718 |
+
wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/e2phs2yb
|
| 719 |
+
wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 720 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 721 |
+
wandb: Find logs at: ./wandb/run-20260615_141733-e2phs2yb/logs
|
| 722 |
+
Traceback (most recent call last):
|
| 723 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 724 |
+
run_yaml_experiment(
|
| 725 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 726 |
+
main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 727 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 728 |
+
experiment.train()
|
| 729 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 730 |
+
self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 731 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 732 |
+
return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 733 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 734 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 735 |
+
return inner_training_loop(
|
| 736 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 737 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 738 |
+
tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 739 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 740 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 741 |
+
loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 742 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 743 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 744 |
+
outputs = model(inputs)
|
| 745 |
+
^^^^^^^^^^^^^
|
| 746 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 747 |
+
return self._call_impl(*args, **kwargs)
|
| 748 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 749 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 750 |
+
return forward_call(*args, **kwargs)
|
| 751 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 752 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 753 |
+
else self._run_ddp_forward(*inputs, **kwargs)
|
| 754 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 755 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 756 |
+
return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 757 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 758 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 759 |
+
return self._call_impl(*args, **kwargs)
|
| 760 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 761 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 762 |
+
return forward_call(*args, **kwargs)
|
| 763 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 764 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 765 |
+
return model_forward(*args, **kwargs)
|
| 766 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 767 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 768 |
+
return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 769 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 770 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 771 |
+
return func(*args, **kwargs)
|
| 772 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 773 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 774 |
+
backbone_outputs = self.backbone(backbone_inputs)
|
| 775 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 776 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 777 |
+
return self._call_impl(*args, **kwargs)
|
| 778 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 779 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 780 |
+
return forward_call(*args, **kwargs)
|
| 781 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 782 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 783 |
+
eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 784 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 785 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 786 |
+
eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 787 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 788 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 789 |
+
return self._call_impl(*args, **kwargs)
|
| 790 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 791 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 792 |
+
return forward_call(*args, **kwargs)
|
| 793 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 794 |
+
File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 795 |
+
outputs = self.language_model(
|
| 796 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 797 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 798 |
+
return self._call_impl(*args, **kwargs)
|
| 799 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 800 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 801 |
+
return forward_call(*args, **kwargs)
|
| 802 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 803 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 804 |
+
output = func(self, *args, **kwargs)
|
| 805 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 806 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 807 |
+
return func(*args, **kwargs)
|
| 808 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 809 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 866, in forward
|
| 810 |
+
logits = self.lm_head(hidden_states[:, slice_indices, :])
|
| 811 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 812 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 813 |
+
return self._call_impl(*args, **kwargs)
|
| 814 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 815 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 816 |
+
return forward_call(*args, **kwargs)
|
| 817 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 818 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
|
| 819 |
+
return F.linear(input, self.weight, self.bias)
|
| 820 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 821 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 7.55 GiB. GPU 0 has a total capacity of 79.25 GiB of which 2.73 GiB is free. Including non-PyTorch memory, this process has 76.50 GiB memory in use. Of the allocated memory 74.82 GiB is allocated by PyTorch, and 1.05 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 822 |
+
[rank0]: Traceback (most recent call last):
|
| 823 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 824 |
+
[rank0]: run_yaml_experiment(
|
| 825 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 826 |
+
[rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 827 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 828 |
+
[rank0]: experiment.train()
|
| 829 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 830 |
+
[rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 831 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 832 |
+
[rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 833 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 834 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 835 |
+
[rank0]: return inner_training_loop(
|
| 836 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 837 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 838 |
+
[rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 839 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 840 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 841 |
+
[rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 842 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 843 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 844 |
+
[rank0]: outputs = model(inputs)
|
| 845 |
+
[rank0]: ^^^^^^^^^^^^^
|
| 846 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 847 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 848 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 849 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 850 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 851 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 852 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 853 |
+
[rank0]: else self._run_ddp_forward(*inputs, **kwargs)
|
| 854 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 855 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 856 |
+
[rank0]: return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 857 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 858 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 859 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 860 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 861 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 862 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 863 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 864 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 865 |
+
[rank0]: return model_forward(*args, **kwargs)
|
| 866 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 867 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 868 |
+
[rank0]: return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 869 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 870 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 871 |
+
[rank0]: return func(*args, **kwargs)
|
| 872 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^
|
| 873 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 874 |
+
[rank0]: backbone_outputs = self.backbone(backbone_inputs)
|
| 875 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 876 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 877 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 878 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 879 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 880 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 881 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 882 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 883 |
+
[rank0]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 884 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 885 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 886 |
+
[rank0]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 887 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 888 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 889 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 890 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 891 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 892 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 893 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 894 |
+
[rank0]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 895 |
+
[rank0]: outputs = self.language_model(
|
| 896 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 897 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 898 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 899 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 900 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 901 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 902 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 903 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 904 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 905 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 906 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 907 |
+
[rank0]: return func(*args, **kwargs)
|
| 908 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^
|
| 909 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 866, in forward
|
| 910 |
+
[rank0]: logits = self.lm_head(hidden_states[:, slice_indices, :])
|
| 911 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 912 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 913 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 914 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 915 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 916 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 917 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 918 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
|
| 919 |
+
[rank0]: return F.linear(input, self.weight, self.bias)
|
| 920 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 921 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 7.55 GiB. GPU 0 has a total capacity of 79.25 GiB of which 2.73 GiB is free. Including non-PyTorch memory, this process has 76.50 GiB memory in use. Of the allocated memory 74.82 GiB is allocated by PyTorch, and 1.05 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 922 |
+
[rank0]:[W615 14:17:43.385716740 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
|
| 923 |
+
W0615 14:17:43.882000 478197 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 484952 closing signal SIGTERM
|
| 924 |
+
E0615 14:17:44.246000 478197 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 484957) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
|
| 925 |
+
Traceback (most recent call last):
|
| 926 |
+
File "<frozen runpy>", line 198, in _run_module_as_main
|
| 927 |
+
File "<frozen runpy>", line 88, in _run_code
|
| 928 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
|
| 929 |
+
main()
|
| 930 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
|
| 931 |
+
return f(*args, **kwargs)
|
| 932 |
+
^^^^^^^^^^^^^^^^^^
|
| 933 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
|
| 934 |
+
run(args)
|
| 935 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
|
| 936 |
+
elastic_launch(
|
| 937 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
|
| 938 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 939 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 940 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
|
| 941 |
+
raise ChildFailedError(
|
| 942 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 943 |
+
============================================================
|
| 944 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
|
| 945 |
+
------------------------------------------------------------
|
| 946 |
+
Failures:
|
| 947 |
+
<NO_OTHER_FAILURES>
|
| 948 |
+
------------------------------------------------------------
|
| 949 |
+
Root Cause (first observed failure):
|
| 950 |
+
[0]:
|
| 951 |
+
time : 2026-06-15_14:17:43
|
| 952 |
+
host : worker1
|
| 953 |
+
rank : 1 (local_rank: 1)
|
| 954 |
+
exitcode : 1 (pid: 484957)
|
| 955 |
+
error_file: <N/A>
|
| 956 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 957 |
+
============================================================
|
| 958 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 959 |
+
|
| 960 |
+
==================================================
|
| 961 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 962 |
+
==================================================
|
| 963 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 964 |
+
dataset_soup: None
|
| 965 |
+
output_dir: /tmp/gr00t
|
| 966 |
+
output_root: None
|
| 967 |
+
data_config: panda_omron
|
| 968 |
+
batch_size: 32
|
| 969 |
+
max_steps: 300000
|
| 970 |
+
num_gpus: 2
|
| 971 |
+
save_steps: 20000
|
| 972 |
+
run_name: None
|
| 973 |
+
save_total_limit: 100
|
| 974 |
+
seed: 42
|
| 975 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 976 |
+
tune_llm: False
|
| 977 |
+
tune_visual: False
|
| 978 |
+
tune_projector: True
|
| 979 |
+
tune_diffusion_model: True
|
| 980 |
+
resume: False
|
| 981 |
+
learning_rate: 3e-05
|
| 982 |
+
weight_decay: 1e-05
|
| 983 |
+
warmup_ratio: 0.05
|
| 984 |
+
lora_rank: 0
|
| 985 |
+
lora_alpha: 16
|
| 986 |
+
lora_dropout: 0.1
|
| 987 |
+
lora_full_model: False
|
| 988 |
+
dataloader_num_workers: 8
|
| 989 |
+
report_to: wandb
|
| 990 |
+
embodiment_tag: new_embodiment
|
| 991 |
+
video_backend: opencv
|
| 992 |
+
balance_dataset_weights: True
|
| 993 |
+
balance_trajectory_weights: True
|
| 994 |
+
ds_weights_alpha: 0.4
|
| 995 |
+
==================================================
|
| 996 |
+
|
| 997 |
+
Using 2 GPUs
|
| 998 |
+
Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '32', '--num-gpus', '2']
|
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs32_gpu2_single_probe.log
ADDED
|
@@ -0,0 +1,401 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/30000 [00:00<?, ?it/s]wandb: updating run metadata; uploading wandb-metadata.json; uploading requirements.txt
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 2 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 3 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 4 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 5 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 6 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 7 |
+
check_for_updates()
|
| 8 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 9 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 10 |
+
|
| 11 |
+
==================================================
|
| 12 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 13 |
+
==================================================
|
| 14 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 15 |
+
dataset_soup: None
|
| 16 |
+
output_dir: /tmp/gr00t
|
| 17 |
+
output_root: None
|
| 18 |
+
data_config: panda_omron
|
| 19 |
+
batch_size: 32
|
| 20 |
+
max_steps: 300000
|
| 21 |
+
num_gpus: 1
|
| 22 |
+
save_steps: 20000
|
| 23 |
+
run_name: None
|
| 24 |
+
save_total_limit: 100
|
| 25 |
+
seed: 42
|
| 26 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 27 |
+
tune_llm: False
|
| 28 |
+
tune_visual: False
|
| 29 |
+
tune_projector: True
|
| 30 |
+
tune_diffusion_model: True
|
| 31 |
+
resume: False
|
| 32 |
+
learning_rate: 3e-05
|
| 33 |
+
weight_decay: 1e-05
|
| 34 |
+
warmup_ratio: 0.05
|
| 35 |
+
lora_rank: 0
|
| 36 |
+
lora_alpha: 16
|
| 37 |
+
lora_dropout: 0.1
|
| 38 |
+
lora_full_model: False
|
| 39 |
+
dataloader_num_workers: 8
|
| 40 |
+
report_to: wandb
|
| 41 |
+
embodiment_tag: new_embodiment
|
| 42 |
+
video_backend: opencv
|
| 43 |
+
balance_dataset_weights: True
|
| 44 |
+
balance_trajectory_weights: True
|
| 45 |
+
ds_weights_alpha: 0.4
|
| 46 |
+
==================================================
|
| 47 |
+
|
| 48 |
+
Using 1 GPUs
|
| 49 |
+
|
| 50 |
+
================================================================================
|
| 51 |
+
Starting sweep branch: default
|
| 52 |
+
Sweep vars: {}
|
| 53 |
+
================================================================================
|
| 54 |
+
|
| 55 |
+
--------------------------------------------------------------------------------
|
| 56 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 57 |
+
Policy type: groot_rkd_v2_raw
|
| 58 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 59 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 60 |
+
Trainable preset: freeze_processing_line
|
| 61 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 62 |
+
--------------------------------------------------------------------------------
|
| 63 |
+
|
| 64 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 65 |
+
Using 100 subset demos for filter_key: 100_demos/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 66 |
+
self.statistics[key] = torch.tensor(value)
|
| 67 |
+
|
| 68 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 69 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 70 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 71 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 72 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 73 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 74 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 75 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 76 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 77 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 78 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 79 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 80 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 81 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 82 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 83 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 84 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 85 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 86 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 87 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 88 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 89 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 90 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 91 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 92 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 93 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 94 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 95 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 96 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 97 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 98 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 99 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 100 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 101 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 102 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 103 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 104 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 105 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 106 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 107 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 108 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 109 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 110 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 111 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 112 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 113 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 114 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 115 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 116 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 117 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 118 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 119 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 120 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 121 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 122 |
+
0.75517122 0.7973985 ]
|
| 123 |
+
Loaded 26 datasets
|
| 124 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 125 |
+
Tune backbone vision tower: True
|
| 126 |
+
Tune backbone LLM: True
|
| 127 |
+
Tune action head projector: False
|
| 128 |
+
Tune action head DiT: False
|
| 129 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 130 |
+
Tune backbone llm: False
|
| 131 |
+
Tune backbone visual: True
|
| 132 |
+
Total number of DiT parameters: 550386688
|
| 133 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 134 |
+
Tune action head projector: True
|
| 135 |
+
Tune action head diffusion model: True
|
| 136 |
+
|
| 137 |
+
Tune backbone llm: True
|
| 138 |
+
Tune backbone visual: True
|
| 139 |
+
Tune action head projector: False
|
| 140 |
+
Tune action head diffusion model: False
|
| 141 |
+
Action head trainable parameter: future_tokens.weight
|
| 142 |
+
Action head trainable parameter: vlln.weight
|
| 143 |
+
Action head trainable parameter: vlln.bias
|
| 144 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 145 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 146 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 147 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 148 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 149 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 150 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 151 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 152 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 153 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 154 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 155 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 156 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 157 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 158 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 159 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 160 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 161 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 162 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 163 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 164 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 165 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 166 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 167 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 168 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 169 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 170 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 171 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 172 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 173 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 174 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 175 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 176 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 177 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 178 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 179 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 180 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 181 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 182 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 183 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 184 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 185 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 186 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 187 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 188 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 189 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 190 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 191 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 192 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 193 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 194 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 195 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 196 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 197 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 198 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 199 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 200 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 201 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 202 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 203 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 204 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 205 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 206 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 207 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 208 |
+
Applied trainable preset: freeze_processing_line
|
| 209 |
+
Trainable parameter tensors after preset: 584
|
| 210 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 211 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 212 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 213 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 214 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 215 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 216 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 217 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 218 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 219 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 220 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 221 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 222 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 223 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 224 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 225 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 226 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 227 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 228 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 229 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 230 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 231 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 232 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 233 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 234 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 235 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 236 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 237 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 238 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 239 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 240 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 241 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 242 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 243 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 244 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 245 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 246 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 247 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 248 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 249 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 250 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 251 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 252 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 253 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 254 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 255 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 256 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 257 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 258 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 259 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 260 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 261 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 262 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 263 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 264 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 265 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 266 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 267 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 268 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 269 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 270 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 271 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 272 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 273 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 274 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 275 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 276 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 277 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 278 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 279 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 280 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 281 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 282 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 283 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 284 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 285 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 286 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 287 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 288 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 289 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 290 |
+
... 504 more
|
| 291 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
|
| 292 |
+
Run name: default_phase2
|
| 293 |
+
train dataloader length: 13746
|
| 294 |
+
train dataset length: 439854
|
| 295 |
+
GPU memory before training: 7.076685905456543 GB
|
| 296 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
|
| 297 |
+
wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 298 |
+
wandb: setting up run kloq3gk7
|
| 299 |
+
wandb: Tracking run with wandb version 0.25.0
|
| 300 |
+
wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_143415-kloq3gk7
|
| 301 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 302 |
+
wandb: Syncing run default_phase2
|
| 303 |
+
wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 304 |
+
wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/kloq3gk7
|
| 305 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 306 |
+
|
| 307 |
0%| | 0/30000 [00:00<?, ?it/s]wandb: updating run metadata; uploading wandb-metadata.json; uploading requirements.txt
|
| 308 |
+
wandb: uploading wandb-metadata.json; uploading requirements.txt
|
| 309 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 310 |
+
wandb: uploading summary
|
| 311 |
+
wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/kloq3gk7
|
| 312 |
+
wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 313 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 314 |
+
wandb: Find logs at: ./wandb/run-20260615_143415-kloq3gk7/logs
|
| 315 |
+
Traceback (most recent call last):
|
| 316 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1015, in <module>
|
| 317 |
+
run_yaml_experiment(
|
| 318 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
|
| 319 |
+
main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 320 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
|
| 321 |
+
experiment.train()
|
| 322 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 323 |
+
self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 324 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 325 |
+
return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 326 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 327 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 328 |
+
return inner_training_loop(
|
| 329 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 330 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 331 |
+
tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 332 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 333 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 334 |
+
loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 335 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 336 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 337 |
+
outputs = model(inputs)
|
| 338 |
+
^^^^^^^^^^^^^
|
| 339 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 340 |
+
return self._call_impl(*args, **kwargs)
|
| 341 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 342 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 343 |
+
return forward_call(*args, **kwargs)
|
| 344 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 345 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 346 |
+
return model_forward(*args, **kwargs)
|
| 347 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 348 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 349 |
+
return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 350 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 351 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 352 |
+
return func(*args, **kwargs)
|
| 353 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 354 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 355 |
+
backbone_outputs = self.backbone(backbone_inputs)
|
| 356 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 357 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 358 |
+
return self._call_impl(*args, **kwargs)
|
| 359 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 360 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 361 |
+
return forward_call(*args, **kwargs)
|
| 362 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 363 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 364 |
+
eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 365 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 366 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 367 |
+
eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 368 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 369 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 370 |
+
return self._call_impl(*args, **kwargs)
|
| 371 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 372 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 373 |
+
return forward_call(*args, **kwargs)
|
| 374 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 375 |
+
File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 376 |
+
outputs = self.language_model(
|
| 377 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 378 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 379 |
+
return self._call_impl(*args, **kwargs)
|
| 380 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 381 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 382 |
+
return forward_call(*args, **kwargs)
|
| 383 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 384 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 385 |
+
output = func(self, *args, **kwargs)
|
| 386 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 387 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 388 |
+
return func(*args, **kwargs)
|
| 389 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 390 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 866, in forward
|
| 391 |
+
logits = self.lm_head(hidden_states[:, slice_indices, :])
|
| 392 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 393 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 394 |
+
return self._call_impl(*args, **kwargs)
|
| 395 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 396 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 397 |
+
return forward_call(*args, **kwargs)
|
| 398 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 399 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
|
| 400 |
+
return F.linear(input, self.weight, self.bias)
|
| 401 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 402 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 7.55 GiB. GPU 0 has a total capacity of 79.25 GiB of which 6.50 GiB is free. Including non-PyTorch memory, this process has 72.73 GiB memory in use. Of the allocated memory 71.75 GiB is allocated by PyTorch, and 492.34 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs64_gpu2_3.log
ADDED
|
@@ -0,0 +1,1095 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 2 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 3 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 4 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 5 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 6 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 7 |
+
check_for_updates()
|
| 8 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 9 |
+
|
| 10 |
+
*****************************************
|
| 11 |
+
Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
|
| 12 |
+
*****************************************
|
| 13 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 14 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 15 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 16 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 17 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 18 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 19 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 20 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 21 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 22 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 23 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 24 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 25 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 26 |
+
check_for_updates()
|
| 27 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 28 |
+
check_for_updates()
|
| 29 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 30 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 31 |
+
|
| 32 |
+
==================================================
|
| 33 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 34 |
+
==================================================
|
| 35 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 36 |
+
dataset_soup: None
|
| 37 |
+
output_dir: /tmp/gr00t
|
| 38 |
+
output_root: None
|
| 39 |
+
data_config: panda_omron
|
| 40 |
+
batch_size: 64
|
| 41 |
+
max_steps: 300000
|
| 42 |
+
num_gpus: 2
|
| 43 |
+
save_steps: 20000
|
| 44 |
+
run_name: None
|
| 45 |
+
save_total_limit: 100
|
| 46 |
+
seed: 42
|
| 47 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 48 |
+
tune_llm: False
|
| 49 |
+
tune_visual: False
|
| 50 |
+
tune_projector: True
|
| 51 |
+
tune_diffusion_model: True
|
| 52 |
+
resume: False
|
| 53 |
+
learning_rate: 3e-05
|
| 54 |
+
weight_decay: 1e-05
|
| 55 |
+
warmup_ratio: 0.05
|
| 56 |
+
lora_rank: 0
|
| 57 |
+
lora_alpha: 16
|
| 58 |
+
lora_dropout: 0.1
|
| 59 |
+
lora_full_model: False
|
| 60 |
+
dataloader_num_workers: 8
|
| 61 |
+
report_to: wandb
|
| 62 |
+
embodiment_tag: new_embodiment
|
| 63 |
+
video_backend: opencv
|
| 64 |
+
balance_dataset_weights: True
|
| 65 |
+
balance_trajectory_weights: True
|
| 66 |
+
ds_weights_alpha: 0.4
|
| 67 |
+
==================================================
|
| 68 |
+
|
| 69 |
+
Using 2 GPUs
|
| 70 |
+
|
| 71 |
+
================================================================================
|
| 72 |
+
Starting sweep branch: default
|
| 73 |
+
Sweep vars: {}
|
| 74 |
+
================================================================================
|
| 75 |
+
|
| 76 |
+
--------------------------------------------------------------------------------
|
| 77 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 78 |
+
Policy type: groot_rkd_v2_raw
|
| 79 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 80 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 81 |
+
Trainable preset: freeze_processing_line
|
| 82 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 83 |
+
--------------------------------------------------------------------------------
|
| 84 |
+
|
| 85 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 86 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 87 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 88 |
+
self.statistics[key] = torch.tensor(value)
|
| 89 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 90 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 91 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 92 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 93 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 94 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 95 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 96 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 97 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 98 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 99 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 100 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 101 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 102 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 103 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 104 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 105 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 106 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 107 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 108 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 109 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 110 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 111 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 112 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 113 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 114 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 115 |
+
|
| 116 |
+
==================================================
|
| 117 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 118 |
+
==================================================
|
| 119 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 120 |
+
dataset_soup: None
|
| 121 |
+
output_dir: /tmp/gr00t
|
| 122 |
+
output_root: None
|
| 123 |
+
data_config: panda_omron
|
| 124 |
+
batch_size: 64
|
| 125 |
+
max_steps: 300000
|
| 126 |
+
num_gpus: 2
|
| 127 |
+
save_steps: 20000
|
| 128 |
+
run_name: None
|
| 129 |
+
save_total_limit: 100Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 130 |
+
seed: 42
|
| 131 |
+
|
| 132 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 133 |
+
tune_llm: False
|
| 134 |
+
tune_visual: False
|
| 135 |
+
tune_projector: True
|
| 136 |
+
tune_diffusion_model: True
|
| 137 |
+
resume: False
|
| 138 |
+
learning_rate: 3e-05
|
| 139 |
+
weight_decay: 1e-05
|
| 140 |
+
warmup_ratio: 0.05
|
| 141 |
+
lora_rank: 0
|
| 142 |
+
lora_alpha: 16
|
| 143 |
+
lora_dropout: 0.1
|
| 144 |
+
lora_full_model: False
|
| 145 |
+
dataloader_num_workers: 8
|
| 146 |
+
report_to: wandb
|
| 147 |
+
embodiment_tag: new_embodiment
|
| 148 |
+
video_backend: opencv
|
| 149 |
+
balance_dataset_weights: True
|
| 150 |
+
balance_trajectory_weights: True
|
| 151 |
+
ds_weights_alpha: 0.4
|
| 152 |
+
==================================================
|
| 153 |
+
|
| 154 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 155 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 156 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 157 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 158 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 159 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 160 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 161 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 162 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 163 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 164 |
+
Using 2 GPUs
|
| 165 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 166 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 167 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 168 |
+
|
| 169 |
+
================================================================================
|
| 170 |
+
Starting sweep branch: default
|
| 171 |
+
Sweep vars: {}
|
| 172 |
+
================================================================================
|
| 173 |
+
|
| 174 |
+
--------------------------------------------------------------------------------
|
| 175 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 176 |
+
Policy type: groot_rkd_v2_raw
|
| 177 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 178 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 179 |
+
Trainable preset: freeze_processing_line
|
| 180 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 181 |
+
--------------------------------------------------------------------------------
|
| 182 |
+
|
| 183 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 184 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 185 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 186 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 187 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 188 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 189 |
+
self.statistics[key] = torch.tensor(value)
|
| 190 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 191 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 192 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 193 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 194 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 195 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 196 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 197 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 198 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 199 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 200 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 201 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 202 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 203 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 204 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 205 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 206 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 207 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 208 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 209 |
+
0.75517122 0.7973985 ]
|
| 210 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 211 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 212 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 213 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 214 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 215 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 216 |
+
Loaded 26 datasets
|
| 217 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 218 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 219 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 220 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 221 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 222 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 223 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 224 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 225 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 226 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 227 |
+
Tune backbone vision tower: True
|
| 228 |
+
Tune backbone LLM: True
|
| 229 |
+
Tune action head projector: False
|
| 230 |
+
Tune action head DiT: False
|
| 231 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 232 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 233 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 234 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 235 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 236 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 237 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 238 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 239 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 240 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 241 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 242 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 243 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 244 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 245 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 246 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 247 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 248 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 249 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 250 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 251 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 252 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 253 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 254 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 255 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 256 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 257 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 258 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 259 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 260 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 261 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 262 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 263 |
+
0.75517122 0.7973985 ]
|
| 264 |
+
Loaded 26 datasets
|
| 265 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 266 |
+
Tune backbone vision tower: True
|
| 267 |
+
Tune backbone LLM: True
|
| 268 |
+
Tune action head projector: False
|
| 269 |
+
Tune action head DiT: False
|
| 270 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 271 |
+
Tune backbone llm: False
|
| 272 |
+
Tune backbone visual: True
|
| 273 |
+
Total number of DiT parameters: 550386688
|
| 274 |
+
Tune backbone llm: False
|
| 275 |
+
Tune backbone visual: True
|
| 276 |
+
Total number of DiT parameters: 550386688
|
| 277 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 278 |
+
Tune action head projector: True
|
| 279 |
+
Tune action head diffusion model: True
|
| 280 |
+
|
| 281 |
+
Tune backbone llm: True
|
| 282 |
+
Tune backbone visual: True
|
| 283 |
+
Tune action head projector: False
|
| 284 |
+
Tune action head diffusion model: False
|
| 285 |
+
Action head trainable parameter: future_tokens.weight
|
| 286 |
+
Action head trainable parameter: vlln.weight
|
| 287 |
+
Action head trainable parameter: vlln.bias
|
| 288 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 289 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 290 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 291 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 292 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 293 |
+
Total number of SelfAttentionTransformer parameters: Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 294 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 295 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 296 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 297 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 298 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 299 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias201433088
|
| 300 |
+
|
| 301 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 302 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 303 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 304 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 305 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 306 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 307 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 308 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 309 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 310 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 311 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 312 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 313 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 314 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 315 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 316 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 317 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 318 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 319 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 320 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 321 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 322 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 323 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 324 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 325 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 326 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 327 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 328 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 329 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 330 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 331 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 332 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 333 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 334 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 335 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 336 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 337 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 338 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 339 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 340 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 341 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 342 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 343 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 344 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 345 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 346 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 347 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 348 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 349 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 350 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 351 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 352 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 353 |
+
Tune action head projector: True
|
| 354 |
+
Tune action head diffusion model: True
|
| 355 |
+
Applied trainable preset: freeze_processing_line
|
| 356 |
+
Trainable parameter tensors after preset: 585
|
| 357 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 358 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 359 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 360 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 361 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 362 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 363 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 364 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 365 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 366 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 367 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 368 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 369 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 370 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 371 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 372 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 373 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 374 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 375 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 376 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 377 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 378 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 379 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 380 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 381 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 382 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 383 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 384 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 385 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 386 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 387 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 388 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 389 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 390 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 391 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 392 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 393 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 394 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 395 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 396 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 397 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 398 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 399 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 400 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 401 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 402 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 403 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 404 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 405 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 406 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 407 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 408 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 409 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 410 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 411 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 412 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 413 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 414 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 415 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 416 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 417 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 418 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 419 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 420 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 421 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 422 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 423 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 424 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 425 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 426 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 427 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 428 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 429 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 430 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 431 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 432 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 433 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 434 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 435 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 436 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 437 |
+
... 505 more
|
| 438 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
|
| 439 |
+
|
| 440 |
+
|
| 441 |
+
Tune backbone llm: True
|
| 442 |
+
Tune backbone visual: True
|
| 443 |
+
Tune action head projector: False
|
| 444 |
+
Tune action head diffusion model: False
|
| 445 |
+
Action head trainable parameter: future_tokens.weight
|
| 446 |
+
Action head trainable parameter: vlln.weight
|
| 447 |
+
Action head trainable parameter: vlln.bias
|
| 448 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 449 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 450 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 451 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 452 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 453 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 454 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 455 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 456 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 457 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 458 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 459 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 460 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 461 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 462 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 463 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 464 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 465 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 466 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 467 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 468 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 469 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 470 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 471 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 472 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 473 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 474 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 475 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 476 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 477 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 478 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 479 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 480 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 481 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 482 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 483 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 484 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 485 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 486 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 487 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 488 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 489 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 490 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 491 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 492 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 493 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 494 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 495 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 496 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 497 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 498 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 499 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 500 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 501 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 502 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 503 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 504 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 505 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 506 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 507 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 508 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 509 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 510 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 511 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 512 |
+
Applied trainable preset: freeze_processing_line
|
| 513 |
+
Trainable parameter tensors after preset: 585
|
| 514 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 515 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 516 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 517 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 518 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 519 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 520 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 521 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 522 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 523 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 524 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 525 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 526 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 527 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 528 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 529 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 530 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 531 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 532 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 533 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 534 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 535 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 536 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 537 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 538 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 539 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 540 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 541 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 542 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 543 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 544 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 545 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 546 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 547 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 548 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 549 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 550 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 551 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 552 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 553 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 554 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 555 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 556 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 557 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 558 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 559 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 560 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 561 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 562 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 563 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 564 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 565 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 566 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 567 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 568 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 569 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 570 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 571 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 572 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 573 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 574 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 575 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 576 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 577 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 578 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 579 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 580 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 581 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 582 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 583 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 584 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 585 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 586 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 587 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 588 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 589 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 590 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 591 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 592 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 593 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 594 |
+
... 505 more
|
| 595 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(1,225,315,328), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 1225315328, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1655357376, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.6076571262506297}
|
| 596 |
+
Run name: default_phase2
|
| 597 |
+
train dataloader length: 3437
|
| 598 |
+
train dataset length: 439854
|
| 599 |
+
GPU memory before training: 7.076685905456543 GB
|
| 600 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 601 |
+
train dataloader length: 3437
|
| 602 |
+
train dataset length: 439854
|
| 603 |
+
GPU memory before training: 7.076685905456543 GB
|
| 604 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 605 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
|
| 606 |
+
wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 607 |
+
wandb: setting up run 3uyg5mqq
|
| 608 |
+
wandb: Tracking run with wandb version 0.25.0
|
| 609 |
+
wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_141245-3uyg5mqq
|
| 610 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 611 |
+
wandb: Syncing run default_phase2
|
| 612 |
+
wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 613 |
+
wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/3uyg5mqq
|
| 614 |
+
|
| 615 |
0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
|
| 616 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 617 |
+
[rank1]: run_yaml_experiment(
|
| 618 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 619 |
+
[rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 620 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 621 |
+
[rank1]: experiment.train()
|
| 622 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 623 |
+
[rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 624 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 625 |
+
[rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 626 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 627 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 628 |
+
[rank1]: return inner_training_loop(
|
| 629 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 630 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 631 |
+
[rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 632 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 633 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 634 |
+
[rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 635 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 636 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 637 |
+
[rank1]: outputs = model(inputs)
|
| 638 |
+
[rank1]: ^^^^^^^^^^^^^
|
| 639 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 640 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 641 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 642 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 643 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 644 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 645 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 646 |
+
[rank1]: else self._run_ddp_forward(*inputs, **kwargs)
|
| 647 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 648 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 649 |
+
[rank1]: return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 650 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 651 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 652 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 653 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 654 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 655 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 656 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 657 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 658 |
+
[rank1]: return model_forward(*args, **kwargs)
|
| 659 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 660 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 661 |
+
[rank1]: return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 662 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 663 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 664 |
+
[rank1]: return func(*args, **kwargs)
|
| 665 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^
|
| 666 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 667 |
+
[rank1]: backbone_outputs = self.backbone(backbone_inputs)
|
| 668 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 669 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 670 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 671 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 672 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 673 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 674 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 675 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 676 |
+
[rank1]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 677 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 678 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 679 |
+
[rank1]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 680 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 681 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 682 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 683 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 684 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 685 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 686 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 687 |
+
[rank1]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 688 |
+
[rank1]: outputs = self.language_model(
|
| 689 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 690 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 691 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 692 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 693 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 694 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 695 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 696 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 697 |
+
[rank1]: output = func(self, *args, **kwargs)
|
| 698 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 699 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 700 |
+
[rank1]: return func(*args, **kwargs)
|
| 701 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^
|
| 702 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 703 |
+
[rank1]: outputs: BaseModelOutputWithPast = self.model(
|
| 704 |
+
[rank1]: ^^^^^^^^^^^
|
| 705 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 706 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 707 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 708 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 709 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 710 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 711 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 712 |
+
[rank1]: output = func(self, *args, **kwargs)
|
| 713 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 714 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 715 |
+
[rank1]: layer_outputs = decoder_layer(
|
| 716 |
+
[rank1]: ^^^^^^^^^^^^^^
|
| 717 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 718 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 719 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 720 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 721 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 722 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 723 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
|
| 724 |
+
[rank1]: hidden_states = self.mlp(hidden_states)
|
| 725 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 726 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 727 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 728 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 729 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 730 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 731 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 732 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
|
| 733 |
+
[rank1]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 734 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 735 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 736 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 737 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 738 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 739 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 740 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 741 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/activation.py", line 432, in forward
|
| 742 |
+
[rank1]: return F.silu(input, inplace=self.inplace)
|
| 743 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 744 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/functional.py", line 2380, in silu
|
| 745 |
+
[rank1]: return torch._C._nn.silu(input)
|
| 746 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^
|
| 747 |
+
[rank1]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 1 has a total capacity of 79.25 GiB of which 469.94 MiB is free. Including non-PyTorch memory, this process has 78.77 GiB memory in use. Of the allocated memory 76.81 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 748 |
+
wandb: updating run metadata
|
| 749 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 750 |
+
wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/3uyg5mqq
|
| 751 |
+
wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 752 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 753 |
+
wandb: Find logs at: ./wandb/run-20260615_141245-3uyg5mqq/logs
|
| 754 |
+
Traceback (most recent call last):
|
| 755 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 756 |
+
run_yaml_experiment(
|
| 757 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 758 |
+
main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 759 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 760 |
+
experiment.train()
|
| 761 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 762 |
+
self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 763 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 764 |
+
return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 765 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 766 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 767 |
+
return inner_training_loop(
|
| 768 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 769 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 770 |
+
tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 771 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 772 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 773 |
+
loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 774 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 775 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 776 |
+
outputs = model(inputs)
|
| 777 |
+
^^^^^^^^^^^^^
|
| 778 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 779 |
+
return self._call_impl(*args, **kwargs)
|
| 780 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 781 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 782 |
+
return forward_call(*args, **kwargs)
|
| 783 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 784 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 785 |
+
else self._run_ddp_forward(*inputs, **kwargs)
|
| 786 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 787 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 788 |
+
return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 789 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 790 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 791 |
+
return self._call_impl(*args, **kwargs)
|
| 792 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 793 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 794 |
+
return forward_call(*args, **kwargs)
|
| 795 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 796 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 797 |
+
return model_forward(*args, **kwargs)
|
| 798 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 799 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 800 |
+
return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 801 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 802 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 803 |
+
return func(*args, **kwargs)
|
| 804 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 805 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 806 |
+
backbone_outputs = self.backbone(backbone_inputs)
|
| 807 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 808 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 809 |
+
return self._call_impl(*args, **kwargs)
|
| 810 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 811 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 812 |
+
return forward_call(*args, **kwargs)
|
| 813 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 814 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 815 |
+
eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 816 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 817 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 818 |
+
eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 819 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 820 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 821 |
+
return self._call_impl(*args, **kwargs)
|
| 822 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 823 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 824 |
+
return forward_call(*args, **kwargs)
|
| 825 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 826 |
+
File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 827 |
+
outputs = self.language_model(
|
| 828 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 829 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 830 |
+
return self._call_impl(*args, **kwargs)
|
| 831 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 832 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 833 |
+
return forward_call(*args, **kwargs)
|
| 834 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 835 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 836 |
+
output = func(self, *args, **kwargs)
|
| 837 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 838 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 839 |
+
return func(*args, **kwargs)
|
| 840 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 841 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 842 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 843 |
+
^^^^^^^^^^^
|
| 844 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 845 |
+
return self._call_impl(*args, **kwargs)
|
| 846 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 847 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 848 |
+
return forward_call(*args, **kwargs)
|
| 849 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 850 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 851 |
+
output = func(self, *args, **kwargs)
|
| 852 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 853 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 854 |
+
layer_outputs = decoder_layer(
|
| 855 |
+
^^^^^^^^^^^^^^
|
| 856 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 857 |
+
return self._call_impl(*args, **kwargs)
|
| 858 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 859 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 860 |
+
return forward_call(*args, **kwargs)
|
| 861 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 862 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
|
| 863 |
+
hidden_states = self.mlp(hidden_states)
|
| 864 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 865 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 866 |
+
return self._call_impl(*args, **kwargs)
|
| 867 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 868 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 869 |
+
return forward_call(*args, **kwargs)
|
| 870 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 871 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
|
| 872 |
+
down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 873 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 874 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 875 |
+
return self._call_impl(*args, **kwargs)
|
| 876 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 877 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 878 |
+
return forward_call(*args, **kwargs)
|
| 879 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 880 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/activation.py", line 432, in forward
|
| 881 |
+
return F.silu(input, inplace=self.inplace)
|
| 882 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 883 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/functional.py", line 2380, in silu
|
| 884 |
+
return torch._C._nn.silu(input)
|
| 885 |
+
^^^^^^^^^^^^^^^^^^^^^^^^
|
| 886 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 469.94 MiB is free. Including non-PyTorch memory, this process has 78.77 GiB memory in use. Of the allocated memory 76.81 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 887 |
+
[rank0]: Traceback (most recent call last):
|
| 888 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1024, in <module>
|
| 889 |
+
[rank0]: run_yaml_experiment(
|
| 890 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 973, in run_yaml_experiment
|
| 891 |
+
[rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 892 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 707, in main
|
| 893 |
+
[rank0]: experiment.train()
|
| 894 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 895 |
+
[rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 896 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 897 |
+
[rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 898 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 899 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 900 |
+
[rank0]: return inner_training_loop(
|
| 901 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 902 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 903 |
+
[rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 904 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 905 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 906 |
+
[rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 907 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 908 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 909 |
+
[rank0]: outputs = model(inputs)
|
| 910 |
+
[rank0]: ^^^^^^^^^^^^^
|
| 911 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 912 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 913 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 914 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 915 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 916 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 917 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 918 |
+
[rank0]: else self._run_ddp_forward(*inputs, **kwargs)
|
| 919 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 920 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 921 |
+
[rank0]: return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 922 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 923 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 924 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 925 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 926 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 927 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 928 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 929 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 930 |
+
[rank0]: return model_forward(*args, **kwargs)
|
| 931 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 932 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 933 |
+
[rank0]: return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 934 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 935 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 936 |
+
[rank0]: return func(*args, **kwargs)
|
| 937 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^
|
| 938 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 939 |
+
[rank0]: backbone_outputs = self.backbone(backbone_inputs)
|
| 940 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 941 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 942 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 943 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 944 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 945 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 946 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 947 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 948 |
+
[rank0]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 949 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 950 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 951 |
+
[rank0]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 952 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 953 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 954 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 955 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 956 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 957 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 958 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 959 |
+
[rank0]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 960 |
+
[rank0]: outputs = self.language_model(
|
| 961 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 962 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 963 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 964 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 965 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 966 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 967 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 968 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 969 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 970 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 971 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 972 |
+
[rank0]: return func(*args, **kwargs)
|
| 973 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^
|
| 974 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 975 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 976 |
+
[rank0]: ^^^^^^^^^^^
|
| 977 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 978 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 979 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 980 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 981 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 982 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 983 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 984 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 985 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 986 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 987 |
+
[rank0]: layer_outputs = decoder_layer(
|
| 988 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 989 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 990 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 991 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 992 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 993 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 994 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 995 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
|
| 996 |
+
[rank0]: hidden_states = self.mlp(hidden_states)
|
| 997 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 998 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 999 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 1000 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1001 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 1002 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 1003 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1004 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
|
| 1005 |
+
[rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 1006 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1007 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 1008 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 1009 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1010 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 1011 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 1012 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1013 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/activation.py", line 432, in forward
|
| 1014 |
+
[rank0]: return F.silu(input, inplace=self.inplace)
|
| 1015 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1016 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/functional.py", line 2380, in silu
|
| 1017 |
+
[rank0]: return torch._C._nn.silu(input)
|
| 1018 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1019 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 469.94 MiB is free. Including non-PyTorch memory, this process has 78.77 GiB memory in use. Of the allocated memory 76.81 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 1020 |
+
[rank0]:[W615 14:13:02.012395307 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
|
| 1021 |
+
W0615 14:13:02.646000 3538434 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 3544058 closing signal SIGTERM
|
| 1022 |
+
E0615 14:13:03.061000 3538434 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 3544064) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
|
| 1023 |
+
Traceback (most recent call last):
|
| 1024 |
+
File "<frozen runpy>", line 198, in _run_module_as_main
|
| 1025 |
+
File "<frozen runpy>", line 88, in _run_code
|
| 1026 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
|
| 1027 |
+
main()
|
| 1028 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
|
| 1029 |
+
return f(*args, **kwargs)
|
| 1030 |
+
^^^^^^^^^^^^^^^^^^
|
| 1031 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
|
| 1032 |
+
run(args)
|
| 1033 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
|
| 1034 |
+
elastic_launch(
|
| 1035 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
|
| 1036 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 1037 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1038 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
|
| 1039 |
+
raise ChildFailedError(
|
| 1040 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 1041 |
+
============================================================
|
| 1042 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
|
| 1043 |
+
------------------------------------------------------------
|
| 1044 |
+
Failures:
|
| 1045 |
+
<NO_OTHER_FAILURES>
|
| 1046 |
+
------------------------------------------------------------
|
| 1047 |
+
Root Cause (first observed failure):
|
| 1048 |
+
[0]:
|
| 1049 |
+
time : 2026-06-15_14:13:02
|
| 1050 |
+
host : worker1
|
| 1051 |
+
rank : 1 (local_rank: 1)
|
| 1052 |
+
exitcode : 1 (pid: 3544064)
|
| 1053 |
+
error_file: <N/A>
|
| 1054 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 1055 |
+
============================================================
|
| 1056 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 1057 |
+
|
| 1058 |
+
==================================================
|
| 1059 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 1060 |
+
==================================================
|
| 1061 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 1062 |
+
dataset_soup: None
|
| 1063 |
+
output_dir: /tmp/gr00t
|
| 1064 |
+
output_root: None
|
| 1065 |
+
data_config: panda_omron
|
| 1066 |
+
batch_size: 64
|
| 1067 |
+
max_steps: 300000
|
| 1068 |
+
num_gpus: 2
|
| 1069 |
+
save_steps: 20000
|
| 1070 |
+
run_name: None
|
| 1071 |
+
save_total_limit: 100
|
| 1072 |
+
seed: 42
|
| 1073 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 1074 |
+
tune_llm: False
|
| 1075 |
+
tune_visual: False
|
| 1076 |
+
tune_projector: True
|
| 1077 |
+
tune_diffusion_model: True
|
| 1078 |
+
resume: False
|
| 1079 |
+
learning_rate: 3e-05
|
| 1080 |
+
weight_decay: 1e-05
|
| 1081 |
+
warmup_ratio: 0.05
|
| 1082 |
+
lora_rank: 0
|
| 1083 |
+
lora_alpha: 16
|
| 1084 |
+
lora_dropout: 0.1
|
| 1085 |
+
lora_full_model: False
|
| 1086 |
+
dataloader_num_workers: 8
|
| 1087 |
+
report_to: wandb
|
| 1088 |
+
embodiment_tag: new_embodiment
|
| 1089 |
+
video_backend: opencv
|
| 1090 |
+
balance_dataset_weights: True
|
| 1091 |
+
balance_trajectory_weights: True
|
| 1092 |
+
ds_weights_alpha: 0.4
|
| 1093 |
+
==================================================
|
| 1094 |
+
|
| 1095 |
+
Using 2 GPUs
|
| 1096 |
+
Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '64', '--num-gpus', '2']
|
VLM_Only_v2/RKD_A_VLM_Only/default/train_bs64_gpu2_3_lmheadfreeze.log
ADDED
|
@@ -0,0 +1,1087 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 2 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 3 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 4 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 5 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 6 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 7 |
+
check_for_updates()
|
| 8 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 9 |
+
|
| 10 |
+
*****************************************
|
| 11 |
+
Setting OMP_NUM_THREADS environment variable for each process to be 1 in default, to avoid your system being overloaded, please further tune the variable for optimal performance in your application as needed.
|
| 12 |
+
*****************************************
|
| 13 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 14 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 15 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 16 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 17 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 18 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 19 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 20 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 21 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 22 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 23 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 24 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 25 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 26 |
+
check_for_updates()
|
| 27 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 28 |
+
check_for_updates()
|
| 29 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 30 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 31 |
+
|
| 32 |
+
==================================================
|
| 33 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 34 |
+
==================================================
|
| 35 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 36 |
+
dataset_soup: None
|
| 37 |
+
output_dir: /tmp/gr00t
|
| 38 |
+
output_root: None
|
| 39 |
+
data_config: panda_omron
|
| 40 |
+
batch_size: 64
|
| 41 |
+
max_steps: 300000
|
| 42 |
+
num_gpus: 2
|
| 43 |
+
save_steps: 20000
|
| 44 |
+
run_name: None
|
| 45 |
+
save_total_limit: 100
|
| 46 |
+
seed: 42
|
| 47 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 48 |
+
tune_llm: False
|
| 49 |
+
tune_visual: False
|
| 50 |
+
tune_projector: True
|
| 51 |
+
tune_diffusion_model: True
|
| 52 |
+
resume: False
|
| 53 |
+
learning_rate: 3e-05
|
| 54 |
+
weight_decay: 1e-05
|
| 55 |
+
warmup_ratio: 0.05
|
| 56 |
+
lora_rank: 0
|
| 57 |
+
lora_alpha: 16
|
| 58 |
+
lora_dropout: 0.1
|
| 59 |
+
lora_full_model: False
|
| 60 |
+
dataloader_num_workers: 8
|
| 61 |
+
report_to: wandb
|
| 62 |
+
embodiment_tag: new_embodiment
|
| 63 |
+
video_backend: opencv
|
| 64 |
+
balance_dataset_weights: True
|
| 65 |
+
balance_trajectory_weights: True
|
| 66 |
+
ds_weights_alpha: 0.4
|
| 67 |
+
==================================================
|
| 68 |
+
|
| 69 |
+
Using 2 GPUs
|
| 70 |
+
|
| 71 |
+
================================================================================
|
| 72 |
+
Starting sweep branch: default
|
| 73 |
+
Sweep vars: {}
|
| 74 |
+
================================================================================
|
| 75 |
+
|
| 76 |
+
--------------------------------------------------------------------------------
|
| 77 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 78 |
+
Policy type: groot_rkd_v2_raw
|
| 79 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 80 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 81 |
+
Trainable preset: freeze_processing_line
|
| 82 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 83 |
+
--------------------------------------------------------------------------------
|
| 84 |
+
|
| 85 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 86 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 87 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 88 |
+
self.statistics[key] = torch.tensor(value)
|
| 89 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 90 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 91 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 92 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 93 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 94 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 95 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 96 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 97 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 98 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 99 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 100 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 101 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 102 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 103 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 104 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 105 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 106 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 107 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 108 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 109 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 110 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 111 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 112 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 113 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 114 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 115 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 116 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 117 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 118 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 119 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 120 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 121 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 122 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 123 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 124 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 125 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 126 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 127 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 128 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 129 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 130 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 131 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 132 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 133 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 134 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 135 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 136 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 137 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 138 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 139 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 140 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 141 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 142 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 143 |
+
0.75517122 0.7973985 ]
|
| 144 |
+
Loaded 26 datasets
|
| 145 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 146 |
+
Tune backbone vision tower: True
|
| 147 |
+
Tune backbone LLM: True
|
| 148 |
+
Tune action head projector: False
|
| 149 |
+
Tune action head DiT: False
|
| 150 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 151 |
+
|
| 152 |
+
==================================================
|
| 153 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 154 |
+
==================================================
|
| 155 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 156 |
+
dataset_soup: None
|
| 157 |
+
output_dir: /tmp/gr00t
|
| 158 |
+
output_root: None
|
| 159 |
+
data_config: panda_omron
|
| 160 |
+
batch_size: 64
|
| 161 |
+
max_steps: 300000
|
| 162 |
+
num_gpus: 2
|
| 163 |
+
save_steps: 20000
|
| 164 |
+
run_name: None
|
| 165 |
+
save_total_limit: 100
|
| 166 |
+
seed: 42
|
| 167 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 168 |
+
tune_llm: False
|
| 169 |
+
tune_visual: False
|
| 170 |
+
tune_projector: True
|
| 171 |
+
tune_diffusion_model: True
|
| 172 |
+
resume: False
|
| 173 |
+
learning_rate: 3e-05
|
| 174 |
+
weight_decay: 1e-05
|
| 175 |
+
warmup_ratio: 0.05
|
| 176 |
+
lora_rank: 0
|
| 177 |
+
lora_alpha: 16
|
| 178 |
+
lora_dropout: 0.1
|
| 179 |
+
lora_full_model: False
|
| 180 |
+
dataloader_num_workers: 8
|
| 181 |
+
report_to: wandb
|
| 182 |
+
embodiment_tag: new_embodiment
|
| 183 |
+
video_backend: opencv
|
| 184 |
+
balance_dataset_weights: True
|
| 185 |
+
balance_trajectory_weights: True
|
| 186 |
+
ds_weights_alpha: 0.4
|
| 187 |
+
==================================================
|
| 188 |
+
|
| 189 |
+
Using 2 GPUs
|
| 190 |
+
|
| 191 |
+
================================================================================
|
| 192 |
+
Starting sweep branch: default
|
| 193 |
+
Sweep vars: {}
|
| 194 |
+
================================================================================
|
| 195 |
+
|
| 196 |
+
--------------------------------------------------------------------------------
|
| 197 |
+
Running phase 1: phase2_rkd_a_vlm_only
|
| 198 |
+
Policy type: groot_rkd_v2_raw
|
| 199 |
+
Base model path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 200 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2
|
| 201 |
+
Trainable preset: freeze_processing_line
|
| 202 |
+
Policy overrides: {'rkd_enabled': True, 'rkd_fm_loss_weight': 0.0, 'rkd_loss_weight': 1.0, 'rkd_relation_mode': 'token_pair_mean', 'rkd_loss_type': 'angle', 'rkd_exclude_diagonal': True}
|
| 203 |
+
--------------------------------------------------------------------------------
|
| 204 |
+
|
| 205 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]
|
| 206 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 207 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 208 |
+
self.statistics[key] = torch.tensor(value)
|
| 209 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 210 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 211 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 212 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 213 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 214 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 215 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 216 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 217 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 218 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 219 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 220 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 221 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 222 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 223 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 224 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 225 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 226 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 227 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 228 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 229 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 230 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 231 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 232 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 233 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 234 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 235 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 236 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 237 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 238 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 239 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 240 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 241 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 242 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 243 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 244 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 245 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 246 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 247 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 248 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 249 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 250 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 251 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 252 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 253 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 254 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 255 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 256 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 257 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 258 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 259 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 260 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 261 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 262 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 263 |
+
0.75517122 0.7973985 ]
|
| 264 |
+
Loaded 26 datasets
|
| 265 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 266 |
+
Tune backbone vision tower: True
|
| 267 |
+
Tune backbone LLM: True
|
| 268 |
+
Tune action head projector: False
|
| 269 |
+
Tune action head DiT: False
|
| 270 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 271 |
+
Tune backbone llm: False
|
| 272 |
+
Tune backbone visual: True
|
| 273 |
+
Total number of DiT parameters: 550386688
|
| 274 |
+
Tune backbone llm: False
|
| 275 |
+
Tune backbone visual: True
|
| 276 |
+
Total number of DiT parameters: 550386688
|
| 277 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 278 |
+
Tune action head projector: True
|
| 279 |
+
Tune action head diffusion model: True
|
| 280 |
+
|
| 281 |
+
Tune backbone llm: True
|
| 282 |
+
Tune backbone visual: True
|
| 283 |
+
Tune action head projector: False
|
| 284 |
+
Tune action head diffusion model: False
|
| 285 |
+
Action head trainable parameter: future_tokens.weight
|
| 286 |
+
Action head trainable parameter: vlln.weight
|
| 287 |
+
Action head trainable parameter: vlln.bias
|
| 288 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 289 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 290 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 291 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 292 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 293 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 294 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 295 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 296 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 297 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 298 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 299 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 300 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 301 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 302 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 303 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 304 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 305 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 306 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 307 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 308 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 309 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 310 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 311 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 312 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 313 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 314 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 315 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 316 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 317 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 318 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 319 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 320 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 321 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 322 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 323 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 324 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 325 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 326 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 327 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 328 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 329 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 330 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 331 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 332 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 333 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 334 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 335 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 336 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 337 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 338 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 339 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 340 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 341 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 342 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 343 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 344 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 345 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 346 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 347 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 348 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 349 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 350 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 351 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 352 |
+
Applied trainable preset: freeze_processing_line
|
| 353 |
+
Trainable parameter tensors after preset: 584
|
| 354 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 355 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 356 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 357 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 358 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 359 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 360 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 361 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 362 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 363 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 364 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 365 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 366 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 367 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 368 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 369 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 370 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 371 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 372 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 373 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 374 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 375 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 376 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 377 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 378 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 379 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 380 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 381 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 382 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 383 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 384 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 385 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 386 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 387 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 388 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 389 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 390 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 391 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 392 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 393 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 394 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 395 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 396 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 397 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 398 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 399 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 400 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 401 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 402 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 403 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 404 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 405 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 406 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 407 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 408 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 409 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 410 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 411 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 412 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 413 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 414 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 415 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 416 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 417 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 418 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 419 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 420 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 421 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 422 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 423 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 424 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 425 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 426 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 427 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 428 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 429 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 430 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 431 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 432 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 433 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 434 |
+
... 504 more
|
| 435 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
|
| 436 |
+
Run name: default_phase2
|
| 437 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 438 |
+
Tune action head projector: True
|
| 439 |
+
Tune action head diffusion model: True
|
| 440 |
+
|
| 441 |
+
Tune backbone llm: True
|
| 442 |
+
Tune backbone visual: True
|
| 443 |
+
Tune action head projector: False
|
| 444 |
+
Tune action head diffusion model: False
|
| 445 |
+
Action head trainable parameter: future_tokens.weight
|
| 446 |
+
Action head trainable parameter: vlln.weight
|
| 447 |
+
Action head trainable parameter: vlln.bias
|
| 448 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.weight
|
| 449 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm1.bias
|
| 450 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.weight
|
| 451 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_q.bias
|
| 452 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.weight
|
| 453 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_k.bias
|
| 454 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.weight
|
| 455 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_v.bias
|
| 456 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.weight
|
| 457 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.attn1.to_out.0.bias
|
| 458 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.weight
|
| 459 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.norm3.bias
|
| 460 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.weight
|
| 461 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.0.proj.bias
|
| 462 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.weight
|
| 463 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.0.ff.net.2.bias
|
| 464 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.weight
|
| 465 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm1.bias
|
| 466 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.weight
|
| 467 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_q.bias
|
| 468 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.weight
|
| 469 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_k.bias
|
| 470 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.weight
|
| 471 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_v.bias
|
| 472 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.weight
|
| 473 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.attn1.to_out.0.bias
|
| 474 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.weight
|
| 475 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.norm3.bias
|
| 476 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.weight
|
| 477 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.0.proj.bias
|
| 478 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.weight
|
| 479 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.1.ff.net.2.bias
|
| 480 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.weight
|
| 481 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm1.bias
|
| 482 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.weight
|
| 483 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_q.bias
|
| 484 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.weight
|
| 485 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_k.bias
|
| 486 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.weight
|
| 487 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_v.bias
|
| 488 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.weight
|
| 489 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.attn1.to_out.0.bias
|
| 490 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.weight
|
| 491 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.norm3.bias
|
| 492 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.weight
|
| 493 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.0.proj.bias
|
| 494 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.weight
|
| 495 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.2.ff.net.2.bias
|
| 496 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.weight
|
| 497 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm1.bias
|
| 498 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.weight
|
| 499 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_q.bias
|
| 500 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.weight
|
| 501 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_k.bias
|
| 502 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.weight
|
| 503 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_v.bias
|
| 504 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.weight
|
| 505 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.attn1.to_out.0.bias
|
| 506 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.weight
|
| 507 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.norm3.bias
|
| 508 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.weight
|
| 509 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.0.proj.bias
|
| 510 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.weight
|
| 511 |
+
Action head trainable parameter: vl_self_attention.transformer_blocks.3.ff.net.2.bias
|
| 512 |
+
Applied trainable preset: freeze_processing_line
|
| 513 |
+
Trainable parameter tensors after preset: 584
|
| 514 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.weight
|
| 515 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.patch_embedding.bias
|
| 516 |
+
trainable: backbone.eagle_model.vision_model.vision_model.embeddings.position_embedding.weight
|
| 517 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.weight
|
| 518 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm1.bias
|
| 519 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.weight
|
| 520 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.k_proj.bias
|
| 521 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.weight
|
| 522 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.v_proj.bias
|
| 523 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.weight
|
| 524 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.q_proj.bias
|
| 525 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.weight
|
| 526 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.self_attn.out_proj.bias
|
| 527 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.weight
|
| 528 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.layer_norm2.bias
|
| 529 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.weight
|
| 530 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc1.bias
|
| 531 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.weight
|
| 532 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.0.mlp.fc2.bias
|
| 533 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.weight
|
| 534 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm1.bias
|
| 535 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.weight
|
| 536 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.k_proj.bias
|
| 537 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.weight
|
| 538 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.v_proj.bias
|
| 539 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.weight
|
| 540 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.q_proj.bias
|
| 541 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.weight
|
| 542 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.self_attn.out_proj.bias
|
| 543 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.weight
|
| 544 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.layer_norm2.bias
|
| 545 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.weight
|
| 546 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc1.bias
|
| 547 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.weight
|
| 548 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.1.mlp.fc2.bias
|
| 549 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.weight
|
| 550 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm1.bias
|
| 551 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.weight
|
| 552 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.k_proj.bias
|
| 553 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.weight
|
| 554 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.v_proj.bias
|
| 555 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.weight
|
| 556 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.q_proj.bias
|
| 557 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.weight
|
| 558 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.self_attn.out_proj.bias
|
| 559 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.weight
|
| 560 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.layer_norm2.bias
|
| 561 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.weight
|
| 562 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc1.bias
|
| 563 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.weight
|
| 564 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.2.mlp.fc2.bias
|
| 565 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.weight
|
| 566 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm1.bias
|
| 567 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.weight
|
| 568 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.k_proj.bias
|
| 569 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.weight
|
| 570 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.v_proj.bias
|
| 571 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.weight
|
| 572 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.q_proj.bias
|
| 573 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.weight
|
| 574 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.self_attn.out_proj.bias
|
| 575 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.weight
|
| 576 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.layer_norm2.bias
|
| 577 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.weight
|
| 578 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc1.bias
|
| 579 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.weight
|
| 580 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.3.mlp.fc2.bias
|
| 581 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.weight
|
| 582 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm1.bias
|
| 583 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.weight
|
| 584 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.k_proj.bias
|
| 585 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.weight
|
| 586 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.v_proj.bias
|
| 587 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.weight
|
| 588 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.q_proj.bias
|
| 589 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.weight
|
| 590 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.self_attn.out_proj.bias
|
| 591 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.weight
|
| 592 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.layer_norm2.bias
|
| 593 |
+
trainable: backbone.eagle_model.vision_model.vision_model.encoder.layers.4.mlp.fc1.weight
|
| 594 |
+
... 504 more
|
| 595 |
+
Trainable summary: {'trainable_preset': 'freeze_processing_line', 'trainable_modules': 'backbone.eagle_model.language_model(914,674,688), backbone.eagle_model.vision_model(427,680,704), backbone.eagle_model.mlp1(2,361,344)', 'trainable_module_names': 'backbone.eagle_model.language_model, backbone.eagle_model.vision_model, backbone.eagle_model.mlp1', 'trainable_module_param_counts': {'backbone.eagle_model.language_model': 914674688, 'backbone.eagle_model.vision_model': 427680704, 'backbone.eagle_model.mlp1': 2361344}, 'trainable_param_count': 1344716736, 'total_param_count': 2724163520, 'trainable_param_ratio': 0.4936255573967895}
|
| 596 |
+
Run name: default_phase2
|
| 597 |
+
train dataloader length: 3437
|
| 598 |
+
train dataset length: 439854
|
| 599 |
+
GPU memory before training: 7.076685905456543 GB
|
| 600 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 601 |
+
train dataloader length: 3437
|
| 602 |
+
train dataset length: 439854
|
| 603 |
+
GPU memory before training: 7.076685905456543 GB
|
| 604 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/VLM_Only_v2/RKD_A_VLM_Only/default/phase2/runs
|
| 605 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
|
| 606 |
+
wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 607 |
+
wandb: setting up run 8wkyy6nk
|
| 608 |
+
wandb: Tracking run with wandb version 0.25.0
|
| 609 |
+
wandb: Run data is saved locally in /home/seonho/wandb/run-20260615_143025-8wkyy6nk
|
| 610 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 611 |
+
wandb: Syncing run default_phase2
|
| 612 |
+
wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 613 |
+
wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/8wkyy6nk
|
| 614 |
+
|
| 615 |
0%| | 0/30000 [00:00<?, ?it/s][rank1]: Traceback (most recent call last):
|
| 616 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
|
| 617 |
+
[rank1]: run_yaml_experiment(
|
| 618 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
|
| 619 |
+
[rank1]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 620 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
|
| 621 |
+
[rank1]: experiment.train()
|
| 622 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 623 |
+
[rank1]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 624 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 625 |
+
[rank1]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 626 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 627 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 628 |
+
[rank1]: return inner_training_loop(
|
| 629 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 630 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 631 |
+
[rank1]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 632 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 633 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 634 |
+
[rank1]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 635 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 636 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 637 |
+
[rank1]: outputs = model(inputs)
|
| 638 |
+
[rank1]: ^^^^^^^^^^^^^
|
| 639 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 640 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 641 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 642 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 643 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 644 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 645 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 646 |
+
[rank1]: else self._run_ddp_forward(*inputs, **kwargs)
|
| 647 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 648 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 649 |
+
[rank1]: return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 650 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 651 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 652 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 653 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 654 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 655 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 656 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 657 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 658 |
+
[rank1]: return model_forward(*args, **kwargs)
|
| 659 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 660 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 661 |
+
[rank1]: return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 662 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 663 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 664 |
+
[rank1]: return func(*args, **kwargs)
|
| 665 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^
|
| 666 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 667 |
+
[rank1]: backbone_outputs = self.backbone(backbone_inputs)
|
| 668 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 669 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 670 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 671 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 672 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 673 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 674 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 675 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 676 |
+
[rank1]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 677 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 678 |
+
[rank1]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 679 |
+
[rank1]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 680 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 681 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 682 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 683 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 684 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 685 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 686 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 687 |
+
[rank1]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 688 |
+
[rank1]: outputs = self.language_model(
|
| 689 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^
|
| 690 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 691 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 692 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 693 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 694 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 695 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 696 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 697 |
+
[rank1]: output = func(self, *args, **kwargs)
|
| 698 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 699 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 700 |
+
[rank1]: return func(*args, **kwargs)
|
| 701 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^
|
| 702 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 703 |
+
[rank1]: outputs: BaseModelOutputWithPast = self.model(
|
| 704 |
+
[rank1]: ^^^^^^^^^^^
|
| 705 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 706 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 707 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 708 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 709 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 710 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 711 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 712 |
+
[rank1]: output = func(self, *args, **kwargs)
|
| 713 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 714 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 715 |
+
[rank1]: layer_outputs = decoder_layer(
|
| 716 |
+
[rank1]: ^^^^^^^^^^^^^^
|
| 717 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 718 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 719 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 720 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 721 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 722 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 723 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
|
| 724 |
+
[rank1]: hidden_states = self.mlp(hidden_states)
|
| 725 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 726 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 727 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 728 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 729 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 730 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 731 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 732 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
|
| 733 |
+
[rank1]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 734 |
+
[rank1]: ^^^^^^^^^^^^^^^
|
| 735 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 736 |
+
[rank1]: return self._call_impl(*args, **kwargs)
|
| 737 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 738 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 739 |
+
[rank1]: return forward_call(*args, **kwargs)
|
| 740 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 741 |
+
[rank1]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
|
| 742 |
+
[rank1]: return F.linear(input, self.weight, self.bias)
|
| 743 |
+
[rank1]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 744 |
+
[rank1]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 1 has a total capacity of 79.25 GiB of which 433.94 MiB is free. Including non-PyTorch memory, this process has 78.80 GiB memory in use. Of the allocated memory 76.84 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 745 |
+
wandb: updating run metadata
|
| 746 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 747 |
+
wandb: uploading wandb-summary.json; uploading config.yaml
|
| 748 |
+
wandb: 🚀 View run default_phase2 at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/8wkyy6nk
|
| 749 |
+
wandb: ⭐️ View project at: https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 750 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 751 |
+
wandb: Find logs at: ./wandb/run-20260615_143025-8wkyy6nk/logs
|
| 752 |
+
Traceback (most recent call last):
|
| 753 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
|
| 754 |
+
run_yaml_experiment(
|
| 755 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
|
| 756 |
+
main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 757 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
|
| 758 |
+
experiment.train()
|
| 759 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 760 |
+
self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 761 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 762 |
+
return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 763 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 764 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 765 |
+
return inner_training_loop(
|
| 766 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 767 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 768 |
+
tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 769 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 770 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 771 |
+
loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 772 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 773 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 774 |
+
outputs = model(inputs)
|
| 775 |
+
^^^^^^^^^^^^^
|
| 776 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 777 |
+
return self._call_impl(*args, **kwargs)
|
| 778 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 779 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 780 |
+
return forward_call(*args, **kwargs)
|
| 781 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 782 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 783 |
+
else self._run_ddp_forward(*inputs, **kwargs)
|
| 784 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 785 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 786 |
+
return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 787 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 788 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 789 |
+
return self._call_impl(*args, **kwargs)
|
| 790 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 791 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 792 |
+
return forward_call(*args, **kwargs)
|
| 793 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 794 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 795 |
+
return model_forward(*args, **kwargs)
|
| 796 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 797 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 798 |
+
return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 799 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 800 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 801 |
+
return func(*args, **kwargs)
|
| 802 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 803 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 804 |
+
backbone_outputs = self.backbone(backbone_inputs)
|
| 805 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 806 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 807 |
+
return self._call_impl(*args, **kwargs)
|
| 808 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 809 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 810 |
+
return forward_call(*args, **kwargs)
|
| 811 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 812 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 813 |
+
eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 814 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 815 |
+
File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 816 |
+
eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 817 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 818 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 819 |
+
return self._call_impl(*args, **kwargs)
|
| 820 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 821 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 822 |
+
return forward_call(*args, **kwargs)
|
| 823 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 824 |
+
File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 825 |
+
outputs = self.language_model(
|
| 826 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 827 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 828 |
+
return self._call_impl(*args, **kwargs)
|
| 829 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 830 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 831 |
+
return forward_call(*args, **kwargs)
|
| 832 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 833 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 834 |
+
output = func(self, *args, **kwargs)
|
| 835 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 836 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 837 |
+
return func(*args, **kwargs)
|
| 838 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 839 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 840 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 841 |
+
^^^^^^^^^^^
|
| 842 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 843 |
+
return self._call_impl(*args, **kwargs)
|
| 844 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 845 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 846 |
+
return forward_call(*args, **kwargs)
|
| 847 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 848 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 849 |
+
output = func(self, *args, **kwargs)
|
| 850 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 851 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 852 |
+
layer_outputs = decoder_layer(
|
| 853 |
+
^^^^^^^^^^^^^^
|
| 854 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 855 |
+
return self._call_impl(*args, **kwargs)
|
| 856 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 857 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 858 |
+
return forward_call(*args, **kwargs)
|
| 859 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 860 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
|
| 861 |
+
hidden_states = self.mlp(hidden_states)
|
| 862 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 863 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 864 |
+
return self._call_impl(*args, **kwargs)
|
| 865 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 866 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 867 |
+
return forward_call(*args, **kwargs)
|
| 868 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 869 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
|
| 870 |
+
down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 871 |
+
^^^^^^^^^^^^^^^
|
| 872 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 873 |
+
return self._call_impl(*args, **kwargs)
|
| 874 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 875 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 876 |
+
return forward_call(*args, **kwargs)
|
| 877 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 878 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
|
| 879 |
+
return F.linear(input, self.weight, self.bias)
|
| 880 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 881 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 433.94 MiB is free. Including non-PyTorch memory, this process has 78.80 GiB memory in use. Of the allocated memory 76.84 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 882 |
+
[rank0]: Traceback (most recent call last):
|
| 883 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 1025, in <module>
|
| 884 |
+
[rank0]: run_yaml_experiment(
|
| 885 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 974, in run_yaml_experiment
|
| 886 |
+
[rank0]: main(args, policy_type=policy_type, policy_overrides=policy_overrides, trainable_preset=preset)
|
| 887 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py", line 708, in main
|
| 888 |
+
[rank0]: experiment.train()
|
| 889 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/runner.py", line 172, in train
|
| 890 |
+
[rank0]: self.trainer.train(resume_from_checkpoint=self.resume_from_checkpoint)
|
| 891 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 176, in train
|
| 892 |
+
[rank0]: return super().train(resume_from_checkpoint, trial, ignore_keys_for_eval, **kwargs)
|
| 893 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 894 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2245, in train
|
| 895 |
+
[rank0]: return inner_training_loop(
|
| 896 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 897 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 2560, in _inner_training_loop
|
| 898 |
+
[rank0]: tr_loss_step = self.training_step(model, inputs, num_items_in_batch)
|
| 899 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 900 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/trainer.py", line 3736, in training_step
|
| 901 |
+
[rank0]: loss = self.compute_loss(model, inputs, num_items_in_batch=num_items_in_batch)
|
| 902 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 903 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/experiment/trainer.py", line 77, in compute_loss
|
| 904 |
+
[rank0]: outputs = model(inputs)
|
| 905 |
+
[rank0]: ^^^^^^^^^^^^^
|
| 906 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 907 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 908 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 909 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 910 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 911 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 912 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1637, in forward
|
| 913 |
+
[rank0]: else self._run_ddp_forward(*inputs, **kwargs)
|
| 914 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 915 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/parallel/distributed.py", line 1464, in _run_ddp_forward
|
| 916 |
+
[rank0]: return self.module(*inputs, **kwargs) # type: ignore[index]
|
| 917 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 918 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 919 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 920 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 921 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 922 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 923 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 924 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 823, in forward
|
| 925 |
+
[rank0]: return model_forward(*args, **kwargs)
|
| 926 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 927 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/accelerate/utils/operations.py", line 811, in __call__
|
| 928 |
+
[rank0]: return convert_to_fp32(self.model_forward(*args, **kwargs))
|
| 929 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 930 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/amp/autocast_mode.py", line 44, in decorate_autocast
|
| 931 |
+
[rank0]: return func(*args, **kwargs)
|
| 932 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^
|
| 933 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/gr00t_n1.py", line 451, in forward
|
| 934 |
+
[rank0]: backbone_outputs = self.backbone(backbone_inputs)
|
| 935 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 936 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 937 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 938 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 939 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 940 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 941 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 942 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 118, in forward
|
| 943 |
+
[rank0]: eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
|
| 944 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 945 |
+
[rank0]: File "/home/seonho/clvla/benchmarks/robocasa_v2/my_policies/groot_rkd_v2_raw/backbone/eagle_backbone.py", line 109, in forward_eagle
|
| 946 |
+
[rank0]: eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
|
| 947 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 948 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 949 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 950 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 951 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 952 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 953 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 954 |
+
[rank0]: File "/home/seonho/.cache/huggingface/modules/transformers_modules/eagle2_hg_model/modeling_eagle2_5_vl.py", line 261, in forward
|
| 955 |
+
[rank0]: outputs = self.language_model(
|
| 956 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 957 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 958 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 959 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 960 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 961 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 962 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 963 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 964 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 965 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 966 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/deprecation.py", line 172, in wrapped_func
|
| 967 |
+
[rank0]: return func(*args, **kwargs)
|
| 968 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^
|
| 969 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 850, in forward
|
| 970 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 971 |
+
[rank0]: ^^^^^^^^^^^
|
| 972 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 973 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 974 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 975 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 976 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 977 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 978 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/utils/generic.py", line 965, in wrapper
|
| 979 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 980 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 981 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 576, in forward
|
| 982 |
+
[rank0]: layer_outputs = decoder_layer(
|
| 983 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 984 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 985 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 986 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 987 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 988 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 989 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 990 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 305, in forward
|
| 991 |
+
[rank0]: hidden_states = self.mlp(hidden_states)
|
| 992 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 993 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 994 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 995 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 996 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 997 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 998 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 999 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 94, in forward
|
| 1000 |
+
[rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 1001 |
+
[rank0]: ^^^^^^^^^^^^^^^
|
| 1002 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
|
| 1003 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 1004 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1005 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
|
| 1006 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 1007 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1008 |
+
[rank0]: File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/nn/modules/linear.py", line 125, in forward
|
| 1009 |
+
[rank0]: return F.linear(input, self.weight, self.bias)
|
| 1010 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1011 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 628.00 MiB. GPU 0 has a total capacity of 79.25 GiB of which 433.94 MiB is free. Including non-PyTorch memory, this process has 78.80 GiB memory in use. Of the allocated memory 76.84 GiB is allocated by PyTorch, and 1.33 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
|
| 1012 |
+
[rank0]:[W615 14:30:41.759054055 ProcessGroupNCCL.cpp:1479] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
|
| 1013 |
+
W0615 14:30:42.535000 3405061 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:900] Sending process 3429266 closing signal SIGTERM
|
| 1014 |
+
E0615 14:30:42.850000 3405061 miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/api.py:874] failed (exitcode: 1) local_rank: 1 (pid: 3429267) of binary: /home/seonho/miniconda3/envs/robocasa/bin/python
|
| 1015 |
+
Traceback (most recent call last):
|
| 1016 |
+
File "<frozen runpy>", line 198, in _run_module_as_main
|
| 1017 |
+
File "<frozen runpy>", line 88, in _run_code
|
| 1018 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 896, in <module>
|
| 1019 |
+
main()
|
| 1020 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 355, in wrapper
|
| 1021 |
+
return f(*args, **kwargs)
|
| 1022 |
+
^^^^^^^^^^^^^^^^^^
|
| 1023 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 892, in main
|
| 1024 |
+
run(args)
|
| 1025 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/run.py", line 883, in run
|
| 1026 |
+
elastic_launch(
|
| 1027 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 139, in __call__
|
| 1028 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 1029 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 1030 |
+
File "/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/torch/distributed/launcher/api.py", line 270, in launch_agent
|
| 1031 |
+
raise ChildFailedError(
|
| 1032 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 1033 |
+
============================================================
|
| 1034 |
+
/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py FAILED
|
| 1035 |
+
------------------------------------------------------------
|
| 1036 |
+
Failures:
|
| 1037 |
+
<NO_OTHER_FAILURES>
|
| 1038 |
+
------------------------------------------------------------
|
| 1039 |
+
Root Cause (first observed failure):
|
| 1040 |
+
[0]:
|
| 1041 |
+
time : 2026-06-15_14:30:42
|
| 1042 |
+
host : worker1
|
| 1043 |
+
rank : 1 (local_rank: 1)
|
| 1044 |
+
exitcode : 1 (pid: 3429267)
|
| 1045 |
+
error_file: <N/A>
|
| 1046 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 1047 |
+
============================================================
|
| 1048 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 1049 |
+
|
| 1050 |
+
==================================================
|
| 1051 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 1052 |
+
==================================================
|
| 1053 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml
|
| 1054 |
+
dataset_soup: None
|
| 1055 |
+
output_dir: /tmp/gr00t
|
| 1056 |
+
output_root: None
|
| 1057 |
+
data_config: panda_omron
|
| 1058 |
+
batch_size: 64
|
| 1059 |
+
max_steps: 300000
|
| 1060 |
+
num_gpus: 2
|
| 1061 |
+
save_steps: 20000
|
| 1062 |
+
run_name: None
|
| 1063 |
+
save_total_limit: 100
|
| 1064 |
+
seed: 42
|
| 1065 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 1066 |
+
tune_llm: False
|
| 1067 |
+
tune_visual: False
|
| 1068 |
+
tune_projector: True
|
| 1069 |
+
tune_diffusion_model: True
|
| 1070 |
+
resume: False
|
| 1071 |
+
learning_rate: 3e-05
|
| 1072 |
+
weight_decay: 1e-05
|
| 1073 |
+
warmup_ratio: 0.05
|
| 1074 |
+
lora_rank: 0
|
| 1075 |
+
lora_alpha: 16
|
| 1076 |
+
lora_dropout: 0.1
|
| 1077 |
+
lora_full_model: False
|
| 1078 |
+
dataloader_num_workers: 8
|
| 1079 |
+
report_to: wandb
|
| 1080 |
+
embodiment_tag: new_embodiment
|
| 1081 |
+
video_backend: opencv
|
| 1082 |
+
balance_dataset_weights: True
|
| 1083 |
+
balance_trajectory_weights: True
|
| 1084 |
+
ds_weights_alpha: 0.4
|
| 1085 |
+
==================================================
|
| 1086 |
+
|
| 1087 |
+
Using 2 GPUs
|
| 1088 |
+
Running torchrun command: ['/home/seonho/miniconda3/envs/robocasa/bin/python', '-m', 'torch.distributed.run', '--standalone', '--nproc_per_node=2', '--nnodes=1', '/home/seonho/clvla/benchmarks/robocasa_v2/my_scripts/gr00t_finetune.py', '--config', '/home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/VLM_only_v2/rkd_v2_a_vlm_only.yaml', '--batch-size', '64', '--num-gpus', '2']
|
experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p1_train.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p2_train.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
experiment_cfg/processing_line_only_v2/MGD_v2/train.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
experiment_cfg/processing_line_only_v2/MGD_v2/train_mse.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/trainer_state.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/trainer_state.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_cosine_mse.yaml
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_loss_cosine_mse
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.3
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase2_mgd_only
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
preset: processing_line_only
|
| 21 |
+
tune_llm: false
|
| 22 |
+
tune_visual: false
|
| 23 |
+
tune_projector: false
|
| 24 |
+
tune_diffusion_model: false
|
| 25 |
+
losses:
|
| 26 |
+
mgd_enabled: true
|
| 27 |
+
mgd_fm_loss_weight: 0.0
|
| 28 |
+
mgd_loss_weight: 1.0
|
| 29 |
+
mgd_sequence_hidden_dim: 512
|
| 30 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 31 |
+
mgd_use_cosine_loss: true
|
| 32 |
+
mgd_use_mse_loss: true
|
| 33 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 34 |
+
max_steps: 30000
|
| 35 |
+
save_steps: 0
|
| 36 |
+
trainable:
|
| 37 |
+
tune_llm: false
|
| 38 |
+
tune_visual: false
|
| 39 |
+
tune_projector: true
|
| 40 |
+
tune_diffusion_model: true
|
| 41 |
+
losses:
|
| 42 |
+
mgd_enabled: true
|
| 43 |
+
mgd_fm_loss_weight: 1.0
|
| 44 |
+
mgd_loss_weight: 0.5
|
| 45 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 46 |
+
mgd_use_cosine_loss: true
|
| 47 |
+
mgd_use_mse_loss: true
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_cosine_mse_phase3_only.yaml
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_loss_cosine_mse_phase3
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.3
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
tune_llm: false
|
| 21 |
+
tune_visual: false
|
| 22 |
+
tune_projector: true
|
| 23 |
+
tune_diffusion_model: true
|
| 24 |
+
losses:
|
| 25 |
+
mgd_enabled: true
|
| 26 |
+
mgd_fm_loss_weight: 1.0
|
| 27 |
+
mgd_loss_weight: 0.5
|
| 28 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 29 |
+
mgd_use_cosine_loss: true
|
| 30 |
+
mgd_use_mse_loss: true
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/config.json
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"action_dim": 32,
|
| 3 |
+
"action_head_cfg": {
|
| 4 |
+
"action_dim": 32,
|
| 5 |
+
"action_horizon": 16,
|
| 6 |
+
"add_pos_embed": true,
|
| 7 |
+
"backbone_embedding_dim": 2048,
|
| 8 |
+
"diffusion_model_cfg": {
|
| 9 |
+
"attention_head_dim": 48,
|
| 10 |
+
"cross_attention_dim": 2048,
|
| 11 |
+
"dropout": 0.2,
|
| 12 |
+
"final_dropout": true,
|
| 13 |
+
"interleave_self_attention": true,
|
| 14 |
+
"norm_type": "ada_norm",
|
| 15 |
+
"num_attention_heads": 32,
|
| 16 |
+
"num_layers": 16,
|
| 17 |
+
"output_dim": 1024,
|
| 18 |
+
"positional_embeddings": null
|
| 19 |
+
},
|
| 20 |
+
"hidden_size": 1024,
|
| 21 |
+
"input_embedding_dim": 1536,
|
| 22 |
+
"max_action_dim": 32,
|
| 23 |
+
"max_state_dim": 64,
|
| 24 |
+
"model_dtype": "float32",
|
| 25 |
+
"noise_beta_alpha": 1.5,
|
| 26 |
+
"noise_beta_beta": 1.0,
|
| 27 |
+
"noise_s": 0.999,
|
| 28 |
+
"num_inference_timesteps": 4,
|
| 29 |
+
"num_target_vision_tokens": 32,
|
| 30 |
+
"num_timestep_buckets": 1000,
|
| 31 |
+
"tune_diffusion_model": true,
|
| 32 |
+
"tune_projector": true,
|
| 33 |
+
"use_vlln": true,
|
| 34 |
+
"vl_self_attention_cfg": {
|
| 35 |
+
"attention_head_dim": 64,
|
| 36 |
+
"dropout": 0.2,
|
| 37 |
+
"final_dropout": true,
|
| 38 |
+
"num_attention_heads": 32,
|
| 39 |
+
"num_layers": 4,
|
| 40 |
+
"positional_embeddings": null
|
| 41 |
+
}
|
| 42 |
+
},
|
| 43 |
+
"action_horizon": 16,
|
| 44 |
+
"architectures": [
|
| 45 |
+
"GR00T_N1_5_MGD"
|
| 46 |
+
],
|
| 47 |
+
"attn_implementation": null,
|
| 48 |
+
"backbone_cfg": {
|
| 49 |
+
"eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
|
| 50 |
+
"load_bf16": false,
|
| 51 |
+
"project_to_dim": null,
|
| 52 |
+
"reproject_vision": false,
|
| 53 |
+
"select_layer": 12,
|
| 54 |
+
"tune_llm": false,
|
| 55 |
+
"tune_visual": true,
|
| 56 |
+
"use_flash_attention": true
|
| 57 |
+
},
|
| 58 |
+
"compute_dtype": "bfloat16",
|
| 59 |
+
"hidden_size": 2048,
|
| 60 |
+
"mgd_enabled": true,
|
| 61 |
+
"mgd_fm_loss_weight": 0.0,
|
| 62 |
+
"mgd_loss_weight": 1.0,
|
| 63 |
+
"mgd_loss_weight_end": 0.0,
|
| 64 |
+
"mgd_loss_weight_schedule": null,
|
| 65 |
+
"mgd_loss_weight_start": 0.05,
|
| 66 |
+
"mgd_pretrained_projector_path": null,
|
| 67 |
+
"mgd_sequence_hidden_dim": 512,
|
| 68 |
+
"mgd_target_dim": 512,
|
| 69 |
+
"mgd_target_pooling": "flatten",
|
| 70 |
+
"mgd_target_projection": "frozen_random",
|
| 71 |
+
"mgd_token_mask_ratio": 0.3,
|
| 72 |
+
"mgd_use_cosine_loss": true,
|
| 73 |
+
"mgd_use_mse_loss": true,
|
| 74 |
+
"model_dtype": "float32",
|
| 75 |
+
"model_type": "gr00t_n1_5",
|
| 76 |
+
"torch_dtype": "bfloat16",
|
| 77 |
+
"transformers_version": "4.51.3"
|
| 78 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/trainer_state.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/config.json
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"action_dim": 32,
|
| 3 |
+
"action_head_cfg": {
|
| 4 |
+
"action_dim": 32,
|
| 5 |
+
"action_horizon": 16,
|
| 6 |
+
"add_pos_embed": true,
|
| 7 |
+
"backbone_embedding_dim": 2048,
|
| 8 |
+
"diffusion_model_cfg": {
|
| 9 |
+
"attention_head_dim": 48,
|
| 10 |
+
"cross_attention_dim": 2048,
|
| 11 |
+
"dropout": 0.2,
|
| 12 |
+
"final_dropout": true,
|
| 13 |
+
"interleave_self_attention": true,
|
| 14 |
+
"norm_type": "ada_norm",
|
| 15 |
+
"num_attention_heads": 32,
|
| 16 |
+
"num_layers": 16,
|
| 17 |
+
"output_dim": 1024,
|
| 18 |
+
"positional_embeddings": null
|
| 19 |
+
},
|
| 20 |
+
"hidden_size": 1024,
|
| 21 |
+
"input_embedding_dim": 1536,
|
| 22 |
+
"max_action_dim": 32,
|
| 23 |
+
"max_state_dim": 64,
|
| 24 |
+
"model_dtype": "float32",
|
| 25 |
+
"noise_beta_alpha": 1.5,
|
| 26 |
+
"noise_beta_beta": 1.0,
|
| 27 |
+
"noise_s": 0.999,
|
| 28 |
+
"num_inference_timesteps": 4,
|
| 29 |
+
"num_target_vision_tokens": 32,
|
| 30 |
+
"num_timestep_buckets": 1000,
|
| 31 |
+
"tune_diffusion_model": true,
|
| 32 |
+
"tune_projector": true,
|
| 33 |
+
"use_vlln": true,
|
| 34 |
+
"vl_self_attention_cfg": {
|
| 35 |
+
"attention_head_dim": 64,
|
| 36 |
+
"dropout": 0.2,
|
| 37 |
+
"final_dropout": true,
|
| 38 |
+
"num_attention_heads": 32,
|
| 39 |
+
"num_layers": 4,
|
| 40 |
+
"positional_embeddings": null
|
| 41 |
+
}
|
| 42 |
+
},
|
| 43 |
+
"action_horizon": 16,
|
| 44 |
+
"architectures": [
|
| 45 |
+
"GR00T_N1_5_MGD"
|
| 46 |
+
],
|
| 47 |
+
"attn_implementation": null,
|
| 48 |
+
"backbone_cfg": {
|
| 49 |
+
"eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
|
| 50 |
+
"load_bf16": false,
|
| 51 |
+
"project_to_dim": null,
|
| 52 |
+
"reproject_vision": false,
|
| 53 |
+
"select_layer": 12,
|
| 54 |
+
"tune_llm": false,
|
| 55 |
+
"tune_visual": true,
|
| 56 |
+
"use_flash_attention": true
|
| 57 |
+
},
|
| 58 |
+
"compute_dtype": "bfloat16",
|
| 59 |
+
"hidden_size": 2048,
|
| 60 |
+
"mgd_enabled": true,
|
| 61 |
+
"mgd_fm_loss_weight": 1.0,
|
| 62 |
+
"mgd_loss_weight": 0.5,
|
| 63 |
+
"mgd_loss_weight_end": 0.0,
|
| 64 |
+
"mgd_loss_weight_schedule": null,
|
| 65 |
+
"mgd_loss_weight_start": 0.05,
|
| 66 |
+
"mgd_pretrained_projector_path": null,
|
| 67 |
+
"mgd_sequence_hidden_dim": 512,
|
| 68 |
+
"mgd_target_dim": 512,
|
| 69 |
+
"mgd_target_pooling": "flatten",
|
| 70 |
+
"mgd_target_projection": "frozen_random",
|
| 71 |
+
"mgd_token_mask_ratio": 0.3,
|
| 72 |
+
"mgd_use_cosine_loss": true,
|
| 73 |
+
"mgd_use_mse_loss": true,
|
| 74 |
+
"model_dtype": "float32",
|
| 75 |
+
"model_type": "gr00t_n1_5",
|
| 76 |
+
"torch_dtype": "bfloat16",
|
| 77 |
+
"transformers_version": "4.51.3"
|
| 78 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/trainer_state.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3_only_train.log
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/30000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
|
|
|
|
| 1 |
0%| | 1/30000 [00:12<105:19:47, 12.64s/it]
|
| 2 |
0%| | 2/30000 [00:14<52:20:08, 6.28s/it]
|
| 3 |
0%| | 3/30000 [00:16<35:11:45, 4.22s/it]
|
| 4 |
0%| | 4/30000 [00:18<27:10:39, 3.26s/it]
|
| 5 |
0%| | 5/30000 [00:19<22:43:51, 2.73s/it]
|
| 6 |
0%| | 6/30000 [00:21<20:05:04, 2.41s/it]
|
| 7 |
0%| | 7/30000 [00:23<18:23:19, 2.21s/it]
|
| 8 |
0%| | 8/30000 [00:25<17:24:08, 2.09s/it]
|
| 9 |
0%| | 9/30000 [00:27<16:37:23, 2.00s/it]
|
| 10 |
0%| | 10/30000 [00:28<16:10:55, 1.94s/it]
|
| 11 |
|
| 12 |
0%| | 10/30000 [00:28<16:10:55, 1.94s/it]
|
| 13 |
0%| | 11/30000 [00:30<15:49:20, 1.90s/it]
|
| 14 |
0%| | 12/30000 [00:32<15:33:32, 1.87s/it]
|
| 15 |
0%| | 13/30000 [00:34<15:24:15, 1.85s/it]
|
| 16 |
0%| | 14/30000 [00:36<15:15:55, 1.83s/it]
|
| 17 |
0%| | 15/30000 [00:37<15:12:53, 1.83s/it]
|
| 18 |
0%| | 16/30000 [00:39<15:09:48, 1.82s/it]
|
| 19 |
0%| | 17/30000 [00:41<15:07:04, 1.82s/it]
|
| 20 |
0%| | 18/30000 [00:43<15:05:37, 1.81s/it]
|
| 21 |
0%| | 19/30000 [00:45<15:05:00, 1.81s/it]
|
| 22 |
0%| | 20/30000 [00:46<15:04:44, 1.81s/it]
|
| 23 |
|
| 24 |
0%| | 20/30000 [00:46<15:04:44, 1.81s/it]
|
| 25 |
0%| | 21/30000 [00:48<15:07:17, 1.82s/it]
|
| 26 |
0%| | 22/30000 [00:50<15:10:17, 1.82s/it]
|
| 27 |
0%| | 23/30000 [00:52<15:06:41, 1.81s/it]
|
| 28 |
0%| | 24/30000 [00:54<15:05:48, 1.81s/it]
|
| 29 |
0%| | 25/30000 [00:55<15:03:36, 1.81s/it]
|
| 30 |
0%| | 26/30000 [00:57<15:04:39, 1.81s/it]
|
| 31 |
0%| | 27/30000 [00:59<15:05:03, 1.81s/it]
|
| 32 |
0%| | 28/30000 [01:01<15:05:37, 1.81s/it]
|
| 33 |
0%| | 29/30000 [01:03<15:05:24, 1.81s/it]
|
| 34 |
0%| | 30/30000 [01:05<15:05:52, 1.81s/it]
|
| 35 |
|
| 36 |
0%| | 30/30000 [01:05<15:05:52, 1.81s/it]
|
| 37 |
0%| | 31/30000 [01:06<15:03:54, 1.81s/it]
|
| 38 |
0%| | 32/30000 [01:08<15:04:22, 1.81s/it]
|
| 39 |
0%| | 33/30000 [01:10<15:04:56, 1.81s/it]
|
| 40 |
0%| | 34/30000 [01:12<15:05:47, 1.81s/it]
|
| 41 |
0%| | 35/30000 [01:14<15:06:06, 1.81s/it]
|
| 42 |
0%| | 36/30000 [01:15<15:06:55, 1.82s/it]
|
| 43 |
0%| | 37/30000 [01:17<15:07:31, 1.82s/it]
|
| 44 |
0%| | 38/30000 [01:19<15:10:34, 1.82s/it]
|
| 45 |
0%| | 39/30000 [01:21<15:09:30, 1.82s/it]
|
| 46 |
0%| | 40/30000 [01:23<15:10:54, 1.82s/it]
|
| 47 |
|
| 48 |
0%| | 40/30000 [01:23<15:10:54, 1.82s/it]
|
| 49 |
0%| | 41/30000 [01:25<15:14:52, 1.83s/it]
|
| 50 |
0%| | 42/30000 [01:26<15:13:02, 1.83s/it]
|
| 51 |
0%| | 43/30000 [01:28<15:14:17, 1.83s/it]
|
| 52 |
0%| | 44/30000 [01:30<15:19:36, 1.84s/it]
|
| 53 |
0%| | 45/30000 [01:32<15:18:50, 1.84s/it]
|
| 54 |
0%| | 46/30000 [01:34<15:19:09, 1.84s/it]
|
| 55 |
0%| | 47/30000 [01:36<15:17:18, 1.84s/it]
|
| 56 |
0%| | 48/30000 [01:37<15:16:58, 1.84s/it]
|
| 57 |
0%| | 49/30000 [01:39<15:15:04, 1.83s/it]
|
| 58 |
0%| | 50/30000 [01:41<15:16:32, 1.84s/it]
|
| 59 |
|
| 60 |
0%| | 50/30000 [01:41<15:16:32, 1.84s/it]
|
| 61 |
0%| | 51/30000 [01:43<15:18:53, 1.84s/it]
|
| 62 |
0%| | 52/30000 [01:45<15:22:58, 1.85s/it]
|
| 63 |
0%| | 53/30000 [01:47<15:19:36, 1.84s/it]
|
| 64 |
0%| | 54/30000 [01:48<15:17:48, 1.84s/it]
|
| 65 |
0%| | 55/30000 [01:50<15:17:19, 1.84s/it]
|
| 66 |
0%| | 56/30000 [01:52<15:16:55, 1.84s/it]
|
| 67 |
0%| | 57/30000 [01:54<15:16:19, 1.84s/it]
|
| 68 |
0%| | 58/30000 [01:56<15:16:38, 1.84s/it]
|
| 69 |
0%| | 59/30000 [01:58<15:16:30, 1.84s/it]
|
| 70 |
0%| | 60/30000 [02:00<15:18:29, 1.84s/it]
|
| 71 |
|
| 72 |
0%| | 60/30000 [02:00<15:18:29, 1.84s/it]
|
| 73 |
0%| | 61/30000 [02:01<15:18:43, 1.84s/it]
|
| 74 |
0%| | 62/30000 [02:03<15:19:39, 1.84s/it]
|
| 75 |
0%| | 63/30000 [02:05<15:18:38, 1.84s/it]
|
|
|
|
| 1 |
+
[robosuite WARNING] No private macro file found! (macros.py:57)
|
| 2 |
+
[robosuite WARNING] It is recommended to use a private macro file (macros.py:58)
|
| 3 |
+
[robosuite WARNING] To setup, run: python /home/seonho/clvla/benchmarks/robocasa_v2/robosuite/robosuite/scripts/setup_macros.py (macros.py:59)
|
| 4 |
+
[robosuite WARNING] Could not import robosuite_models. Some robots may not be available. If you want to use these robots, please install robosuite_models from source (https://github.com/ARISE-Initiative/robosuite_models) or through pip install. (__init__.py:30)
|
| 5 |
+
[robosuite WARNING] Could not load the mink-based whole-body IK. Make sure you install related import properly (e.g. pip install mink==0.0.5), otherwise you will not be able to use the default IK controller setting for GR1 robot. (__init__.py:40)
|
| 6 |
+
/home/seonho/miniconda3/envs/robocasa/lib/python3.11/site-packages/albumentations/__init__.py:13: UserWarning: A new version of Albumentations is available: 2.0.8 (you have 1.4.18). Upgrade using: pip install -U albumentations. To disable automatic update checks, set the environment variable NO_ALBUMENTATIONS_UPDATE to 1.
|
| 7 |
+
check_for_updates()
|
| 8 |
+
`use_fast` is set to `True` but the image processor class does not have a fast version. Falling back to the slow version.
|
| 9 |
+
WARNING: mimicgen environments not imported since mimicgen is not installed!
|
| 10 |
+
|
| 11 |
+
==================================================
|
| 12 |
+
GR00T FINE-TUNING CONFIGURATION:
|
| 13 |
+
==================================================
|
| 14 |
+
config: /home/seonho/clvla/benchmarks/robocasa_v2/experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse_phase3_only.yaml
|
| 15 |
+
dataset_soup: None
|
| 16 |
+
output_dir: /tmp/gr00t
|
| 17 |
+
output_root: None
|
| 18 |
+
data_config: panda_omron
|
| 19 |
+
batch_size: 64
|
| 20 |
+
max_steps: 300000
|
| 21 |
+
num_gpus: 1
|
| 22 |
+
save_steps: 20000
|
| 23 |
+
run_name: None
|
| 24 |
+
save_total_limit: 100
|
| 25 |
+
seed: 42
|
| 26 |
+
base_model_path: nvidia/GR00T-N1.5-3B
|
| 27 |
+
tune_llm: False
|
| 28 |
+
tune_visual: False
|
| 29 |
+
tune_projector: True
|
| 30 |
+
tune_diffusion_model: True
|
| 31 |
+
resume: False
|
| 32 |
+
learning_rate: 3e-05
|
| 33 |
+
weight_decay: 1e-05
|
| 34 |
+
warmup_ratio: 0.05
|
| 35 |
+
lora_rank: 0
|
| 36 |
+
lora_alpha: 16
|
| 37 |
+
lora_dropout: 0.1
|
| 38 |
+
lora_full_model: False
|
| 39 |
+
dataloader_num_workers: 8
|
| 40 |
+
report_to: wandb
|
| 41 |
+
embodiment_tag: new_embodiment
|
| 42 |
+
video_backend: opencv
|
| 43 |
+
balance_dataset_weights: True
|
| 44 |
+
balance_trajectory_weights: True
|
| 45 |
+
ds_weights_alpha: 0.4
|
| 46 |
+
==================================================
|
| 47 |
+
|
| 48 |
+
Using 1 GPUs
|
| 49 |
+
|
| 50 |
+
================================================================================
|
| 51 |
+
Starting sweep branch: mgd_token_mask_ratio_0.3
|
| 52 |
+
Sweep vars: {'mgd_token_mask_ratio': 0.3}
|
| 53 |
+
================================================================================
|
| 54 |
+
|
| 55 |
+
--------------------------------------------------------------------------------
|
| 56 |
+
Running phase 1: phase3_fm_mgd_fixed_0p5
|
| 57 |
+
Policy type: groot_mgd
|
| 58 |
+
Base model path: /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
|
| 59 |
+
Output dir: /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3
|
| 60 |
+
Trainable preset: None
|
| 61 |
+
Policy overrides: {'mgd_enabled': True, 'mgd_fm_loss_weight': 1.0, 'mgd_loss_weight': 0.5, 'mgd_token_mask_ratio': 0.3, 'mgd_use_cosine_loss': True, 'mgd_use_mse_loss': True}
|
| 62 |
+
--------------------------------------------------------------------------------
|
| 63 |
+
|
| 64 |
+
[{'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseBlenderLid/20250822/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseBlenderLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseFridge/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'CloseFridge', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CloseToasterOvenDoor/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'CloseToasterOvenDoor', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/CoffeeSetupMug/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'CoffeeSetupMug', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/NavigateKitchen/20250821/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'NavigateKitchen', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenCabinet/20250819/lerobot', 'horizon': 1050, 'filter_key': '100_demos', 'task': 'OpenCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenDrawer/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'OpenDrawer', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenStandMixerHead/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'OpenStandMixerHead', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToCabinet/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToCabinet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCounterToStove/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceCounterToStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceDrawerToCounter/20250820/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'PickPlaceDrawerToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceSinkToCounter/20250819/lerobot', 'horizon': 900, 'filter_key': '100_demos', 'task': 'PickPlaceSinkToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceToasterToCounter/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'PickPlaceToasterToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/PickPlaceCabinetToCounter/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'PickPlaceCabinetToCounter', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnElectricKettle/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnElectricKettle', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnMicrowave/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffMicrowave/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffMicrowave', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/OpenElectricKettleLid/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'OpenElectricKettleLid', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/SlideDishwasherRack/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'SlideDishwasherRack', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnSinkFaucet/20250819/lerobot', 'horizon': 600, 'filter_key': '100_demos', 'task': 'TurnOnSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnSinkSpout/20250820/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnSinkSpout', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffSinkFaucet/20250819/lerobot', 'horizon': 300, 'filter_key': '100_demos', 'task': 'TurnOffSinkFaucet', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustWaterTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustWaterTemperature', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOffStove/20250819/lerobot', 'horizon': 750, 'filter_key': '100_demos', 'task': 'TurnOffStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/TurnOnStove/20250819/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'TurnOnStove', 'split': 'pretrain', 'source': 'human'}, {'path': '/home/seonho/clvla/benchmarks/robocasa_v2/robocasa/datasets/v1.0/pretrain/atomic/AdjustToasterOvenTemperature/20250820/lerobot', 'horizon': 450, 'filter_key': '100_demos', 'task': 'AdjustToasterOvenTemperature', 'split': 'pretrain', 'source': 'human'}]/home/seonho/clvla/benchmarks/robocasa_v2/Isaac-GR00T/gr00t/data/transform/state_action.py:257: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.detach().clone() or sourceTensor.detach().clone().requires_grad_(True), rather than torch.tensor(sourceTensor).
|
| 65 |
+
self.statistics[key] = torch.tensor(value)
|
| 66 |
+
|
| 67 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 68 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 69 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 70 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 71 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 72 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 73 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 74 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 75 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 76 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 77 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 78 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 79 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 80 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 81 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 82 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 83 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 84 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 85 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 86 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 87 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 88 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 89 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 90 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 91 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 92 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 93 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 94 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 95 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 96 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 97 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 98 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 99 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 100 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 101 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 102 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 103 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 104 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 105 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 106 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 107 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 108 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 109 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 110 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 111 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 112 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 113 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 114 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 115 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 116 |
+
Using 100 subset demos for filter_key: 100_demos
|
| 117 |
+
Initialized dataset lerobot with EmbodimentTag.NEW_EMBODIMENT
|
| 118 |
+
dataset weights: [1. 0.88494669 0.76443286 0.83888607 0.73785464 1.00501644
|
| 119 |
+
0.80029956 0.66065525 0.83967491 0.83515002 0.95047759 0.86791692
|
| 120 |
+
0.88329477 0.78510653 0.64381843 0.67404766 0.69697218 0.60443684
|
| 121 |
+
0.78448103 0.83447486 0.61146215 0.64384058 0.79545025 0.93845856
|
| 122 |
+
0.75517122 0.7973985 ]
|
| 123 |
+
Loaded 26 datasets
|
| 124 |
+
Loading pretrained dual brain from /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
|
| 125 |
+
Tune backbone vision tower: False
|
| 126 |
+
Tune backbone LLM: False
|
| 127 |
+
Tune action head projector: True
|
| 128 |
+
Tune action head DiT: True
|
| 129 |
+
Model not found or avail in the huggingface hub. Loading from local path: /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
|
| 130 |
+
Tune backbone llm: False
|
| 131 |
+
Tune backbone visual: True
|
| 132 |
+
Total number of DiT parameters: 550386688
|
| 133 |
+
Total number of SelfAttentionTransformer parameters: 201433088
|
| 134 |
+
Tune action head projector: True
|
| 135 |
+
Tune action head diffusion model: True
|
| 136 |
+
|
| 137 |
+
Tune backbone llm: False
|
| 138 |
+
Tune backbone visual: False
|
| 139 |
+
Warning: No backbone trainable parameters found.
|
| 140 |
+
Tune action head projector: True
|
| 141 |
+
Tune action head diffusion model: True
|
| 142 |
+
Trainable summary: {'trainable_preset': 'default', 'trainable_modules': 'action_head.future_tokens(49,152), action_head.vlln(4,096), action_head.vl_self_attention(201,433,088), action_head.state_encoder(52,510,720), action_head.action_encoder(228,212,736), action_head.action_decoder(34,636,800), action_head.position_embedding(1,572,864), action_head.model(550,386,688), sequence_mgd_head(1,574,913)', 'trainable_module_names': 'action_head.future_tokens, action_head.vlln, action_head.vl_self_attention, action_head.state_encoder, action_head.action_encoder, action_head.action_decoder, action_head.position_embedding, action_head.model, sequence_mgd_head', 'trainable_module_param_counts': {'action_head.future_tokens': 49152, 'action_head.vlln': 4096, 'action_head.vl_self_attention': 201433088, 'action_head.state_encoder': 52510720, 'action_head.action_encoder': 228212736, 'action_head.action_decoder': 34636800, 'action_head.position_embedding': 1572864, 'action_head.model': 550386688, 'sequence_mgd_head': 1574913}, 'trainable_param_count': 1070381057, 'total_param_count': 2738321857, 'trainable_param_ratio': 0.3908894253112628}
|
| 143 |
+
Run name: mgd_token_mask_ratio_0.3_phase3
|
| 144 |
+
train dataloader length: 6873
|
| 145 |
+
train dataset length: 439854
|
| 146 |
+
GPU memory before training: 7.117771625518799 GB
|
| 147 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/seonho/.netrc.
|
| 148 |
+
wandb: Currently logged in as: minje227_hyu (minje227_hyu-hanyang-university) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 149 |
+
wandb: setting up run 4hsojbcy
|
| 150 |
+
wandb: Tracking run with wandb version 0.25.0
|
| 151 |
+
wandb: Run data is saved locally in /home/seonho/wandb/run-20260614_013832-4hsojbcy
|
| 152 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 153 |
+
wandb: Syncing run mgd_token_mask_ratio_0.3_phase3
|
| 154 |
+
wandb: ⭐️ View project at https://wandb.ai/minje227_hyu-hanyang-university/huggingface
|
| 155 |
+
wandb: 🚀 View run at https://wandb.ai/minje227_hyu-hanyang-university/huggingface/runs/4hsojbcy
|
| 156 |
+
TensorBoard logs will be saved to: /home/seonho/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/runs
|
| 157 |
+
|
| 158 |
0%| | 0/30000 [00:00<?, ?it/s]Could not estimate the number of tokens of the input, floating-point operations will not be computed
|
| 159 |
+
|
| 160 |
0%| | 1/30000 [00:12<105:19:47, 12.64s/it]
|
| 161 |
0%| | 2/30000 [00:14<52:20:08, 6.28s/it]
|
| 162 |
0%| | 3/30000 [00:16<35:11:45, 4.22s/it]
|
| 163 |
0%| | 4/30000 [00:18<27:10:39, 3.26s/it]
|
| 164 |
0%| | 5/30000 [00:19<22:43:51, 2.73s/it]
|
| 165 |
0%| | 6/30000 [00:21<20:05:04, 2.41s/it]
|
| 166 |
0%| | 7/30000 [00:23<18:23:19, 2.21s/it]
|
| 167 |
0%| | 8/30000 [00:25<17:24:08, 2.09s/it]
|
| 168 |
0%| | 9/30000 [00:27<16:37:23, 2.00s/it]
|
| 169 |
0%| | 10/30000 [00:28<16:10:55, 1.94s/it]
|
| 170 |
|
| 171 |
0%| | 10/30000 [00:28<16:10:55, 1.94s/it]
|
| 172 |
0%| | 11/30000 [00:30<15:49:20, 1.90s/it]
|
| 173 |
0%| | 12/30000 [00:32<15:33:32, 1.87s/it]
|
| 174 |
0%| | 13/30000 [00:34<15:24:15, 1.85s/it]
|
| 175 |
0%| | 14/30000 [00:36<15:15:55, 1.83s/it]
|
| 176 |
0%| | 15/30000 [00:37<15:12:53, 1.83s/it]
|
| 177 |
0%| | 16/30000 [00:39<15:09:48, 1.82s/it]
|
| 178 |
0%| | 17/30000 [00:41<15:07:04, 1.82s/it]
|
| 179 |
0%| | 18/30000 [00:43<15:05:37, 1.81s/it]
|
| 180 |
0%| | 19/30000 [00:45<15:05:00, 1.81s/it]
|
| 181 |
0%| | 20/30000 [00:46<15:04:44, 1.81s/it]
|
| 182 |
|
| 183 |
0%| | 20/30000 [00:46<15:04:44, 1.81s/it]
|
| 184 |
0%| | 21/30000 [00:48<15:07:17, 1.82s/it]
|
| 185 |
0%| | 22/30000 [00:50<15:10:17, 1.82s/it]
|
| 186 |
0%| | 23/30000 [00:52<15:06:41, 1.81s/it]
|
| 187 |
0%| | 24/30000 [00:54<15:05:48, 1.81s/it]
|
| 188 |
0%| | 25/30000 [00:55<15:03:36, 1.81s/it]
|
| 189 |
0%| | 26/30000 [00:57<15:04:39, 1.81s/it]
|
| 190 |
0%| | 27/30000 [00:59<15:05:03, 1.81s/it]
|
| 191 |
0%| | 28/30000 [01:01<15:05:37, 1.81s/it]
|
| 192 |
0%| | 29/30000 [01:03<15:05:24, 1.81s/it]
|
| 193 |
0%| | 30/30000 [01:05<15:05:52, 1.81s/it]
|
| 194 |
|
| 195 |
0%| | 30/30000 [01:05<15:05:52, 1.81s/it]
|
| 196 |
0%| | 31/30000 [01:06<15:03:54, 1.81s/it]
|
| 197 |
0%| | 32/30000 [01:08<15:04:22, 1.81s/it]
|
| 198 |
0%| | 33/30000 [01:10<15:04:56, 1.81s/it]
|
| 199 |
0%| | 34/30000 [01:12<15:05:47, 1.81s/it]
|
| 200 |
0%| | 35/30000 [01:14<15:06:06, 1.81s/it]
|
| 201 |
0%| | 36/30000 [01:15<15:06:55, 1.82s/it]
|
| 202 |
0%| | 37/30000 [01:17<15:07:31, 1.82s/it]
|
| 203 |
0%| | 38/30000 [01:19<15:10:34, 1.82s/it]
|
| 204 |
0%| | 39/30000 [01:21<15:09:30, 1.82s/it]
|
| 205 |
0%| | 40/30000 [01:23<15:10:54, 1.82s/it]
|
| 206 |
|
| 207 |
0%| | 40/30000 [01:23<15:10:54, 1.82s/it]
|
| 208 |
0%| | 41/30000 [01:25<15:14:52, 1.83s/it]
|
| 209 |
0%| | 42/30000 [01:26<15:13:02, 1.83s/it]
|
| 210 |
0%| | 43/30000 [01:28<15:14:17, 1.83s/it]
|
| 211 |
0%| | 44/30000 [01:30<15:19:36, 1.84s/it]
|
| 212 |
0%| | 45/30000 [01:32<15:18:50, 1.84s/it]
|
| 213 |
0%| | 46/30000 [01:34<15:19:09, 1.84s/it]
|
| 214 |
0%| | 47/30000 [01:36<15:17:18, 1.84s/it]
|
| 215 |
0%| | 48/30000 [01:37<15:16:58, 1.84s/it]
|
| 216 |
0%| | 49/30000 [01:39<15:15:04, 1.83s/it]
|
| 217 |
0%| | 50/30000 [01:41<15:16:32, 1.84s/it]
|
| 218 |
|
| 219 |
0%| | 50/30000 [01:41<15:16:32, 1.84s/it]
|
| 220 |
0%| | 51/30000 [01:43<15:18:53, 1.84s/it]
|
| 221 |
0%| | 52/30000 [01:45<15:22:58, 1.85s/it]
|
| 222 |
0%| | 53/30000 [01:47<15:19:36, 1.84s/it]
|
| 223 |
0%| | 54/30000 [01:48<15:17:48, 1.84s/it]
|
| 224 |
0%| | 55/30000 [01:50<15:17:19, 1.84s/it]
|
| 225 |
0%| | 56/30000 [01:52<15:16:55, 1.84s/it]
|
| 226 |
0%| | 57/30000 [01:54<15:16:19, 1.84s/it]
|
| 227 |
0%| | 58/30000 [01:56<15:16:38, 1.84s/it]
|
| 228 |
0%| | 59/30000 [01:58<15:16:30, 1.84s/it]
|
| 229 |
0%| | 60/30000 [02:00<15:18:29, 1.84s/it]
|
| 230 |
|
| 231 |
0%| | 60/30000 [02:00<15:18:29, 1.84s/it]
|
| 232 |
0%| | 61/30000 [02:01<15:18:43, 1.84s/it]
|
| 233 |
0%| | 62/30000 [02:03<15:19:39, 1.84s/it]
|
| 234 |
0%| | 63/30000 [02:05<15:18:38, 1.84s/it]
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3_only_train_gpu01.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/resolved_config.yaml
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_loss_cosine_mse_phase3
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.3
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
tune_llm: false
|
| 21 |
+
tune_visual: false
|
| 22 |
+
tune_projector: true
|
| 23 |
+
tune_diffusion_model: true
|
| 24 |
+
losses:
|
| 25 |
+
mgd_enabled: true
|
| 26 |
+
mgd_fm_loss_weight: 1.0
|
| 27 |
+
mgd_loss_weight: 0.5
|
| 28 |
+
mgd_token_mask_ratio: 0.3
|
| 29 |
+
mgd_use_cosine_loss: true
|
| 30 |
+
mgd_use_mse_loss: true
|
| 31 |
+
resolved_sweep:
|
| 32 |
+
mgd_token_mask_ratio: 0.3
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/mgd_v2_mask_ratio_ablation_0p1.yaml
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_mask_ratio_ablation
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.1
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase2_mgd_only
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
preset: processing_line_only
|
| 21 |
+
tune_llm: false
|
| 22 |
+
tune_visual: false
|
| 23 |
+
tune_projector: false
|
| 24 |
+
tune_diffusion_model: false
|
| 25 |
+
losses:
|
| 26 |
+
mgd_enabled: true
|
| 27 |
+
mgd_fm_loss_weight: 0.0
|
| 28 |
+
mgd_loss_weight: 1.0
|
| 29 |
+
mgd_sequence_hidden_dim: 512
|
| 30 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 31 |
+
mgd_use_cosine_loss: true
|
| 32 |
+
mgd_use_mse_loss: false
|
| 33 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 34 |
+
max_steps: 30000
|
| 35 |
+
save_steps: 0
|
| 36 |
+
trainable:
|
| 37 |
+
tune_llm: false
|
| 38 |
+
tune_visual: false
|
| 39 |
+
tune_projector: true
|
| 40 |
+
tune_diffusion_model: true
|
| 41 |
+
losses:
|
| 42 |
+
mgd_enabled: true
|
| 43 |
+
mgd_fm_loss_weight: 1.0
|
| 44 |
+
mgd_loss_weight: 0.5
|
| 45 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 46 |
+
mgd_use_cosine_loss: true
|
| 47 |
+
mgd_use_mse_loss: false
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/config.json
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"action_dim": 32,
|
| 3 |
+
"action_head_cfg": {
|
| 4 |
+
"action_dim": 32,
|
| 5 |
+
"action_horizon": 16,
|
| 6 |
+
"add_pos_embed": true,
|
| 7 |
+
"backbone_embedding_dim": 2048,
|
| 8 |
+
"diffusion_model_cfg": {
|
| 9 |
+
"attention_head_dim": 48,
|
| 10 |
+
"cross_attention_dim": 2048,
|
| 11 |
+
"dropout": 0.2,
|
| 12 |
+
"final_dropout": true,
|
| 13 |
+
"interleave_self_attention": true,
|
| 14 |
+
"norm_type": "ada_norm",
|
| 15 |
+
"num_attention_heads": 32,
|
| 16 |
+
"num_layers": 16,
|
| 17 |
+
"output_dim": 1024,
|
| 18 |
+
"positional_embeddings": null
|
| 19 |
+
},
|
| 20 |
+
"hidden_size": 1024,
|
| 21 |
+
"input_embedding_dim": 1536,
|
| 22 |
+
"max_action_dim": 32,
|
| 23 |
+
"max_state_dim": 64,
|
| 24 |
+
"model_dtype": "float32",
|
| 25 |
+
"noise_beta_alpha": 1.5,
|
| 26 |
+
"noise_beta_beta": 1.0,
|
| 27 |
+
"noise_s": 0.999,
|
| 28 |
+
"num_inference_timesteps": 4,
|
| 29 |
+
"num_target_vision_tokens": 32,
|
| 30 |
+
"num_timestep_buckets": 1000,
|
| 31 |
+
"tune_diffusion_model": true,
|
| 32 |
+
"tune_projector": true,
|
| 33 |
+
"use_vlln": true,
|
| 34 |
+
"vl_self_attention_cfg": {
|
| 35 |
+
"attention_head_dim": 64,
|
| 36 |
+
"dropout": 0.2,
|
| 37 |
+
"final_dropout": true,
|
| 38 |
+
"num_attention_heads": 32,
|
| 39 |
+
"num_layers": 4,
|
| 40 |
+
"positional_embeddings": null
|
| 41 |
+
}
|
| 42 |
+
},
|
| 43 |
+
"action_horizon": 16,
|
| 44 |
+
"architectures": [
|
| 45 |
+
"GR00T_N1_5_MGD"
|
| 46 |
+
],
|
| 47 |
+
"attn_implementation": null,
|
| 48 |
+
"backbone_cfg": {
|
| 49 |
+
"eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
|
| 50 |
+
"load_bf16": false,
|
| 51 |
+
"project_to_dim": null,
|
| 52 |
+
"reproject_vision": false,
|
| 53 |
+
"select_layer": 12,
|
| 54 |
+
"tune_llm": false,
|
| 55 |
+
"tune_visual": true,
|
| 56 |
+
"use_flash_attention": true
|
| 57 |
+
},
|
| 58 |
+
"compute_dtype": "bfloat16",
|
| 59 |
+
"hidden_size": 2048,
|
| 60 |
+
"mgd_enabled": true,
|
| 61 |
+
"mgd_fm_loss_weight": 0.0,
|
| 62 |
+
"mgd_loss_weight": 1.0,
|
| 63 |
+
"mgd_loss_weight_end": 0.0,
|
| 64 |
+
"mgd_loss_weight_schedule": null,
|
| 65 |
+
"mgd_loss_weight_start": 0.05,
|
| 66 |
+
"mgd_pretrained_projector_path": null,
|
| 67 |
+
"mgd_sequence_hidden_dim": 512,
|
| 68 |
+
"mgd_target_dim": 512,
|
| 69 |
+
"mgd_target_pooling": "flatten",
|
| 70 |
+
"mgd_target_projection": "frozen_random",
|
| 71 |
+
"mgd_token_mask_ratio": 0.1,
|
| 72 |
+
"mgd_use_cosine_loss": true,
|
| 73 |
+
"mgd_use_mse_loss": false,
|
| 74 |
+
"model_dtype": "float32",
|
| 75 |
+
"model_type": "gr00t_n1_5",
|
| 76 |
+
"torch_dtype": "bfloat16",
|
| 77 |
+
"transformers_version": "4.51.3"
|
| 78 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/trainer_state.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/config.json
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"action_dim": 32,
|
| 3 |
+
"action_head_cfg": {
|
| 4 |
+
"action_dim": 32,
|
| 5 |
+
"action_horizon": 16,
|
| 6 |
+
"add_pos_embed": true,
|
| 7 |
+
"backbone_embedding_dim": 2048,
|
| 8 |
+
"diffusion_model_cfg": {
|
| 9 |
+
"attention_head_dim": 48,
|
| 10 |
+
"cross_attention_dim": 2048,
|
| 11 |
+
"dropout": 0.2,
|
| 12 |
+
"final_dropout": true,
|
| 13 |
+
"interleave_self_attention": true,
|
| 14 |
+
"norm_type": "ada_norm",
|
| 15 |
+
"num_attention_heads": 32,
|
| 16 |
+
"num_layers": 16,
|
| 17 |
+
"output_dim": 1024,
|
| 18 |
+
"positional_embeddings": null
|
| 19 |
+
},
|
| 20 |
+
"hidden_size": 1024,
|
| 21 |
+
"input_embedding_dim": 1536,
|
| 22 |
+
"max_action_dim": 32,
|
| 23 |
+
"max_state_dim": 64,
|
| 24 |
+
"model_dtype": "float32",
|
| 25 |
+
"noise_beta_alpha": 1.5,
|
| 26 |
+
"noise_beta_beta": 1.0,
|
| 27 |
+
"noise_s": 0.999,
|
| 28 |
+
"num_inference_timesteps": 4,
|
| 29 |
+
"num_target_vision_tokens": 32,
|
| 30 |
+
"num_timestep_buckets": 1000,
|
| 31 |
+
"tune_diffusion_model": true,
|
| 32 |
+
"tune_projector": true,
|
| 33 |
+
"use_vlln": true,
|
| 34 |
+
"vl_self_attention_cfg": {
|
| 35 |
+
"attention_head_dim": 64,
|
| 36 |
+
"dropout": 0.2,
|
| 37 |
+
"final_dropout": true,
|
| 38 |
+
"num_attention_heads": 32,
|
| 39 |
+
"num_layers": 4,
|
| 40 |
+
"positional_embeddings": null
|
| 41 |
+
}
|
| 42 |
+
},
|
| 43 |
+
"action_horizon": 16,
|
| 44 |
+
"architectures": [
|
| 45 |
+
"GR00T_N1_5_MGD"
|
| 46 |
+
],
|
| 47 |
+
"attn_implementation": null,
|
| 48 |
+
"backbone_cfg": {
|
| 49 |
+
"eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
|
| 50 |
+
"load_bf16": false,
|
| 51 |
+
"project_to_dim": null,
|
| 52 |
+
"reproject_vision": false,
|
| 53 |
+
"select_layer": 12,
|
| 54 |
+
"tune_llm": false,
|
| 55 |
+
"tune_visual": true,
|
| 56 |
+
"use_flash_attention": true
|
| 57 |
+
},
|
| 58 |
+
"compute_dtype": "bfloat16",
|
| 59 |
+
"hidden_size": 2048,
|
| 60 |
+
"mgd_enabled": true,
|
| 61 |
+
"mgd_fm_loss_weight": 1.0,
|
| 62 |
+
"mgd_loss_weight": 0.5,
|
| 63 |
+
"mgd_loss_weight_end": 0.0,
|
| 64 |
+
"mgd_loss_weight_schedule": null,
|
| 65 |
+
"mgd_loss_weight_start": 0.05,
|
| 66 |
+
"mgd_pretrained_projector_path": null,
|
| 67 |
+
"mgd_sequence_hidden_dim": 512,
|
| 68 |
+
"mgd_target_dim": 512,
|
| 69 |
+
"mgd_target_pooling": "flatten",
|
| 70 |
+
"mgd_target_projection": "frozen_random",
|
| 71 |
+
"mgd_token_mask_ratio": 0.1,
|
| 72 |
+
"mgd_use_cosine_loss": true,
|
| 73 |
+
"mgd_use_mse_loss": false,
|
| 74 |
+
"model_dtype": "float32",
|
| 75 |
+
"model_type": "gr00t_n1_5",
|
| 76 |
+
"torch_dtype": "bfloat16",
|
| 77 |
+
"transformers_version": "4.51.3"
|
| 78 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/resolved_config.yaml
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_mask_ratio_ablation
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.1
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase2_mgd_only
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
preset: processing_line_only
|
| 21 |
+
tune_llm: false
|
| 22 |
+
tune_visual: false
|
| 23 |
+
tune_projector: false
|
| 24 |
+
tune_diffusion_model: false
|
| 25 |
+
losses:
|
| 26 |
+
mgd_enabled: true
|
| 27 |
+
mgd_fm_loss_weight: 0.0
|
| 28 |
+
mgd_loss_weight: 1.0
|
| 29 |
+
mgd_sequence_hidden_dim: 512
|
| 30 |
+
mgd_token_mask_ratio: 0.1
|
| 31 |
+
mgd_use_cosine_loss: true
|
| 32 |
+
mgd_use_mse_loss: false
|
| 33 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 34 |
+
max_steps: 30000
|
| 35 |
+
save_steps: 0
|
| 36 |
+
trainable:
|
| 37 |
+
tune_llm: false
|
| 38 |
+
tune_visual: false
|
| 39 |
+
tune_projector: true
|
| 40 |
+
tune_diffusion_model: true
|
| 41 |
+
losses:
|
| 42 |
+
mgd_enabled: true
|
| 43 |
+
mgd_fm_loss_weight: 1.0
|
| 44 |
+
mgd_loss_weight: 0.5
|
| 45 |
+
mgd_token_mask_ratio: 0.1
|
| 46 |
+
mgd_use_cosine_loss: true
|
| 47 |
+
mgd_use_mse_loss: false
|
| 48 |
+
resolved_sweep:
|
| 49 |
+
mgd_token_mask_ratio: 0.1
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/mgd_v2_mask_ratio_ablation_0p2.yaml
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_mask_ratio_ablation
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.2
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase2_mgd_only
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
preset: processing_line_only
|
| 21 |
+
tune_llm: false
|
| 22 |
+
tune_visual: false
|
| 23 |
+
tune_projector: false
|
| 24 |
+
tune_diffusion_model: false
|
| 25 |
+
losses:
|
| 26 |
+
mgd_enabled: true
|
| 27 |
+
mgd_fm_loss_weight: 0.0
|
| 28 |
+
mgd_loss_weight: 1.0
|
| 29 |
+
mgd_sequence_hidden_dim: 512
|
| 30 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 31 |
+
mgd_use_cosine_loss: true
|
| 32 |
+
mgd_use_mse_loss: false
|
| 33 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 34 |
+
max_steps: 30000
|
| 35 |
+
save_steps: 0
|
| 36 |
+
trainable:
|
| 37 |
+
tune_llm: false
|
| 38 |
+
tune_visual: false
|
| 39 |
+
tune_projector: true
|
| 40 |
+
tune_diffusion_model: true
|
| 41 |
+
losses:
|
| 42 |
+
mgd_enabled: true
|
| 43 |
+
mgd_fm_loss_weight: 1.0
|
| 44 |
+
mgd_loss_weight: 0.5
|
| 45 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 46 |
+
mgd_use_cosine_loss: true
|
| 47 |
+
mgd_use_mse_loss: false
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase3/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/resolved_config.yaml
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_mask_ratio_ablation
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.2
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase2_mgd_only
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
preset: processing_line_only
|
| 21 |
+
tune_llm: false
|
| 22 |
+
tune_visual: false
|
| 23 |
+
tune_projector: false
|
| 24 |
+
tune_diffusion_model: false
|
| 25 |
+
losses:
|
| 26 |
+
mgd_enabled: true
|
| 27 |
+
mgd_fm_loss_weight: 0.0
|
| 28 |
+
mgd_loss_weight: 1.0
|
| 29 |
+
mgd_sequence_hidden_dim: 512
|
| 30 |
+
mgd_token_mask_ratio: 0.2
|
| 31 |
+
mgd_use_cosine_loss: true
|
| 32 |
+
mgd_use_mse_loss: false
|
| 33 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 34 |
+
max_steps: 30000
|
| 35 |
+
save_steps: 0
|
| 36 |
+
trainable:
|
| 37 |
+
tune_llm: false
|
| 38 |
+
tune_visual: false
|
| 39 |
+
tune_projector: true
|
| 40 |
+
tune_diffusion_model: true
|
| 41 |
+
losses:
|
| 42 |
+
mgd_enabled: true
|
| 43 |
+
mgd_fm_loss_weight: 1.0
|
| 44 |
+
mgd_loss_weight: 0.5
|
| 45 |
+
mgd_token_mask_ratio: 0.2
|
| 46 |
+
mgd_use_cosine_loss: true
|
| 47 |
+
mgd_use_mse_loss: false
|
| 48 |
+
resolved_sweep:
|
| 49 |
+
mgd_token_mask_ratio: 0.2
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/mgd_v2_mask_ratio_ablation_0p4.yaml
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_mask_ratio_ablation
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.4
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase2_mgd_only
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
preset: processing_line_only
|
| 21 |
+
tune_llm: false
|
| 22 |
+
tune_visual: false
|
| 23 |
+
tune_projector: false
|
| 24 |
+
tune_diffusion_model: false
|
| 25 |
+
losses:
|
| 26 |
+
mgd_enabled: true
|
| 27 |
+
mgd_fm_loss_weight: 0.0
|
| 28 |
+
mgd_loss_weight: 1.0
|
| 29 |
+
mgd_sequence_hidden_dim: 512
|
| 30 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 31 |
+
mgd_use_cosine_loss: true
|
| 32 |
+
mgd_use_mse_loss: false
|
| 33 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 34 |
+
max_steps: 30000
|
| 35 |
+
save_steps: 0
|
| 36 |
+
trainable:
|
| 37 |
+
tune_llm: false
|
| 38 |
+
tune_visual: false
|
| 39 |
+
tune_projector: true
|
| 40 |
+
tune_diffusion_model: true
|
| 41 |
+
losses:
|
| 42 |
+
mgd_enabled: true
|
| 43 |
+
mgd_fm_loss_weight: 1.0
|
| 44 |
+
mgd_loss_weight: 0.5
|
| 45 |
+
mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
|
| 46 |
+
mgd_use_cosine_loss: true
|
| 47 |
+
mgd_use_mse_loss: false
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/phase2/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.4/resolved_config.yaml
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: mgd_v2_mask_ratio_ablation
|
| 2 |
+
policy_type: groot_mgd
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation
|
| 5 |
+
sweep:
|
| 6 |
+
mgd_token_mask_ratio:
|
| 7 |
+
- 0.4
|
| 8 |
+
dataset:
|
| 9 |
+
dataset_soup: my_atomic26_human
|
| 10 |
+
training:
|
| 11 |
+
num_gpus: 2
|
| 12 |
+
batch_size: 64
|
| 13 |
+
seed: 42
|
| 14 |
+
model: null
|
| 15 |
+
phases:
|
| 16 |
+
- name: phase2_mgd_only
|
| 17 |
+
max_steps: 30000
|
| 18 |
+
save_steps: 0
|
| 19 |
+
trainable:
|
| 20 |
+
preset: processing_line_only
|
| 21 |
+
tune_llm: false
|
| 22 |
+
tune_visual: false
|
| 23 |
+
tune_projector: false
|
| 24 |
+
tune_diffusion_model: false
|
| 25 |
+
losses:
|
| 26 |
+
mgd_enabled: true
|
| 27 |
+
mgd_fm_loss_weight: 0.0
|
| 28 |
+
mgd_loss_weight: 1.0
|
| 29 |
+
mgd_sequence_hidden_dim: 512
|
| 30 |
+
mgd_token_mask_ratio: 0.4
|
| 31 |
+
mgd_use_cosine_loss: true
|
| 32 |
+
mgd_use_mse_loss: false
|
| 33 |
+
- name: phase3_fm_mgd_fixed_0p5
|
| 34 |
+
max_steps: 30000
|
| 35 |
+
save_steps: 0
|
| 36 |
+
trainable:
|
| 37 |
+
tune_llm: false
|
| 38 |
+
tune_visual: false
|
| 39 |
+
tune_projector: true
|
| 40 |
+
tune_diffusion_model: true
|
| 41 |
+
losses:
|
| 42 |
+
mgd_enabled: true
|
| 43 |
+
mgd_fm_loss_weight: 1.0
|
| 44 |
+
mgd_loss_weight: 0.5
|
| 45 |
+
mgd_token_mask_ratio: 0.4
|
| 46 |
+
mgd_use_cosine_loss: true
|
| 47 |
+
mgd_use_mse_loss: false
|
| 48 |
+
resolved_sweep:
|
| 49 |
+
mgd_token_mask_ratio: 0.4
|
rkd_v2_1/rkd_v2_1_angle/default/phase2/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
rkd_v2_1/rkd_v2_1_angle/default/resolved_config.yaml
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: rkd_v2_1_angle_phase2_phase3_aux_fixed_0p5
|
| 2 |
+
policy_type: groot_rkd_v2
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/rkd_v2_1/rkd_v2_1_angle
|
| 5 |
+
dataset:
|
| 6 |
+
dataset_soup: my_atomic26_human
|
| 7 |
+
training:
|
| 8 |
+
num_gpus: 1
|
| 9 |
+
batch_size: 128
|
| 10 |
+
seed: 42
|
| 11 |
+
model: null
|
| 12 |
+
phases:
|
| 13 |
+
- name: phase2_rkd_a_only
|
| 14 |
+
max_steps: 60000
|
| 15 |
+
save_steps: 30000
|
| 16 |
+
trainable:
|
| 17 |
+
preset: processing_line_only
|
| 18 |
+
tune_llm: false
|
| 19 |
+
tune_visual: false
|
| 20 |
+
tune_projector: false
|
| 21 |
+
tune_diffusion_model: false
|
| 22 |
+
losses:
|
| 23 |
+
rkd_enabled: true
|
| 24 |
+
rkd_fm_loss_weight: 0.0
|
| 25 |
+
rkd_loss_weight: 1.0
|
| 26 |
+
rkd_relation_mode: token_pair_mean
|
| 27 |
+
rkd_loss_type: angle
|
| 28 |
+
rkd_exclude_diagonal: true
|
| 29 |
+
- name: phase3_fm_rkd_a_fixed_0p5
|
| 30 |
+
max_steps: 80000
|
| 31 |
+
save_steps: 20000
|
| 32 |
+
trainable:
|
| 33 |
+
tune_llm: false
|
| 34 |
+
tune_visual: false
|
| 35 |
+
tune_projector: true
|
| 36 |
+
tune_diffusion_model: true
|
| 37 |
+
losses:
|
| 38 |
+
rkd_enabled: true
|
| 39 |
+
rkd_fm_loss_weight: 1.0
|
| 40 |
+
rkd_loss_weight: 0.5
|
| 41 |
+
rkd_relation_mode: token_pair_mean
|
| 42 |
+
rkd_loss_type: angle
|
| 43 |
+
rkd_exclude_diagonal: true
|
| 44 |
+
resolved_sweep: {}
|
rkd_v2_1/rkd_v2_1_angle/default/rkd_v2_1_a_stepup.yaml
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: rkd_v2_1_angle_phase2_phase3_aux_fixed_0p5
|
| 2 |
+
policy_type: groot_rkd_v2
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/rkd_v2_1/rkd_v2_1_angle
|
| 5 |
+
dataset:
|
| 6 |
+
dataset_soup: my_atomic26_human
|
| 7 |
+
training:
|
| 8 |
+
num_gpus: 1
|
| 9 |
+
batch_size: 128
|
| 10 |
+
seed: 42
|
| 11 |
+
model: null
|
| 12 |
+
phases:
|
| 13 |
+
- name: phase2_rkd_a_only
|
| 14 |
+
max_steps: 60000
|
| 15 |
+
save_steps: 30000
|
| 16 |
+
trainable:
|
| 17 |
+
preset: processing_line_only
|
| 18 |
+
tune_llm: false
|
| 19 |
+
tune_visual: false
|
| 20 |
+
tune_projector: false
|
| 21 |
+
tune_diffusion_model: false
|
| 22 |
+
losses:
|
| 23 |
+
rkd_enabled: true
|
| 24 |
+
rkd_fm_loss_weight: 0.0
|
| 25 |
+
rkd_loss_weight: 1.0
|
| 26 |
+
rkd_relation_mode: token_pair_mean
|
| 27 |
+
rkd_loss_type: angle
|
| 28 |
+
rkd_exclude_diagonal: true
|
| 29 |
+
- name: phase3_fm_rkd_a_fixed_0p5
|
| 30 |
+
max_steps: 80000
|
| 31 |
+
save_steps: 20000
|
| 32 |
+
trainable:
|
| 33 |
+
tune_llm: false
|
| 34 |
+
tune_visual: false
|
| 35 |
+
tune_projector: true
|
| 36 |
+
tune_diffusion_model: true
|
| 37 |
+
losses:
|
| 38 |
+
rkd_enabled: true
|
| 39 |
+
rkd_fm_loss_weight: 1.0
|
| 40 |
+
rkd_loss_weight: 0.5
|
| 41 |
+
rkd_relation_mode: token_pair_mean
|
| 42 |
+
rkd_loss_type: angle
|
| 43 |
+
rkd_exclude_diagonal: true
|
rkd_v2_1/rkd_v2_1_distance_angle/default/phase2/experiment_cfg/metadata.json
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"new_embodiment": {
|
| 3 |
+
"statistics": {
|
| 4 |
+
"state": {
|
| 5 |
+
"base_position": {
|
| 6 |
+
"max": [
|
| 7 |
+
7.3139495849609375,
|
| 8 |
+
0.4876587688922882,
|
| 9 |
+
0.7196521759033203
|
| 10 |
+
],
|
| 11 |
+
"min": [
|
| 12 |
+
-4.94293737411499,
|
| 13 |
+
-6.890198230743408,
|
| 14 |
+
0.6996827125549316
|
| 15 |
+
],
|
| 16 |
+
"mean": [
|
| 17 |
+
2.749288365724937,
|
| 18 |
+
-2.0455216239909983,
|
| 19 |
+
0.700532551020533
|
| 20 |
+
],
|
| 21 |
+
"std": [
|
| 22 |
+
1.6786295897046262,
|
| 23 |
+
1.312897969434244,
|
| 24 |
+
0.00133751758906258
|
| 25 |
+
],
|
| 26 |
+
"q01": [
|
| 27 |
+
-1.4241907881148037,
|
| 28 |
+
-5.1716547935950885,
|
| 29 |
+
0.6999670898768614
|
| 30 |
+
],
|
| 31 |
+
"q99": [
|
| 32 |
+
6.198111545831555,
|
| 33 |
+
-0.5873531334104489,
|
| 34 |
+
0.7038833014242587
|
| 35 |
+
]
|
| 36 |
+
},
|
| 37 |
+
"base_rotation": {
|
| 38 |
+
"max": [
|
| 39 |
+
0.0,
|
| 40 |
+
0.0,
|
| 41 |
+
1.0,
|
| 42 |
+
1.0
|
| 43 |
+
],
|
| 44 |
+
"min": [
|
| 45 |
+
0.0,
|
| 46 |
+
0.0,
|
| 47 |
+
-1.0,
|
| 48 |
+
0.0
|
| 49 |
+
],
|
| 50 |
+
"mean": [
|
| 51 |
+
0.0,
|
| 52 |
+
0.0,
|
| 53 |
+
0.23084999587450247,
|
| 54 |
+
0.6238431088581847
|
| 55 |
+
],
|
| 56 |
+
"std": [
|
| 57 |
+
0.0,
|
| 58 |
+
0.0,
|
| 59 |
+
0.6556404371644047,
|
| 60 |
+
0.3572252105732042
|
| 61 |
+
],
|
| 62 |
+
"q01": [
|
| 63 |
+
0.0,
|
| 64 |
+
0.0,
|
| 65 |
+
-0.9999999999999999,
|
| 66 |
+
1.7482354593273349e-06
|
| 67 |
+
],
|
| 68 |
+
"q99": [
|
| 69 |
+
0.0,
|
| 70 |
+
0.0,
|
| 71 |
+
0.9999999999999999,
|
| 72 |
+
0.9999999999999999
|
| 73 |
+
]
|
| 74 |
+
},
|
| 75 |
+
"end_effector_position_relative": {
|
| 76 |
+
"max": [
|
| 77 |
+
0.9014528393745422,
|
| 78 |
+
0.8003877401351929,
|
| 79 |
+
0.9829942584037781
|
| 80 |
+
],
|
| 81 |
+
"min": [
|
| 82 |
+
-0.37471598386764526,
|
| 83 |
+
-0.8472502827644348,
|
| 84 |
+
-0.25070279836654663
|
| 85 |
+
],
|
| 86 |
+
"mean": [
|
| 87 |
+
0.28667332601863704,
|
| 88 |
+
-0.038315473170876704,
|
| 89 |
+
0.45974658906777444
|
| 90 |
+
],
|
| 91 |
+
"std": [
|
| 92 |
+
0.17067058828305154,
|
| 93 |
+
0.23569952037784195,
|
| 94 |
+
0.21950702354181487
|
| 95 |
+
],
|
| 96 |
+
"q01": [
|
| 97 |
+
0.015172948963784228,
|
| 98 |
+
-0.44450502063220615,
|
| 99 |
+
0.23314444103329338
|
| 100 |
+
],
|
| 101 |
+
"q99": [
|
| 102 |
+
0.5609069689572412,
|
| 103 |
+
0.38454179128009547,
|
| 104 |
+
0.7028102736473852
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
"end_effector_rotation_relative": {
|
| 108 |
+
"max": [
|
| 109 |
+
0.9999998807907104,
|
| 110 |
+
0.9984455108642578,
|
| 111 |
+
0.9434727430343628,
|
| 112 |
+
0.9062229990959167
|
| 113 |
+
],
|
| 114 |
+
"min": [
|
| 115 |
+
-0.9999930262565613,
|
| 116 |
+
-0.9988337755203247,
|
| 117 |
+
-0.9618149995803833,
|
| 118 |
+
2.1872274658107926e-07
|
| 119 |
+
],
|
| 120 |
+
"mean": [
|
| 121 |
+
-0.25672297401336,
|
| 122 |
+
0.024425335806687216,
|
| 123 |
+
-0.08740977164483847,
|
| 124 |
+
0.16608739644018397
|
| 125 |
+
],
|
| 126 |
+
"std": [
|
| 127 |
+
0.7932585208392383,
|
| 128 |
+
0.3102260195580416,
|
| 129 |
+
0.3772472576315148,
|
| 130 |
+
0.17451959300158773
|
| 131 |
+
],
|
| 132 |
+
"q01": [
|
| 133 |
+
-0.99379006987882,
|
| 134 |
+
-0.5478102050038828,
|
| 135 |
+
-0.6101777952971636,
|
| 136 |
+
0.002274085014134748
|
| 137 |
+
],
|
| 138 |
+
"q99": [
|
| 139 |
+
0.8804856038896288,
|
| 140 |
+
0.5910611092873037,
|
| 141 |
+
0.5216058042993562,
|
| 142 |
+
0.5231965974012255
|
| 143 |
+
]
|
| 144 |
+
},
|
| 145 |
+
"gripper_qpos": {
|
| 146 |
+
"max": [
|
| 147 |
+
0.055869169533252716,
|
| 148 |
+
0.010916369967162609
|
| 149 |
+
],
|
| 150 |
+
"min": [
|
| 151 |
+
-0.011436971835792065,
|
| 152 |
+
-0.05664053186774254
|
| 153 |
+
],
|
| 154 |
+
"mean": [
|
| 155 |
+
0.031651551516052555,
|
| 156 |
+
-0.03161481387920421
|
| 157 |
+
],
|
| 158 |
+
"std": [
|
| 159 |
+
0.013119769222217526,
|
| 160 |
+
0.01306078225192402
|
| 161 |
+
],
|
| 162 |
+
"q01": [
|
| 163 |
+
0.0060999246446777336,
|
| 164 |
+
-0.04062601653964198
|
| 165 |
+
],
|
| 166 |
+
"q99": [
|
| 167 |
+
0.04054891621517283,
|
| 168 |
+
-0.0059163567113456345
|
| 169 |
+
]
|
| 170 |
+
}
|
| 171 |
+
},
|
| 172 |
+
"action": {
|
| 173 |
+
"base_motion": {
|
| 174 |
+
"max": [
|
| 175 |
+
1.0,
|
| 176 |
+
1.0,
|
| 177 |
+
1.0,
|
| 178 |
+
0.0
|
| 179 |
+
],
|
| 180 |
+
"min": [
|
| 181 |
+
-1.0,
|
| 182 |
+
-1.0,
|
| 183 |
+
-1.0,
|
| 184 |
+
0.0
|
| 185 |
+
],
|
| 186 |
+
"mean": [
|
| 187 |
+
0.008144896753718373,
|
| 188 |
+
-0.00018893135146078263,
|
| 189 |
+
-0.0008739062727221845,
|
| 190 |
+
0.0
|
| 191 |
+
],
|
| 192 |
+
"std": [
|
| 193 |
+
0.11030045424364411,
|
| 194 |
+
0.10148082570313594,
|
| 195 |
+
0.08975698373467506,
|
| 196 |
+
0.0
|
| 197 |
+
],
|
| 198 |
+
"q01": [
|
| 199 |
+
-0.07287373067581518,
|
| 200 |
+
-0.09446994199569948,
|
| 201 |
+
-0.07702818259249572,
|
| 202 |
+
0.0
|
| 203 |
+
],
|
| 204 |
+
"q99": [
|
| 205 |
+
0.04485485414407898,
|
| 206 |
+
0.09626941605965177,
|
| 207 |
+
0.07610665906122742,
|
| 208 |
+
0.0
|
| 209 |
+
]
|
| 210 |
+
},
|
| 211 |
+
"control_mode": {
|
| 212 |
+
"max": [
|
| 213 |
+
1.0
|
| 214 |
+
],
|
| 215 |
+
"min": [
|
| 216 |
+
-1.0
|
| 217 |
+
],
|
| 218 |
+
"mean": [
|
| 219 |
+
-0.9221974313425939
|
| 220 |
+
],
|
| 221 |
+
"std": [
|
| 222 |
+
0.38670194342612674
|
| 223 |
+
],
|
| 224 |
+
"q01": [
|
| 225 |
+
-1.0
|
| 226 |
+
],
|
| 227 |
+
"q99": [
|
| 228 |
+
-0.384693946883099
|
| 229 |
+
]
|
| 230 |
+
},
|
| 231 |
+
"end_effector_position": {
|
| 232 |
+
"max": [
|
| 233 |
+
1.0,
|
| 234 |
+
1.0,
|
| 235 |
+
1.0
|
| 236 |
+
],
|
| 237 |
+
"min": [
|
| 238 |
+
-1.0,
|
| 239 |
+
-1.0,
|
| 240 |
+
-1.0
|
| 241 |
+
],
|
| 242 |
+
"mean": [
|
| 243 |
+
0.01976129923017491,
|
| 244 |
+
-0.020870631344490034,
|
| 245 |
+
-0.05993476517349181
|
| 246 |
+
],
|
| 247 |
+
"std": [
|
| 248 |
+
0.4347024990804053,
|
| 249 |
+
0.4221662976569997,
|
| 250 |
+
0.3845506843044066
|
| 251 |
+
],
|
| 252 |
+
"q01": [
|
| 253 |
+
-0.8835792939767807,
|
| 254 |
+
-0.9280919251713778,
|
| 255 |
+
-0.783361071470956
|
| 256 |
+
],
|
| 257 |
+
"q99": [
|
| 258 |
+
0.7622084438449307,
|
| 259 |
+
0.8625125252609077,
|
| 260 |
+
0.7470974753250706
|
| 261 |
+
]
|
| 262 |
+
},
|
| 263 |
+
"end_effector_rotation": {
|
| 264 |
+
"max": [
|
| 265 |
+
1.0,
|
| 266 |
+
1.0,
|
| 267 |
+
1.0
|
| 268 |
+
],
|
| 269 |
+
"min": [
|
| 270 |
+
-1.0,
|
| 271 |
+
-1.0,
|
| 272 |
+
-1.0
|
| 273 |
+
],
|
| 274 |
+
"mean": [
|
| 275 |
+
0.008089673068544335,
|
| 276 |
+
-0.026498254627927348,
|
| 277 |
+
0.0029083551743059005
|
| 278 |
+
],
|
| 279 |
+
"std": [
|
| 280 |
+
0.11216734503733773,
|
| 281 |
+
0.13206002613798629,
|
| 282 |
+
0.12615870412692864
|
| 283 |
+
],
|
| 284 |
+
"q01": [
|
| 285 |
+
-0.2814627972846672,
|
| 286 |
+
-0.4342386516115439,
|
| 287 |
+
-0.34489755628340574
|
| 288 |
+
],
|
| 289 |
+
"q99": [
|
| 290 |
+
0.31879353266939386,
|
| 291 |
+
0.28411797683220563,
|
| 292 |
+
0.3597718671669894
|
| 293 |
+
]
|
| 294 |
+
},
|
| 295 |
+
"gripper_close": {
|
| 296 |
+
"max": [
|
| 297 |
+
1.0
|
| 298 |
+
],
|
| 299 |
+
"min": [
|
| 300 |
+
-1.0
|
| 301 |
+
],
|
| 302 |
+
"mean": [
|
| 303 |
+
-0.3824228874769045
|
| 304 |
+
],
|
| 305 |
+
"std": [
|
| 306 |
+
0.9240015096469786
|
| 307 |
+
],
|
| 308 |
+
"q01": [
|
| 309 |
+
-1.0
|
| 310 |
+
],
|
| 311 |
+
"q99": [
|
| 312 |
+
0.5597272305813592
|
| 313 |
+
]
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"modalities": {
|
| 318 |
+
"video": {
|
| 319 |
+
"robot0_eye_in_hand": {
|
| 320 |
+
"resolution": [
|
| 321 |
+
256,
|
| 322 |
+
256
|
| 323 |
+
],
|
| 324 |
+
"channels": 3,
|
| 325 |
+
"fps": 20.0
|
| 326 |
+
},
|
| 327 |
+
"robot0_agentview_left": {
|
| 328 |
+
"resolution": [
|
| 329 |
+
256,
|
| 330 |
+
256
|
| 331 |
+
],
|
| 332 |
+
"channels": 3,
|
| 333 |
+
"fps": 20.0
|
| 334 |
+
},
|
| 335 |
+
"robot0_agentview_right": {
|
| 336 |
+
"resolution": [
|
| 337 |
+
256,
|
| 338 |
+
256
|
| 339 |
+
],
|
| 340 |
+
"channels": 3,
|
| 341 |
+
"fps": 20.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
"state": {
|
| 345 |
+
"base_position": {
|
| 346 |
+
"absolute": true,
|
| 347 |
+
"rotation_type": null,
|
| 348 |
+
"shape": [
|
| 349 |
+
3
|
| 350 |
+
],
|
| 351 |
+
"continuous": true
|
| 352 |
+
},
|
| 353 |
+
"base_rotation": {
|
| 354 |
+
"absolute": true,
|
| 355 |
+
"rotation_type": "quaternion",
|
| 356 |
+
"shape": [
|
| 357 |
+
4
|
| 358 |
+
],
|
| 359 |
+
"continuous": true
|
| 360 |
+
},
|
| 361 |
+
"end_effector_position_relative": {
|
| 362 |
+
"absolute": true,
|
| 363 |
+
"rotation_type": null,
|
| 364 |
+
"shape": [
|
| 365 |
+
3
|
| 366 |
+
],
|
| 367 |
+
"continuous": true
|
| 368 |
+
},
|
| 369 |
+
"end_effector_rotation_relative": {
|
| 370 |
+
"absolute": true,
|
| 371 |
+
"rotation_type": "quaternion",
|
| 372 |
+
"shape": [
|
| 373 |
+
4
|
| 374 |
+
],
|
| 375 |
+
"continuous": true
|
| 376 |
+
},
|
| 377 |
+
"gripper_qpos": {
|
| 378 |
+
"absolute": true,
|
| 379 |
+
"rotation_type": null,
|
| 380 |
+
"shape": [
|
| 381 |
+
2
|
| 382 |
+
],
|
| 383 |
+
"continuous": true
|
| 384 |
+
}
|
| 385 |
+
},
|
| 386 |
+
"action": {
|
| 387 |
+
"base_motion": {
|
| 388 |
+
"absolute": true,
|
| 389 |
+
"rotation_type": null,
|
| 390 |
+
"shape": [
|
| 391 |
+
4
|
| 392 |
+
],
|
| 393 |
+
"continuous": true
|
| 394 |
+
},
|
| 395 |
+
"control_mode": {
|
| 396 |
+
"absolute": true,
|
| 397 |
+
"rotation_type": null,
|
| 398 |
+
"shape": [
|
| 399 |
+
1
|
| 400 |
+
],
|
| 401 |
+
"continuous": true
|
| 402 |
+
},
|
| 403 |
+
"end_effector_position": {
|
| 404 |
+
"absolute": true,
|
| 405 |
+
"rotation_type": null,
|
| 406 |
+
"shape": [
|
| 407 |
+
3
|
| 408 |
+
],
|
| 409 |
+
"continuous": true
|
| 410 |
+
},
|
| 411 |
+
"end_effector_rotation": {
|
| 412 |
+
"absolute": true,
|
| 413 |
+
"rotation_type": "axis_angle",
|
| 414 |
+
"shape": [
|
| 415 |
+
3
|
| 416 |
+
],
|
| 417 |
+
"continuous": true
|
| 418 |
+
},
|
| 419 |
+
"gripper_close": {
|
| 420 |
+
"absolute": true,
|
| 421 |
+
"rotation_type": null,
|
| 422 |
+
"shape": [
|
| 423 |
+
1
|
| 424 |
+
],
|
| 425 |
+
"continuous": true
|
| 426 |
+
}
|
| 427 |
+
}
|
| 428 |
+
},
|
| 429 |
+
"embodiment_tag": "new_embodiment"
|
| 430 |
+
}
|
| 431 |
+
}
|
rkd_v2_1/rkd_v2_1_distance_angle/default/resolved_config.yaml
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: rkd_v2_1_distance_angle_phase2_phase3_aux_fixed_0p5
|
| 2 |
+
policy_type: groot_rkd_v2
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/rkd_v2_1/rkd_v2_1_distance_angle
|
| 5 |
+
dataset:
|
| 6 |
+
dataset_soup: my_atomic26_human
|
| 7 |
+
training:
|
| 8 |
+
num_gpus: 1
|
| 9 |
+
batch_size: 128
|
| 10 |
+
seed: 42
|
| 11 |
+
model: null
|
| 12 |
+
phases:
|
| 13 |
+
- name: phase2_rkd_da_only
|
| 14 |
+
max_steps: 60000
|
| 15 |
+
save_steps: 30000
|
| 16 |
+
trainable:
|
| 17 |
+
preset: processing_line_only
|
| 18 |
+
tune_llm: false
|
| 19 |
+
tune_visual: false
|
| 20 |
+
tune_projector: false
|
| 21 |
+
tune_diffusion_model: false
|
| 22 |
+
losses:
|
| 23 |
+
rkd_enabled: true
|
| 24 |
+
rkd_fm_loss_weight: 0.0
|
| 25 |
+
rkd_loss_weight: 1.0
|
| 26 |
+
rkd_relation_mode: token_pair_mean
|
| 27 |
+
rkd_loss_type: distance_angle
|
| 28 |
+
rkd_distance_loss_weight: 1.0
|
| 29 |
+
rkd_angle_loss_weight: 2.0
|
| 30 |
+
rkd_exclude_diagonal: true
|
| 31 |
+
- name: phase3_fm_rkd_da_fixed_0p5
|
| 32 |
+
max_steps: 80000
|
| 33 |
+
save_steps: 20000
|
| 34 |
+
trainable:
|
| 35 |
+
tune_llm: false
|
| 36 |
+
tune_visual: false
|
| 37 |
+
tune_projector: true
|
| 38 |
+
tune_diffusion_model: true
|
| 39 |
+
losses:
|
| 40 |
+
rkd_enabled: true
|
| 41 |
+
rkd_fm_loss_weight: 1.0
|
| 42 |
+
rkd_loss_weight: 0.5
|
| 43 |
+
rkd_relation_mode: token_pair_mean
|
| 44 |
+
rkd_loss_type: distance_angle
|
| 45 |
+
rkd_distance_loss_weight: 1.0
|
| 46 |
+
rkd_angle_loss_weight: 2.0
|
| 47 |
+
rkd_exclude_diagonal: true
|
| 48 |
+
resolved_sweep: {}
|
rkd_v2_1/rkd_v2_1_distance_angle/default/rkd_v2_1_da_stepup.yaml
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
experiment_name: rkd_v2_1_distance_angle_phase2_phase3_aux_fixed_0p5
|
| 2 |
+
policy_type: groot_rkd_v2
|
| 3 |
+
base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
|
| 4 |
+
output_root: ${HOME}/groot_robocasa/robocasa_v2/rkd_v2_1/rkd_v2_1_distance_angle
|
| 5 |
+
dataset:
|
| 6 |
+
dataset_soup: my_atomic26_human
|
| 7 |
+
training:
|
| 8 |
+
num_gpus: 1
|
| 9 |
+
batch_size: 128
|
| 10 |
+
seed: 42
|
| 11 |
+
model: null
|
| 12 |
+
phases:
|
| 13 |
+
- name: phase2_rkd_da_only
|
| 14 |
+
max_steps: 60000
|
| 15 |
+
save_steps: 30000
|
| 16 |
+
trainable:
|
| 17 |
+
preset: processing_line_only
|
| 18 |
+
tune_llm: false
|
| 19 |
+
tune_visual: false
|
| 20 |
+
tune_projector: false
|
| 21 |
+
tune_diffusion_model: false
|
| 22 |
+
losses:
|
| 23 |
+
rkd_enabled: true
|
| 24 |
+
rkd_fm_loss_weight: 0.0
|
| 25 |
+
rkd_loss_weight: 1.0
|
| 26 |
+
rkd_relation_mode: token_pair_mean
|
| 27 |
+
rkd_loss_type: distance_angle
|
| 28 |
+
rkd_distance_loss_weight: 1.0
|
| 29 |
+
rkd_angle_loss_weight: 2.0
|
| 30 |
+
rkd_exclude_diagonal: true
|
| 31 |
+
- name: phase3_fm_rkd_da_fixed_0p5
|
| 32 |
+
max_steps: 80000
|
| 33 |
+
save_steps: 20000
|
| 34 |
+
trainable:
|
| 35 |
+
tune_llm: false
|
| 36 |
+
tune_visual: false
|
| 37 |
+
tune_projector: true
|
| 38 |
+
tune_diffusion_model: true
|
| 39 |
+
losses:
|
| 40 |
+
rkd_enabled: true
|
| 41 |
+
rkd_fm_loss_weight: 1.0
|
| 42 |
+
rkd_loss_weight: 0.5
|
| 43 |
+
rkd_relation_mode: token_pair_mean
|
| 44 |
+
rkd_loss_type: distance_angle
|
| 45 |
+
rkd_distance_loss_weight: 1.0
|
| 46 |
+
rkd_angle_loss_weight: 2.0
|
| 47 |
+
rkd_exclude_diagonal: true
|