Whalswp commited on
Commit
3886dfe
·
verified ·
1 Parent(s): 2fddde6

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +1 -0
  2. experiment_cfg/processing_line_only/retrain_best/rkd_seed44_phase3_from_hf.log +0 -0
  3. experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p4_train.log +0 -0
  4. experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_gpu01_lane.log +3 -0
  5. experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_gpu23_lane.log +0 -0
  6. processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/config.json +73 -0
  7. processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/model-00001-of-00002.safetensors +3 -0
  8. processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/model-00002-of-00002.safetensors +3 -0
  9. processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/model.safetensors.index.json +0 -0
  10. processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/training_args.bin +3 -0
  11. processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/resolved_config.yaml +2 -18
  12. processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/rkd_retrain_best2_phase3_only.yaml +30 -0
  13. processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/config.json +73 -0
  14. processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/model-00001-of-00002.safetensors +3 -0
  15. processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/model-00002-of-00002.safetensors +3 -0
  16. processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/model.safetensors.index.json +0 -0
  17. processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/training_args.bin +3 -0
  18. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/model-00001-of-00002.safetensors +3 -0
  19. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/model-00002-of-00002.safetensors +3 -0
  20. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/training_args.bin +3 -0
  21. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/model-00001-of-00002.safetensors +3 -0
  22. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/model-00002-of-00002.safetensors +3 -0
  23. processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/training_args.bin +3 -0
  24. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_mse.yaml +47 -0
  25. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/config.json +78 -0
  26. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/experiment_cfg/metadata.json +431 -0
  27. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/model-00001-of-00002.safetensors +3 -0
  28. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/model-00002-of-00002.safetensors +3 -0
  29. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/model.safetensors.index.json +0 -0
  30. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/trainer_state.json +0 -0
  31. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/training_args.bin +3 -0
  32. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/config.json +78 -0
  33. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/experiment_cfg/metadata.json +431 -0
  34. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/model-00001-of-00002.safetensors +3 -0
  35. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/model-00002-of-00002.safetensors +3 -0
  36. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/model.safetensors.index.json +0 -0
  37. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/trainer_state.json +0 -0
  38. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/training_args.bin +3 -0
  39. processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/resolved_config.yaml +49 -0
  40. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/model-00001-of-00002.safetensors +3 -0
  41. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/model-00002-of-00002.safetensors +3 -0
  42. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/training_args.bin +3 -0
  43. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/model-00001-of-00002.safetensors +3 -0
  44. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/model-00002-of-00002.safetensors +3 -0
  45. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/trainer_state.json +0 -0
  46. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/training_args.bin +3 -0
  47. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/config.json +78 -0
  48. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/experiment_cfg/metadata.json +431 -0
  49. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/model-00001-of-00002.safetensors +3 -0
  50. processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/model-00002-of-00002.safetensors +3 -0
.gitattributes CHANGED
@@ -36,3 +36,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
36
  RKD_Processing_line_Only/train.log filter=lfs diff=lfs merge=lfs -text
37
  VLM_Only/RKD_Raw_VLM/train.log filter=lfs diff=lfs merge=lfs -text
38
  VLM_Only/MGD_Raw_VLM/train.log filter=lfs diff=lfs merge=lfs -text
 
 
36
  RKD_Processing_line_Only/train.log filter=lfs diff=lfs merge=lfs -text
37
  VLM_Only/RKD_Raw_VLM/train.log filter=lfs diff=lfs merge=lfs -text
38
  VLM_Only/MGD_Raw_VLM/train.log filter=lfs diff=lfs merge=lfs -text
39
+ experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_gpu01_lane.log filter=lfs diff=lfs merge=lfs -text
experiment_cfg/processing_line_only/retrain_best/rkd_seed44_phase3_from_hf.log ADDED
The diff for this file is too large to render. See raw diff
 
experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation_0p4_train.log ADDED
The diff for this file is too large to render. See raw diff
 
experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_gpu01_lane.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9a245b8243ddcf5da916b9d670ccb3744c8659bb323295269fb0d607782ae851
3
+ size 18113927
experiment_cfg/processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_gpu23_lane.log ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/config.json ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_RKD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "model_dtype": "float32",
61
+ "model_type": "gr00t_n1_5",
62
+ "rkd_action_temp": 0.1,
63
+ "rkd_enabled": true,
64
+ "rkd_fm_loss_weight": 1.0,
65
+ "rkd_loss_weight": 0.5,
66
+ "rkd_loss_weight_end": 0.0,
67
+ "rkd_loss_weight_schedule": null,
68
+ "rkd_loss_weight_start": 0.1,
69
+ "rkd_relation_mode": "token_pair_mean",
70
+ "rkd_vlm_temp": 0.1,
71
+ "torch_dtype": "bfloat16",
72
+ "transformers_version": "4.51.3"
73
+ }
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5d907a2ab38eeafc2144f692975a2ea8743fee421f16867eece069e40c72d6ba
3
+ size 4999367032
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12aeab1fe2b8ad478d76dc85e59332af134f3867496c0bde7084f3496497fc8e
3
+ size 2586705312
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase3/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ed18136a5e063abb170e8d790206843cf1594e52fb332edaee1e4510aa88ed25
3
+ size 5905
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/resolved_config.yaml CHANGED
@@ -1,6 +1,6 @@
1
- experiment_name: rkd_phase2_phase3_aux_fixed_0p5_retrain_seed44
2
  policy_type: groot_rkd
3
- base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
  output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only/retrain_best/rkd_seed44
5
  sweep:
6
  rkd_temp:
@@ -13,22 +13,6 @@ training:
13
  seed: 44
14
  model: null
15
  phases:
16
- - name: phase2_rkd_only
17
- max_steps: 30000
18
- save_steps: 0
19
- trainable:
20
- preset: processing_line_only
21
- tune_llm: false
22
- tune_visual: false
23
- tune_projector: false
24
- tune_diffusion_model: false
25
- losses:
26
- rkd_enabled: true
27
- rkd_fm_loss_weight: 0.0
28
- rkd_loss_weight: 1.0
29
- rkd_action_temp: 0.1
30
- rkd_vlm_temp: 0.1
31
- rkd_relation_mode: token_pair_mean
32
  - name: phase3_fm_rkd_fixed_0p5
33
  max_steps: 30000
34
  save_steps: 0
 
1
+ experiment_name: rkd_phase3_aux_fixed_0p5_retrain_seed44
2
  policy_type: groot_rkd
3
+ base_model_path: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase2
4
  output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only/retrain_best/rkd_seed44
5
  sweep:
6
  rkd_temp:
 
13
  seed: 44
14
  model: null
15
  phases:
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
  - name: phase3_fm_rkd_fixed_0p5
17
  max_steps: 30000
18
  save_steps: 0
processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/rkd_retrain_best2_phase3_only.yaml ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: rkd_phase3_aux_fixed_0p5_retrain_seed44
2
+ policy_type: groot_rkd
3
+ base_model_path: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only/retrain_best/rkd_seed44/rkd_temp_0.1/phase2
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only/retrain_best/rkd_seed44
5
+ sweep:
6
+ rkd_temp:
7
+ - 0.1
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 44
14
+ model: null
15
+ phases:
16
+ - name: phase3_fm_rkd_fixed_0p5
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ tune_llm: false
21
+ tune_visual: false
22
+ tune_projector: true
23
+ tune_diffusion_model: true
24
+ losses:
25
+ rkd_enabled: true
26
+ rkd_fm_loss_weight: 1.0
27
+ rkd_loss_weight: 0.5
28
+ rkd_action_temp: ${sweep.rkd_temp}
29
+ rkd_vlm_temp: ${sweep.rkd_temp}
30
+ rkd_relation_mode: token_pair_mean
processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/config.json ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_RKD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "model_dtype": "float32",
61
+ "model_type": "gr00t_n1_5",
62
+ "rkd_action_temp": 0.1,
63
+ "rkd_enabled": true,
64
+ "rkd_fm_loss_weight": 1.0,
65
+ "rkd_loss_weight": 0.5,
66
+ "rkd_loss_weight_end": 0.0,
67
+ "rkd_loss_weight_schedule": null,
68
+ "rkd_loss_weight_start": 0.1,
69
+ "rkd_relation_mode": "token_pair_mean",
70
+ "rkd_vlm_temp": 0.1,
71
+ "torch_dtype": "bfloat16",
72
+ "transformers_version": "4.51.3"
73
+ }
processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a20a4468bc793dfe2673af8caaca55d77d3b51b184d2bf7efc87502b3d2feb66
3
+ size 4999367032
processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3aaf86b6ddc18be35ff4a91b01930fb874f544c118ab380d1ae5d480aca419d4
3
+ size 2586705312
processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only/retrain_best/rkd_seed45/rkd_temp_0.1/phase3/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:36e2c84120c47dcf3d3b3f57a98dbd057cbe716bb9e4a9870f70f8c6ef606a2e
3
+ size 5905
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:98cf98087479b3e570d7fc66a0d397f42423009d07149e2cac936dbf1290a117
3
+ size 4999367032
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c6ee1f00a02297a9f359b7af9710d808ca7208f22399cca90e557629359184da
3
+ size 2643339748
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase2/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6d4617a50d34e2c44ef05433fea3c978b4f60fa0bbe0afb83e73e4a2ddd9607
3
+ size 5969
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7776b2ee616424ceee08de07a06227bb186644520814307016f9b550f46226d1
3
+ size 4999367032
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c236c8d4c3dcfbb24787ef44938f947fb2be5d973f70d9c0650b994b14b73113
3
+ size 2643339748
processing_line_only_v2/MGD_v2/mgd_v2_loss_cosine_mse/mgd_token_mask_ratio_0.3/phase3/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14e966cc29fc571161b938899aea45bb66cf7eff03db1f7263303a966ded550b
3
+ size 5969
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/mgd_v2_loss_mse.yaml ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_loss_mse
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_mse
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.3
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
31
+ mgd_use_cosine_loss: false
32
+ mgd_use_mse_loss: true
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: ${sweep.mgd_token_mask_ratio}
46
+ mgd_use_cosine_loss: false
47
+ mgd_use_mse_loss: true
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/config.json ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_MGD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "mgd_enabled": true,
61
+ "mgd_fm_loss_weight": 0.0,
62
+ "mgd_loss_weight": 1.0,
63
+ "mgd_loss_weight_end": 0.0,
64
+ "mgd_loss_weight_schedule": null,
65
+ "mgd_loss_weight_start": 0.05,
66
+ "mgd_pretrained_projector_path": null,
67
+ "mgd_sequence_hidden_dim": 512,
68
+ "mgd_target_dim": 512,
69
+ "mgd_target_pooling": "flatten",
70
+ "mgd_target_projection": "frozen_random",
71
+ "mgd_token_mask_ratio": 0.3,
72
+ "mgd_use_cosine_loss": false,
73
+ "mgd_use_mse_loss": true,
74
+ "model_dtype": "float32",
75
+ "model_type": "gr00t_n1_5",
76
+ "torch_dtype": "bfloat16",
77
+ "transformers_version": "4.51.3"
78
+ }
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:98cf98087479b3e570d7fc66a0d397f42423009d07149e2cac936dbf1290a117
3
+ size 4999367032
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d983182f2aeca91d08f6fcbb64f4a695f3d4b87e411c082768897249062d8281
3
+ size 2643339748
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/trainer_state.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase2/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:36b22b1e779cb8364f92308094b4e4181a65c15557d955edb39862363a6d3911
3
+ size 5969
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/config.json ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_MGD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "mgd_enabled": true,
61
+ "mgd_fm_loss_weight": 1.0,
62
+ "mgd_loss_weight": 0.5,
63
+ "mgd_loss_weight_end": 0.0,
64
+ "mgd_loss_weight_schedule": null,
65
+ "mgd_loss_weight_start": 0.05,
66
+ "mgd_pretrained_projector_path": null,
67
+ "mgd_sequence_hidden_dim": 512,
68
+ "mgd_target_dim": 512,
69
+ "mgd_target_pooling": "flatten",
70
+ "mgd_target_projection": "frozen_random",
71
+ "mgd_token_mask_ratio": 0.3,
72
+ "mgd_use_cosine_loss": false,
73
+ "mgd_use_mse_loss": true,
74
+ "model_dtype": "float32",
75
+ "model_type": "gr00t_n1_5",
76
+ "torch_dtype": "bfloat16",
77
+ "transformers_version": "4.51.3"
78
+ }
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2b01d8089bcfe5112dcbcb61d3a53975724cde989df4cdbf3fae480dd24f2e55
3
+ size 4999367032
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:accb04d1a46feada7cfcbbcca4135cbe828f5e84672ad66c2832a6bf2a11fe37
3
+ size 2643339748
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/trainer_state.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/phase3/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f54fd331f315b0079b8e5f938a05b82bc6faddf00045a45c9af2fcf22891d59c
3
+ size 5969
processing_line_only_v2/MGD_v2/mgd_v2_loss_mse/mgd_token_mask_ratio_0.3/resolved_config.yaml ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ experiment_name: mgd_v2_loss_mse
2
+ policy_type: groot_mgd
3
+ base_model_path: ${HOME}/groot_robocasa/Atomic26_baseline/checkpoint-60000
4
+ output_root: ${HOME}/groot_robocasa/robocasa_v2/processing_line_only_v2/MGD_v2/mgd_v2_loss_mse
5
+ sweep:
6
+ mgd_token_mask_ratio:
7
+ - 0.3
8
+ dataset:
9
+ dataset_soup: my_atomic26_human
10
+ training:
11
+ num_gpus: 2
12
+ batch_size: 64
13
+ seed: 42
14
+ model: null
15
+ phases:
16
+ - name: phase2_mgd_only
17
+ max_steps: 30000
18
+ save_steps: 0
19
+ trainable:
20
+ preset: processing_line_only
21
+ tune_llm: false
22
+ tune_visual: false
23
+ tune_projector: false
24
+ tune_diffusion_model: false
25
+ losses:
26
+ mgd_enabled: true
27
+ mgd_fm_loss_weight: 0.0
28
+ mgd_loss_weight: 1.0
29
+ mgd_sequence_hidden_dim: 512
30
+ mgd_token_mask_ratio: 0.3
31
+ mgd_use_cosine_loss: false
32
+ mgd_use_mse_loss: true
33
+ - name: phase3_fm_mgd_fixed_0p5
34
+ max_steps: 30000
35
+ save_steps: 0
36
+ trainable:
37
+ tune_llm: false
38
+ tune_visual: false
39
+ tune_projector: true
40
+ tune_diffusion_model: true
41
+ losses:
42
+ mgd_enabled: true
43
+ mgd_fm_loss_weight: 1.0
44
+ mgd_loss_weight: 0.5
45
+ mgd_token_mask_ratio: 0.3
46
+ mgd_use_cosine_loss: false
47
+ mgd_use_mse_loss: true
48
+ resolved_sweep:
49
+ mgd_token_mask_ratio: 0.3
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:98cf98087479b3e570d7fc66a0d397f42423009d07149e2cac936dbf1290a117
3
+ size 4999367032
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:195c04d96fc235692877602d231c57bb35500c91600df74936e5daf4e5adbe63
3
+ size 2643339748
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase2/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd4194b69281f2d668f850a7ed0f001b486597afbcae8d89aec953c507e540a2
3
+ size 5969
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:297ba5d932d6ec1bc6dd6ecc599015a406b97b1ebcb4a815397e60c8d48fa08d
3
+ size 4999367032
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b2d70c3f40e7be4d09538bde42ca534e9e1c222c358f9189eb025649ee07ebbf
3
+ size 2643339748
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/trainer_state.json ADDED
The diff for this file is too large to render. See raw diff
 
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.1/phase3/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e36e4efc7535170dfa141d25fdb685596fc171a8fe10bafd33f03dfb3c37f9b
3
+ size 5969
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/config.json ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "action_dim": 32,
3
+ "action_head_cfg": {
4
+ "action_dim": 32,
5
+ "action_horizon": 16,
6
+ "add_pos_embed": true,
7
+ "backbone_embedding_dim": 2048,
8
+ "diffusion_model_cfg": {
9
+ "attention_head_dim": 48,
10
+ "cross_attention_dim": 2048,
11
+ "dropout": 0.2,
12
+ "final_dropout": true,
13
+ "interleave_self_attention": true,
14
+ "norm_type": "ada_norm",
15
+ "num_attention_heads": 32,
16
+ "num_layers": 16,
17
+ "output_dim": 1024,
18
+ "positional_embeddings": null
19
+ },
20
+ "hidden_size": 1024,
21
+ "input_embedding_dim": 1536,
22
+ "max_action_dim": 32,
23
+ "max_state_dim": 64,
24
+ "model_dtype": "float32",
25
+ "noise_beta_alpha": 1.5,
26
+ "noise_beta_beta": 1.0,
27
+ "noise_s": 0.999,
28
+ "num_inference_timesteps": 4,
29
+ "num_target_vision_tokens": 32,
30
+ "num_timestep_buckets": 1000,
31
+ "tune_diffusion_model": true,
32
+ "tune_projector": true,
33
+ "use_vlln": true,
34
+ "vl_self_attention_cfg": {
35
+ "attention_head_dim": 64,
36
+ "dropout": 0.2,
37
+ "final_dropout": true,
38
+ "num_attention_heads": 32,
39
+ "num_layers": 4,
40
+ "positional_embeddings": null
41
+ }
42
+ },
43
+ "action_horizon": 16,
44
+ "architectures": [
45
+ "GR00T_N1_5_MGD"
46
+ ],
47
+ "attn_implementation": null,
48
+ "backbone_cfg": {
49
+ "eagle_path": "NVEagle/eagle_er-qwen3_1_7B-Siglip2_400M_stage1_5_128gpu_er_v7_1mlp_nops",
50
+ "load_bf16": false,
51
+ "project_to_dim": null,
52
+ "reproject_vision": false,
53
+ "select_layer": 12,
54
+ "tune_llm": false,
55
+ "tune_visual": true,
56
+ "use_flash_attention": true
57
+ },
58
+ "compute_dtype": "bfloat16",
59
+ "hidden_size": 2048,
60
+ "mgd_enabled": true,
61
+ "mgd_fm_loss_weight": 0.0,
62
+ "mgd_loss_weight": 1.0,
63
+ "mgd_loss_weight_end": 0.0,
64
+ "mgd_loss_weight_schedule": null,
65
+ "mgd_loss_weight_start": 0.05,
66
+ "mgd_pretrained_projector_path": null,
67
+ "mgd_sequence_hidden_dim": 512,
68
+ "mgd_target_dim": 512,
69
+ "mgd_target_pooling": "flatten",
70
+ "mgd_target_projection": "frozen_random",
71
+ "mgd_token_mask_ratio": 0.2,
72
+ "mgd_use_cosine_loss": true,
73
+ "mgd_use_mse_loss": false,
74
+ "model_dtype": "float32",
75
+ "model_type": "gr00t_n1_5",
76
+ "torch_dtype": "bfloat16",
77
+ "transformers_version": "4.51.3"
78
+ }
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/experiment_cfg/metadata.json ADDED
@@ -0,0 +1,431 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "new_embodiment": {
3
+ "statistics": {
4
+ "state": {
5
+ "base_position": {
6
+ "max": [
7
+ 7.3139495849609375,
8
+ 0.4876587688922882,
9
+ 0.7196521759033203
10
+ ],
11
+ "min": [
12
+ -4.94293737411499,
13
+ -6.890198230743408,
14
+ 0.6996827125549316
15
+ ],
16
+ "mean": [
17
+ 2.749288365724937,
18
+ -2.0455216239909983,
19
+ 0.700532551020533
20
+ ],
21
+ "std": [
22
+ 1.6786295897046262,
23
+ 1.312897969434244,
24
+ 0.00133751758906258
25
+ ],
26
+ "q01": [
27
+ -1.4241907881148037,
28
+ -5.1716547935950885,
29
+ 0.6999670898768614
30
+ ],
31
+ "q99": [
32
+ 6.198111545831555,
33
+ -0.5873531334104489,
34
+ 0.7038833014242587
35
+ ]
36
+ },
37
+ "base_rotation": {
38
+ "max": [
39
+ 0.0,
40
+ 0.0,
41
+ 1.0,
42
+ 1.0
43
+ ],
44
+ "min": [
45
+ 0.0,
46
+ 0.0,
47
+ -1.0,
48
+ 0.0
49
+ ],
50
+ "mean": [
51
+ 0.0,
52
+ 0.0,
53
+ 0.23084999587450247,
54
+ 0.6238431088581847
55
+ ],
56
+ "std": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.6556404371644047,
60
+ 0.3572252105732042
61
+ ],
62
+ "q01": [
63
+ 0.0,
64
+ 0.0,
65
+ -0.9999999999999999,
66
+ 1.7482354593273349e-06
67
+ ],
68
+ "q99": [
69
+ 0.0,
70
+ 0.0,
71
+ 0.9999999999999999,
72
+ 0.9999999999999999
73
+ ]
74
+ },
75
+ "end_effector_position_relative": {
76
+ "max": [
77
+ 0.9014528393745422,
78
+ 0.8003877401351929,
79
+ 0.9829942584037781
80
+ ],
81
+ "min": [
82
+ -0.37471598386764526,
83
+ -0.8472502827644348,
84
+ -0.25070279836654663
85
+ ],
86
+ "mean": [
87
+ 0.28667332601863704,
88
+ -0.038315473170876704,
89
+ 0.45974658906777444
90
+ ],
91
+ "std": [
92
+ 0.17067058828305154,
93
+ 0.23569952037784195,
94
+ 0.21950702354181487
95
+ ],
96
+ "q01": [
97
+ 0.015172948963784228,
98
+ -0.44450502063220615,
99
+ 0.23314444103329338
100
+ ],
101
+ "q99": [
102
+ 0.5609069689572412,
103
+ 0.38454179128009547,
104
+ 0.7028102736473852
105
+ ]
106
+ },
107
+ "end_effector_rotation_relative": {
108
+ "max": [
109
+ 0.9999998807907104,
110
+ 0.9984455108642578,
111
+ 0.9434727430343628,
112
+ 0.9062229990959167
113
+ ],
114
+ "min": [
115
+ -0.9999930262565613,
116
+ -0.9988337755203247,
117
+ -0.9618149995803833,
118
+ 2.1872274658107926e-07
119
+ ],
120
+ "mean": [
121
+ -0.25672297401336,
122
+ 0.024425335806687216,
123
+ -0.08740977164483847,
124
+ 0.16608739644018397
125
+ ],
126
+ "std": [
127
+ 0.7932585208392383,
128
+ 0.3102260195580416,
129
+ 0.3772472576315148,
130
+ 0.17451959300158773
131
+ ],
132
+ "q01": [
133
+ -0.99379006987882,
134
+ -0.5478102050038828,
135
+ -0.6101777952971636,
136
+ 0.002274085014134748
137
+ ],
138
+ "q99": [
139
+ 0.8804856038896288,
140
+ 0.5910611092873037,
141
+ 0.5216058042993562,
142
+ 0.5231965974012255
143
+ ]
144
+ },
145
+ "gripper_qpos": {
146
+ "max": [
147
+ 0.055869169533252716,
148
+ 0.010916369967162609
149
+ ],
150
+ "min": [
151
+ -0.011436971835792065,
152
+ -0.05664053186774254
153
+ ],
154
+ "mean": [
155
+ 0.031651551516052555,
156
+ -0.03161481387920421
157
+ ],
158
+ "std": [
159
+ 0.013119769222217526,
160
+ 0.01306078225192402
161
+ ],
162
+ "q01": [
163
+ 0.0060999246446777336,
164
+ -0.04062601653964198
165
+ ],
166
+ "q99": [
167
+ 0.04054891621517283,
168
+ -0.0059163567113456345
169
+ ]
170
+ }
171
+ },
172
+ "action": {
173
+ "base_motion": {
174
+ "max": [
175
+ 1.0,
176
+ 1.0,
177
+ 1.0,
178
+ 0.0
179
+ ],
180
+ "min": [
181
+ -1.0,
182
+ -1.0,
183
+ -1.0,
184
+ 0.0
185
+ ],
186
+ "mean": [
187
+ 0.008144896753718373,
188
+ -0.00018893135146078263,
189
+ -0.0008739062727221845,
190
+ 0.0
191
+ ],
192
+ "std": [
193
+ 0.11030045424364411,
194
+ 0.10148082570313594,
195
+ 0.08975698373467506,
196
+ 0.0
197
+ ],
198
+ "q01": [
199
+ -0.07287373067581518,
200
+ -0.09446994199569948,
201
+ -0.07702818259249572,
202
+ 0.0
203
+ ],
204
+ "q99": [
205
+ 0.04485485414407898,
206
+ 0.09626941605965177,
207
+ 0.07610665906122742,
208
+ 0.0
209
+ ]
210
+ },
211
+ "control_mode": {
212
+ "max": [
213
+ 1.0
214
+ ],
215
+ "min": [
216
+ -1.0
217
+ ],
218
+ "mean": [
219
+ -0.9221974313425939
220
+ ],
221
+ "std": [
222
+ 0.38670194342612674
223
+ ],
224
+ "q01": [
225
+ -1.0
226
+ ],
227
+ "q99": [
228
+ -0.384693946883099
229
+ ]
230
+ },
231
+ "end_effector_position": {
232
+ "max": [
233
+ 1.0,
234
+ 1.0,
235
+ 1.0
236
+ ],
237
+ "min": [
238
+ -1.0,
239
+ -1.0,
240
+ -1.0
241
+ ],
242
+ "mean": [
243
+ 0.01976129923017491,
244
+ -0.020870631344490034,
245
+ -0.05993476517349181
246
+ ],
247
+ "std": [
248
+ 0.4347024990804053,
249
+ 0.4221662976569997,
250
+ 0.3845506843044066
251
+ ],
252
+ "q01": [
253
+ -0.8835792939767807,
254
+ -0.9280919251713778,
255
+ -0.783361071470956
256
+ ],
257
+ "q99": [
258
+ 0.7622084438449307,
259
+ 0.8625125252609077,
260
+ 0.7470974753250706
261
+ ]
262
+ },
263
+ "end_effector_rotation": {
264
+ "max": [
265
+ 1.0,
266
+ 1.0,
267
+ 1.0
268
+ ],
269
+ "min": [
270
+ -1.0,
271
+ -1.0,
272
+ -1.0
273
+ ],
274
+ "mean": [
275
+ 0.008089673068544335,
276
+ -0.026498254627927348,
277
+ 0.0029083551743059005
278
+ ],
279
+ "std": [
280
+ 0.11216734503733773,
281
+ 0.13206002613798629,
282
+ 0.12615870412692864
283
+ ],
284
+ "q01": [
285
+ -0.2814627972846672,
286
+ -0.4342386516115439,
287
+ -0.34489755628340574
288
+ ],
289
+ "q99": [
290
+ 0.31879353266939386,
291
+ 0.28411797683220563,
292
+ 0.3597718671669894
293
+ ]
294
+ },
295
+ "gripper_close": {
296
+ "max": [
297
+ 1.0
298
+ ],
299
+ "min": [
300
+ -1.0
301
+ ],
302
+ "mean": [
303
+ -0.3824228874769045
304
+ ],
305
+ "std": [
306
+ 0.9240015096469786
307
+ ],
308
+ "q01": [
309
+ -1.0
310
+ ],
311
+ "q99": [
312
+ 0.5597272305813592
313
+ ]
314
+ }
315
+ }
316
+ },
317
+ "modalities": {
318
+ "video": {
319
+ "robot0_eye_in_hand": {
320
+ "resolution": [
321
+ 256,
322
+ 256
323
+ ],
324
+ "channels": 3,
325
+ "fps": 20.0
326
+ },
327
+ "robot0_agentview_left": {
328
+ "resolution": [
329
+ 256,
330
+ 256
331
+ ],
332
+ "channels": 3,
333
+ "fps": 20.0
334
+ },
335
+ "robot0_agentview_right": {
336
+ "resolution": [
337
+ 256,
338
+ 256
339
+ ],
340
+ "channels": 3,
341
+ "fps": 20.0
342
+ }
343
+ },
344
+ "state": {
345
+ "base_position": {
346
+ "absolute": true,
347
+ "rotation_type": null,
348
+ "shape": [
349
+ 3
350
+ ],
351
+ "continuous": true
352
+ },
353
+ "base_rotation": {
354
+ "absolute": true,
355
+ "rotation_type": "quaternion",
356
+ "shape": [
357
+ 4
358
+ ],
359
+ "continuous": true
360
+ },
361
+ "end_effector_position_relative": {
362
+ "absolute": true,
363
+ "rotation_type": null,
364
+ "shape": [
365
+ 3
366
+ ],
367
+ "continuous": true
368
+ },
369
+ "end_effector_rotation_relative": {
370
+ "absolute": true,
371
+ "rotation_type": "quaternion",
372
+ "shape": [
373
+ 4
374
+ ],
375
+ "continuous": true
376
+ },
377
+ "gripper_qpos": {
378
+ "absolute": true,
379
+ "rotation_type": null,
380
+ "shape": [
381
+ 2
382
+ ],
383
+ "continuous": true
384
+ }
385
+ },
386
+ "action": {
387
+ "base_motion": {
388
+ "absolute": true,
389
+ "rotation_type": null,
390
+ "shape": [
391
+ 4
392
+ ],
393
+ "continuous": true
394
+ },
395
+ "control_mode": {
396
+ "absolute": true,
397
+ "rotation_type": null,
398
+ "shape": [
399
+ 1
400
+ ],
401
+ "continuous": true
402
+ },
403
+ "end_effector_position": {
404
+ "absolute": true,
405
+ "rotation_type": null,
406
+ "shape": [
407
+ 3
408
+ ],
409
+ "continuous": true
410
+ },
411
+ "end_effector_rotation": {
412
+ "absolute": true,
413
+ "rotation_type": "axis_angle",
414
+ "shape": [
415
+ 3
416
+ ],
417
+ "continuous": true
418
+ },
419
+ "gripper_close": {
420
+ "absolute": true,
421
+ "rotation_type": null,
422
+ "shape": [
423
+ 1
424
+ ],
425
+ "continuous": true
426
+ }
427
+ }
428
+ },
429
+ "embodiment_tag": "new_embodiment"
430
+ }
431
+ }
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:98cf98087479b3e570d7fc66a0d397f42423009d07149e2cac936dbf1290a117
3
+ size 4999367032
processing_line_only_v2/MGD_v2/mgd_v2_mask_ratio_ablation/mgd_token_mask_ratio_0.2/phase2/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7546a323a674ff18936f5b04a14c992b8b4388c0d58416e10a286ef15fdec8e8
3
+ size 2643339748