Upload folder using huggingface_hub

Browse files

Files changed (16) hide show

.source_vm +1 -0
config.json +70 -0
embodiment_id.json +11 -0
experiment_cfg/conf.yaml +209 -0
experiment_cfg/config.yaml +242 -0
experiment_cfg/dataset_statistics.json +257 -0
experiment_cfg/final_model_config.json +54 -0
experiment_cfg/final_processor_config.json +0 -0
model-00001-of-00002.safetensors +3 -0
model-00002-of-00002.safetensors +3 -0
model.safetensors.index.json +0 -0
processor_config.json +454 -0
statistics.json +0 -0
trainer_state.json +154 -0
training_args.bin +3 -0
wandb_config.json +1 -0

.source_vm ADDED Viewed

	@@ -0,0 +1 @@


1	+ Gr00t-train

config.json ADDED Viewed

	@@ -0,0 +1,70 @@

+{
+  "action_horizon": 50,
+  "add_pos_embed": true,
+  "apply_sincos_state_encoding": true,
+  "architectures": [
+    "Gr00tN1d6"
+  ],
+  "attn_dropout": 0.2,
+  "attn_implementation": null,
+  "backbone_embedding_dim": 2048,
+  "backbone_model_type": "eagle",
+  "backbone_trainable_params_fp32": true,
+  "collator_overwrite_image_inputs": false,
+  "color_jitter_params": {
+    "brightness": 0.1,
+    "contrast": 0.1,
+    "hue": 0.1,
+    "saturation": 0.1
+  },
+  "crop_fraction": 0.95,
+  "diffusion_model_cfg": {
+    "attention_head_dim": 48,
+    "dropout": 0.2,
+    "final_dropout": true,
+    "interleave_self_attention": true,
+    "norm_type": "ada_norm",
+    "num_attention_heads": 32,
+    "num_layers": 32,
+    "output_dim": 1024,
+    "positional_embeddings": null
+  },
+  "eagle_collator": true,
+  "formalize_language": true,
+  "gemma_collator": false,
+  "hidden_size": 1024,
+  "image_crop_size": null,
+  "image_target_size": null,
+  "input_embedding_dim": 1536,
+  "load_bf16": true,
+  "max_action_dim": 128,
+  "max_num_embodiments": 32,
+  "max_seq_len": 1024,
+  "max_state_dim": 128,
+  "model_dtype": "bfloat16",
+  "model_name": "nvidia/Eagle-Block2A-2B-v2",
+  "model_type": "Gr00tN1d6",
+  "noise_beta_alpha": 1.5,
+  "noise_beta_beta": 1.0,
+  "noise_s": 0.999,
+  "num_inference_timesteps": 4,
+  "num_timestep_buckets": 1000,
+  "random_rotation_angle": null,
+  "reproject_vision": false,
+  "select_layer": 16,
+  "shortest_image_edge": 256,
+  "state_dropout_prob": 0.0,
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.51.3",
+  "tune_diffusion_model": true,
+  "tune_llm": false,
+  "tune_projector": true,
+  "tune_top_llm_layers": 4,
+  "tune_visual": false,
+  "tune_vlln": true,
+  "use_albumentations_transforms": true,
+  "use_alternate_vl_dit": true,
+  "use_flash_attention": true,
+  "use_relative_action": true,
+  "use_vlln": true
+}

embodiment_id.json ADDED Viewed

	@@ -0,0 +1,11 @@

+{
+  "robocasa_panda_omron": 13,
+  "gr1": 20,
+  "behavior_r1_pro": 24,
+  "unitree_g1": 8,
+  "oxe_google": 0,
+  "oxe_widowx": 1,
+  "libero_panda": 2,
+  "oxe_droid": 16,
+  "new_embodiment": 10
+}

experiment_cfg/conf.yaml ADDED Viewed

	@@ -0,0 +1,209 @@

+load_config_path: null
+model:
+  model_type: Gr00tN1d6
+  model_dtype: bfloat16
+  model_name: nvidia/Eagle-Block2A-2B-v2
+  backbone_model_type: eagle
+  model_revision: null
+  tune_top_llm_layers: 4
+  backbone_embedding_dim: 2048
+  tune_llm: false
+  tune_visual: false
+  select_layer: 16
+  reproject_vision: false
+  use_flash_attention: true
+  load_bf16: false
+  collator_overwrite_image_inputs: false
+  eagle_collator: true
+  backbone_trainable_params_fp32: true
+  image_crop_size: null
+  image_target_size: null
+  shortest_image_edge: 256
+  crop_fraction: 0.95
+  random_rotation_angle: null
+  color_jitter_params:
+    brightness: 0.3
+    contrast: 0.4
+    saturation: 0.5
+    hue: 0.08
+  use_albumentations_transforms: true
+  extra_augmentation_config: null
+  formalize_language: true
+  apply_sincos_state_encoding: false
+  use_relative_action: true
+  max_state_dim: 29
+  max_action_dim: 29
+  action_horizon: 16
+  hidden_size: 1024
+  input_embedding_dim: 1536
+  add_pos_embed: true
+  attn_dropout: 0.2
+  use_vlln: true
+  max_seq_len: 1024
+  use_alternate_vl_dit: true
+  attend_text_every_n_blocks: 2
+  diffusion_model_cfg:
+    positional_embeddings: null
+    num_layers: 32
+    num_attention_heads: 32
+    attention_head_dim: 48
+    norm_type: ada_norm
+    dropout: 0.2
+    final_dropout: true
+    output_dim: 1024
+    interleave_self_attention: true
+  num_inference_timesteps: 4
+  noise_beta_alpha: 1.5
+  noise_beta_beta: 1.0
+  noise_s: 0.999
+  num_timestep_buckets: 1000
+  tune_projector: true
+  tune_diffusion_model: true
+  tune_vlln: true
+  state_dropout_prob: 0.0
+  state_additive_noise_scale: 0.0
+  max_num_embodiments: 32
+data:
+  datasets:
+  - dataset_paths:
+    - /data/datasets/Dongkkka/Test_lerobot
+    embodiment_tag: new_embodiment
+    mix_ratio: 1.0
+    dataset_type: physical_embodiment
+    val_dataset_path: null
+  modality_configs:
+    new_embodiment:
+      video:
+        delta_indices:
+        - 0
+        modality_keys:
+        - cam_left_head
+        sin_cos_embedding_keys: null
+        mean_std_embedding_keys: null
+        action_configs: null
+      state:
+        delta_indices:
+        - 0
+        modality_keys:
+        - arm_left
+        - arm_right
+        sin_cos_embedding_keys: null
+        mean_std_embedding_keys: null
+        action_configs: null
+      action:
+        delta_indices:
+        - 0
+        - 1
+        - 2
+        - 3
+        - 4
+        - 5
+        - 6
+        - 7
+        - 8
+        - 9
+        - 10
+        - 11
+        - 12
+        - 13
+        - 14
+        - 15
+        modality_keys:
+        - arm_left
+        - arm_right
+        sin_cos_embedding_keys: null
+        mean_std_embedding_keys: null
+        action_configs:
+        - rep: ABSOLUTE
+          type: NON_EEF
+          format: DEFAULT
+          state_key: null
+        - rep: ABSOLUTE
+          type: NON_EEF
+          format: DEFAULT
+          state_key: null
+      language:
+        delta_indices:
+        - 0
+        modality_keys:
+        - annotation.human.task_description
+        sin_cos_embedding_keys: null
+        mean_std_embedding_keys: null
+        action_configs: null
+  download_cache: false
+  shard_size: 1024
+  episode_sampling_rate: 0.1
+  num_shards_per_epoch: 100000
+  override_pretraining_statistics: false
+  mode: single_turn
+  random_chop: 0.0
+  mock_dataset_mode: false
+  shuffle: true
+  seed: 42
+  multiprocessing_context: fork
+  allow_padding: false
+  subsample_ratio: 1.0
+  image_crop_size:
+  - 244
+  - 244
+  image_target_size:
+  - 224
+  - 224
+  video_backend: torchcodec
+training:
+  output_dir: /data/checkpoints/TestModel4
+  experiment_name: null
+  max_steps: 200
+  global_batch_size: 48
+  batch_size: null
+  gradient_accumulation_steps: 1
+  learning_rate: 0.0001
+  lr_scheduler_type: cosine
+  weight_decay: 1.0e-05
+  warmup_ratio: 0.05
+  warmup_steps: 0
+  max_grad_norm: 1.0
+  optim: adamw_torch
+  start_from_checkpoint: nvidia/GR00T-N1.6-3B
+  tf32: true
+  fp16: false
+  bf16: true
+  eval_bf16: true
+  logging_steps: 10
+  save_steps: 100
+  save_total_limit: 10
+  save_vl_model: false
+  upload_checkpoints: false
+  upload_every: 1000
+  upload_last_n_checkpoints: 5
+  max_concurrent_uploads: 2
+  eval_strategy: 'no'
+  eval_steps: 500
+  eval_set_split_ratio: 0.1
+  eval_batch_size: 2
+  save_best_eval_metric_name: ''
+  save_best_eval_metric_greater_is_better: true
+  deepspeed_stage: 2
+  gradient_checkpointing: false
+  transformers_trust_remote_code: true
+  transformers_local_files_only: false
+  transformers_cache_dir: null
+  transformers_access_token: null
+  use_ddp: false
+  ddp_bucket_cap_mb: 100
+  num_gpus: 1
+  dataloader_num_workers: 8
+  remove_unused_columns: false
+  use_wandb: false
+  wandb_project: finetune-gr00t-n1d6
+  enable_profiling: false
+  max_retries: 3
+  assert_loss_less_than: null
+  add_rl_callback: false
+  enable_open_loop_eval: false
+  open_loop_eval_traj_ids:
+  - 0
+  open_loop_eval_steps_per_traj: 100
+  open_loop_eval_plot_indices: null
+max_steps: 200
+save_steps: 100

experiment_cfg/config.yaml ADDED Viewed

	@@ -0,0 +1,242 @@

+!!python/object:gr00t.configs.base_config.Config
+data: !!python/object:gr00t.configs.data.data_config.DataConfig
+  allow_padding: false
+  datasets:
+  - !!python/object:gr00t.configs.data.data_config.SingleDatasetConfig
+    dataset_paths:
+    - /data/datasets/Dongkkka/Test_lerobot
+    dataset_type: physical_embodiment
+    embodiment_tag: new_embodiment
+    mix_ratio: 1.0
+    val_dataset_path: null
+  download_cache: false
+  episode_sampling_rate: 0.1
+  image_crop_size:
+  - 244
+  - 244
+  image_target_size:
+  - 224
+  - 224
+  mock_dataset_mode: false
+  modality_configs:
+    new_embodiment:
+      action: !!python/object:gr00t.data.types.ModalityConfig
+        action_configs:
+        - !!python/object:gr00t.data.types.ActionConfig
+          format: &id001 !!python/object/apply:gr00t.data.types.ActionFormat
+          - default
+          rep: &id002 !!python/object/apply:gr00t.data.types.ActionRepresentation
+          - absolute
+          state_key: null
+          type: &id003 !!python/object/apply:gr00t.data.types.ActionType
+          - non_eef
+        - !!python/object:gr00t.data.types.ActionConfig
+          format: *id001
+          rep: *id002
+          state_key: null
+          type: *id003
+        delta_indices:
+        - 0
+        - 1
+        - 2
+        - 3
+        - 4
+        - 5
+        - 6
+        - 7
+        - 8
+        - 9
+        - 10
+        - 11
+        - 12
+        - 13
+        - 14
+        - 15
+        mean_std_embedding_keys: null
+        modality_keys:
+        - arm_left
+        - arm_right
+        sin_cos_embedding_keys: null
+      language: !!python/object:gr00t.data.types.ModalityConfig
+        action_configs: null
+        delta_indices:
+        - 0
+        mean_std_embedding_keys: null
+        modality_keys:
+        - annotation.human.task_description
+        sin_cos_embedding_keys: null
+      state: !!python/object:gr00t.data.types.ModalityConfig
+        action_configs: null
+        delta_indices:
+        - 0
+        mean_std_embedding_keys: null
+        modality_keys:
+        - arm_left
+        - arm_right
+        sin_cos_embedding_keys: null
+      video: !!python/object:gr00t.data.types.ModalityConfig
+        action_configs: null
+        delta_indices:
+        - 0
+        mean_std_embedding_keys: null
+        modality_keys:
+        - cam_left_head
+        sin_cos_embedding_keys: null
+  mode: single_turn
+  multiprocessing_context: fork
+  num_shards_per_epoch: 100000
+  override_pretraining_statistics: false
+  random_chop: 0.0
+  seed: 42
+  shard_size: 1024
+  shuffle: true
+  subsample_ratio: 1.0
+  video_backend: torchcodec
+load_config_path: null
+model: !!python/object:gr00t.configs.model.gr00t_n1d6.Gr00tN1d6Config
+  _attn_implementation_autoset: false
+  _attn_implementation_internal: null
+  _commit_hash: null
+  _name_or_path: ''
+  add_cross_attention: false
+  architectures: null
+  backbone_model_type: eagle
+  backbone_trainable_params_fp32: true
+  bad_words_ids: null
+  begin_suppress_tokens: null
+  bos_token_id: null
+  chunk_size_feed_forward: 0
+  color_jitter_params:
+    brightness: 0.3
+    contrast: 0.4
+    hue: 0.08
+    saturation: 0.5
+  cross_attention_hidden_size: null
+  decoder_start_token_id: null
+  diffusion_model_cfg:
+    attention_head_dim: 48
+    dropout: 0.2
+    final_dropout: true
+    interleave_self_attention: true
+    norm_type: ada_norm
+    num_attention_heads: 32
+    num_layers: 32
+    output_dim: 1024
+    positional_embeddings: null
+  diversity_penalty: 0.0
+  do_sample: false
+  eagle_collator: true
+  early_stopping: false
+  encoder_no_repeat_ngram_size: 0
+  eos_token_id: null
+  exponential_decay_length_penalty: null
+  extra_augmentation_config: null
+  finetuning_task: null
+  forced_bos_token_id: null
+  forced_eos_token_id: null
+  id2label:
+    0: LABEL_0
+    1: LABEL_1
+  is_decoder: false
+  is_encoder_decoder: false
+  label2id:
+    LABEL_0: 0
+    LABEL_1: 1
+  length_penalty: 1.0
+  load_bf16: false
+  max_length: 20
+  min_length: 0
+  model_name: nvidia/Eagle-Block2A-2B-v2
+  no_repeat_ngram_size: 0
+  num_beam_groups: 1
+  num_beams: 1
+  num_return_sequences: 1
+  output_attentions: false
+  output_hidden_states: false
+  output_scores: false
+  pad_token_id: null
+  prefix: null
+  problem_type: null
+  pruned_heads: {}
+  random_rotation_angle: null
+  remove_invalid_values: false
+  repetition_penalty: 1.0
+  reproject_vision: false
+  return_dict: true
+  return_dict_in_generate: false
+  sep_token_id: null
+  state_dropout_prob: 0.0
+  suppress_tokens: null
+  task_specific_params: null
+  temperature: 1.0
+  tf_legacy_loss: false
+  tie_encoder_decoder: false
+  tie_word_embeddings: true
+  tokenizer_class: null
+  top_k: 50
+  top_p: 1.0
+  torch_dtype: null
+  torchscript: false
+  transformers_version: null
+  tune_diffusion_model: true
+  tune_llm: false
+  tune_projector: true
+  tune_visual: false
+  typical_p: 1.0
+  use_bfloat16: false
+  use_relative_action: true
+training: !!python/object:gr00t.configs.training.training_config.TrainingConfig
+  add_rl_callback: false
+  assert_loss_less_than: null
+  batch_size: null
+  bf16: true
+  dataloader_num_workers: 8
+  ddp_bucket_cap_mb: 100
+  deepspeed_stage: 2
+  enable_open_loop_eval: false
+  enable_profiling: false
+  eval_batch_size: 2
+  eval_bf16: true
+  eval_set_split_ratio: 0.1
+  eval_steps: 500
+  eval_strategy: 'no'
+  experiment_name: null
+  fp16: false
+  global_batch_size: 48
+  gradient_accumulation_steps: 1
+  gradient_checkpointing: false
+  learning_rate: 0.0001
+  logging_steps: 10
+  lr_scheduler_type: cosine
+  max_concurrent_uploads: 2
+  max_grad_norm: 1.0
+  max_retries: 3
+  max_steps: 200
+  num_gpus: 1
+  open_loop_eval_plot_indices: null
+  open_loop_eval_steps_per_traj: 100
+  open_loop_eval_traj_ids:
+  - 0
+  optim: adamw_torch
+  output_dir: /data/checkpoints/TestModel4
+  remove_unused_columns: false
+  save_best_eval_metric_greater_is_better: true
+  save_best_eval_metric_name: ''
+  save_steps: 100
+  save_total_limit: 10
+  save_vl_model: false
+  start_from_checkpoint: nvidia/GR00T-N1.6-3B
+  tf32: true
+  transformers_access_token: null
+  transformers_cache_dir: null
+  transformers_local_files_only: false
+  transformers_trust_remote_code: true
+  upload_checkpoints: false
+  upload_every: 1000
+  upload_last_n_checkpoints: 5
+  use_ddp: false
+  use_wandb: false
+  wandb_project: finetune-gr00t-n1d6
+  warmup_ratio: 0.05
+  warmup_steps: 0
+  weight_decay: 1.0e-05

experiment_cfg/dataset_statistics.json ADDED Viewed

	@@ -0,0 +1,257 @@

+{
+  "new_embodiment": {
+    "state": {
+      "arm_left": {
+        "min": [
+          0.5357068181037903,
+          0.09819874167442322,
+          -0.005884254351258278,
+          -1.9916343688964844,
+          0.171865776181221,
+          -0.7192093133926392,
+          -0.2347273975610733,
+          0.1835404485464096
+        ],
+        "max": [
+          0.8060949444770813,
+          0.14419420063495636,
+          0.1393405795097351,
+          -1.6996147632598877,
+          0.36047351360321045,
+          -0.3933342397212982,
+          -0.17479148507118225,
+          0.1919773519039154
+        ],
+        "mean": [
+          0.647490382194519,
+          0.11591889709234238,
+          0.05527754873037338,
+          -1.8612351417541504,
+          0.2748093605041504,
+          -0.5980395078659058,
+          -0.20959335565567017,
+          0.1872807741165161
+        ],
+        "std": [
+          0.07161965221166611,
+          0.01286551449447867,
+          0.02708311937749386,
+          0.07693036645650828,
+          0.04468376561999321,
+          0.08223182708024979,
+          0.020106619223952207,
+          0.0019493200816217977
+        ],
+        "q01": [
+          0.542970244884491,
+          0.09969677031040192,
+          0.0053154832031577825,
+          -1.9704224967956543,
+          0.21393687039613724,
+          -0.7124262452125549,
+          -0.23469635844230652,
+          0.18635275959968567
+        ],
+        "q99": [
+          0.7976319086551666,
+          0.1417374312877655,
+          0.1274196347594261,
+          -1.710426688194275,
+          0.35903324604034426,
+          -0.396414190530777,
+          -0.17485354840755463,
+          0.1919773519039154
+        ]
+      },
+      "arm_right": {
+        "min": [
+          -0.12207131832838058,
+          -0.35657861828804016,
+          -0.4954638183116913,
+          -2.450582265853882,
+          0.8439650535583496,
+          -0.33980071544647217,
+          -0.9373581409454346,
+          0.18072815239429474
+        ],
+        "max": [
+          0.31729432940483093,
+          -0.019150791689753532,
+          0.5513582229614258,
+          -2.1091995239257812,
+          1.5962507724761963,
+          0.3736201822757721,
+          -0.6251019239425659,
+          0.20322653651237488
+        ],
+        "mean": [
+          0.07138513773679733,
+          -0.16919372975826263,
+          0.0536830797791481,
+          -2.2932255268096924,
+          1.2343531847000122,
+          0.08430171012878418,
+          -0.763832688331604,
+          0.19362585246562958
+        ],
+        "std": [
+          0.06878731399774551,
+          0.06808101385831833,
+          0.23403118550777435,
+          0.0615885853767395,
+          0.17844322323799133,
+          0.10313137620687485,
+          0.05072656273841858,
+          0.005280985496938146
+        ],
+        "q01": [
+          -0.07233462154865265,
+          -0.31898770451545716,
+          -0.37645135819911957,
+          -2.429511785507202,
+          0.9102486312389374,
+          -0.21467964828014374,
+          -0.899279260635376,
+          0.18072815239429474
+        ],
+        "q99": [
+          0.22330133676528924,
+          -0.05408480763435364,
+          0.48604661107063274,
+          -2.1352156257629393,
+          1.5619227027893066,
+          0.2997923380136489,
+          -0.6622847545146943,
+          0.20041424036026
+        ]
+      }
+    },
+    "action": {
+      "arm_left": {
+        "min": [
+          0.5353593230247498,
+          0.09817477315664291,
+          -0.006135923322290182,
+          -1.9926410913467407,
+          0.16260196268558502,
+          -0.7185632586479187,
+          -0.2346990555524826,
+          0.12024854868650436
+        ],
+        "max": [
+          0.8068739175796509,
+          0.14419420063495636,
+          0.139592245221138,
+          -1.699650764465332,
+          0.3604854941368103,
+          -0.39335930347442627,
+          -0.17487381398677826,
+          0.1303728222846985
+        ],
+        "mean": [
+          0.647443413734436,
+          0.11592874675989151,
+          0.05530872195959091,
+          -1.861175298690796,
+          0.274822860956192,
+          -0.5980128645896912,
+          -0.20963473618030548,
+          0.12473082542419434
+        ],
+        "std": [
+          0.07161091268062592,
+          0.012873547151684761,
+          0.02710757777094841,
+          0.07696112245321238,
+          0.04471345245838165,
+          0.08225028216838837,
+          0.020105436444282532,
+          0.0023378378245981287
+        ],
+        "q01": [
+          0.5430291891098022,
+          0.09970875084400177,
+          0.004601942375302315,
+          -1.9711652994155884,
+          0.21322332322597504,
+          -0.7124273180961609,
+          -0.2346990555524826,
+          0.1236233040690422
+        ],
+        "q99": [
+          0.7976700067520142,
+          0.14112623035907745,
+          0.12732040882110596,
+          -1.7103885412216187,
+          0.35895150899887085,
+          -0.3964272737503052,
+          -0.17487381398677826,
+          0.1303728222846985
+        ]
+      },
+      "arm_right": {
+        "min": [
+          -0.1257864236831665,
+          -0.3604854941368103,
+          -0.5062136650085449,
+          -2.451301336288452,
+          0.8375535011291504,
+          -0.3473398983478546,
+          -0.9418641924858093,
+          0.11687378585338593
+        ],
+        "max": [
+          0.3206019699573517,
+          -0.016873788088560104,
+          0.558368980884552,
+          -2.104621648788452,
+          1.6214176416397095,
+          0.3766990303993225,
+          -0.6258641481399536,
+          0.14387184381484985
+        ],
+        "mean": [
+          0.0714845135807991,
+          -0.16912876069545746,
+          0.053440161049366,
+          -2.2931737899780273,
+          1.2342159748077393,
+          0.08426135778427124,
+          -0.7638657689094543,
+          0.1323314756155014
+        ],
+        "std": [
+          0.06979027390480042,
+          0.07009217888116837,
+          0.24079644680023193,
+          0.062028586864471436,
+          0.18407252430915833,
+          0.10534369945526123,
+          0.05093063414096832,
+          0.006346254609525135
+        ],
+        "q01": [
+          -0.0746128237247467,
+          -0.32032585799694063,
+          -0.3850291669368744,
+          -2.4310834121704104,
+          0.9007228302955628,
+          -0.20928162336349487,
+          -0.901704580783844,
+          0.11687378585338593
+        ],
+        "q99": [
+          0.22702915966510773,
+          -0.05062136426568031,
+          0.49519968152046157,
+          -2.1337673664093018,
+          1.5677284002304077,
+          0.3055836880207053,
+          -0.6626796722412109,
+          0.140497088432312
+        ]
+      }
+    },
+    "relative_action": {}
+  }
+}

experiment_cfg/final_model_config.json ADDED Viewed

	@@ -0,0 +1,54 @@

+{
+  "model_type": "Gr00tN1d6",
+  "model_dtype": "bfloat16",
+  "model_name": "nvidia/Eagle-Block2A-2B-v2",
+  "backbone_model_type": "eagle",
+  "model_revision": null,
+  "tune_top_llm_layers": 4,
+  "backbone_embedding_dim": 2048,
+  "tune_llm": false,
+  "tune_visual": false,
+  "select_layer": 16,
+  "reproject_vision": false,
+  "use_flash_attention": true,
+  "load_bf16": true,
+  "collator_overwrite_image_inputs": false,
+  "eagle_collator": true,
+  "backbone_trainable_params_fp32": true,
+  "extra_augmentation_config": null,
+  "apply_sincos_state_encoding": true,
+  "use_relative_action": true,
+  "max_state_dim": 128,
+  "max_action_dim": 128,
+  "action_horizon": 50,
+  "hidden_size": 1024,
+  "input_embedding_dim": 1536,
+  "add_pos_embed": true,
+  "attn_dropout": 0.2,
+  "use_vlln": true,
+  "max_seq_len": 1024,
+  "use_alternate_vl_dit": true,
+  "attend_text_every_n_blocks": 2,
+  "diffusion_model_cfg": {
+    "attention_head_dim": 48,
+    "dropout": 0.2,
+    "final_dropout": true,
+    "interleave_self_attention": true,
+    "norm_type": "ada_norm",
+    "num_attention_heads": 32,
+    "num_layers": 32,
+    "output_dim": 1024,
+    "positional_embeddings": null
+  },
+  "num_inference_timesteps": 4,
+  "noise_beta_alpha": 1.5,
+  "noise_beta_beta": 1.0,
+  "noise_s": 0.999,
+  "num_timestep_buckets": 1000,
+  "tune_projector": true,
+  "tune_diffusion_model": true,
+  "tune_vlln": true,
+  "state_dropout_prob": 0.0,
+  "state_additive_noise_scale": 0.0,
+  "max_num_embodiments": 32
+}

experiment_cfg/final_processor_config.json ADDED Viewed

The diff for this file is too large to render. See raw diff

model-00001-of-00002.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:49c72b22bcc7e7bf23d62d89f589c6dde1b3d3e2b78bdebe42815c856bcc236d
+size 4990120184

model-00002-of-00002.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:edc430476f0dfe396a7097de20e031448bac1c379a3b0ecda50cf0b831d8db54
+size 4823190320

model.safetensors.index.json ADDED Viewed

The diff for this file is too large to render. See raw diff

processor_config.json ADDED Viewed

	@@ -0,0 +1,454 @@

+{
+  "processor_class": "Gr00tN1d6Processor",
+  "processor_kwargs": {
+    "modality_configs": {
+      "behavior_r1_pro": {
+        "video": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "observation.images.rgb.head_256_256",
+            "observation.images.rgb.left_wrist_256_256",
+            "observation.images.rgb.right_wrist_256_256"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        },
+        "state": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "robot_pos",
+            "robot_ori_cos",
+            "robot_ori_sin",
+            "robot_2d_ori",
+            "robot_2d_ori_cos",
+            "robot_2d_ori_sin",
+            "robot_lin_vel",
+            "robot_ang_vel",
+            "arm_left_qpos",
+            "arm_left_qpos_sin",
+            "arm_left_qpos_cos",
+            "eef_left_pos",
+            "eef_left_quat",
+            "gripper_left_qpos",
+            "arm_right_qpos",
+            "arm_right_qpos_sin",
+            "arm_right_qpos_cos",
+            "eef_right_pos",
+            "eef_right_quat",
+            "gripper_right_qpos",
+            "trunk_qpos"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        },
+        "action": {
+          "delta_indices": [
+            0,
+            1,
+            2,
+            3,
+            4,
+            5,
+            6,
+            7,
+            8,
+            9,
+            10,
+            11,
+            12,
+            13,
+            14,
+            15,
+            16,
+            17,
+            18,
+            19,
+            20,
+            21,
+            22,
+            23,
+            24,
+            25,
+            26,
+            27,
+            28,
+            29,
+            30,
+            31
+          ],
+          "modality_keys": [
+            "base",
+            "torso",
+            "left_arm",
+            "left_gripper",
+            "right_arm",
+            "right_gripper"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": [
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "RELATIVE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": "trunk_qpos"
+            },
+            {
+              "rep": "RELATIVE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": "arm_left_qpos"
+            },
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "RELATIVE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": "arm_right_qpos"
+            },
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            }
+          ]
+        },
+        "language": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "annotation.human.coarse_action"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        }
+      },
+      "gr1": {
+        "video": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "ego_view_bg_crop_pad_res256_freq20"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        },
+        "state": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "left_arm",
+            "right_arm",
+            "left_hand",
+            "right_hand",
+            "waist"
+          ],
+          "sin_cos_embedding_keys": [
+            "left_arm",
+            "right_arm",
+            "left_hand",
+            "right_hand",
+            "waist"
+          ],
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        },
+        "action": {
+          "delta_indices": [
+            0,
+            1,
+            2,
+            3,
+            4,
+            5,
+            6,
+            7,
+            8,
+            9,
+            10,
+            11,
+            12,
+            13,
+            14,
+            15
+          ],
+          "modality_keys": [
+            "left_arm",
+            "right_arm",
+            "left_hand",
+            "right_hand",
+            "waist"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": [
+            {
+              "rep": "RELATIVE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "RELATIVE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "RELATIVE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "RELATIVE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            }
+          ]
+        },
+        "language": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "task"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        }
+      },
+      "robocasa_panda_omron": {
+        "video": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "res256_image_side_0",
+            "res256_image_side_1",
+            "res256_image_wrist_0"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        },
+        "state": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "end_effector_position_relative",
+            "end_effector_rotation_relative",
+            "gripper_qpos",
+            "base_position",
+            "base_rotation"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        },
+        "action": {
+          "delta_indices": [
+            0,
+            1,
+            2,
+            3,
+            4,
+            5,
+            6,
+            7,
+            8,
+            9,
+            10,
+            11,
+            12,
+            13,
+            14,
+            15
+          ],
+          "modality_keys": [
+            "end_effector_position",
+            "end_effector_rotation",
+            "gripper_close",
+            "base_motion",
+            "control_mode"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": [
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            }
+          ]
+        },
+        "language": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "annotation.human.action.task_description"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        }
+      },
+      "new_embodiment": {
+        "video": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "cam_left_head"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        },
+        "state": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "arm_left",
+            "arm_right"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        },
+        "action": {
+          "delta_indices": [
+            0,
+            1,
+            2,
+            3,
+            4,
+            5,
+            6,
+            7,
+            8,
+            9,
+            10,
+            11,
+            12,
+            13,
+            14,
+            15
+          ],
+          "modality_keys": [
+            "arm_left",
+            "arm_right"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": [
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            },
+            {
+              "rep": "ABSOLUTE",
+              "type": "NON_EEF",
+              "format": "DEFAULT",
+              "state_key": null
+            }
+          ]
+        },
+        "language": {
+          "delta_indices": [
+            0
+          ],
+          "modality_keys": [
+            "annotation.human.task_description"
+          ],
+          "sin_cos_embedding_keys": null,
+          "mean_std_embedding_keys": null,
+          "action_configs": null
+        }
+      }
+    },
+    "image_crop_size": null,
+    "image_target_size": null,
+    "use_albumentations": true,
+    "random_rotation_angle": null,
+    "color_jitter_params": {
+      "brightness": 0.3,
+      "contrast": 0.4,
+      "saturation": 0.5,
+      "hue": 0.08
+    },
+    "shortest_image_edge": 256,
+    "crop_fraction": 0.95,
+    "model_name": "nvidia/Eagle-Block2A-2B-v2",
+    "model_type": "eagle",
+    "formalize_language": true,
+    "max_state_dim": 128,
+    "max_action_dim": 128,
+    "max_action_horizon": 50,
+    "use_percentiles": false,
+    "clip_outliers": true,
+    "apply_sincos_state_encoding": true,
+    "use_relative_action": true
+  }
+}

statistics.json ADDED Viewed

The diff for this file is too large to render. See raw diff

trainer_state.json ADDED Viewed

	@@ -0,0 +1,154 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 0.5,
+  "eval_steps": 500,
+  "global_step": 200,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "grad_norm": 0.38351285457611084,
+      "learning_rate": 9.956320346634876e-05,
+      "loss": 1.2186,
+      "step": 10
+    },
+    {
+      "grad_norm": 0.453412801027298,
+      "learning_rate": 9.473646649103818e-05,
+      "loss": 1.176,
+      "step": 20
+    },
+    {
+      "grad_norm": 0.3962445557117462,
+      "learning_rate": 8.506183921362443e-05,
+      "loss": 1.1279,
+      "step": 30
+    },
+    {
+      "grad_norm": 0.33287128806114197,
+      "learning_rate": 7.158771761692464e-05,
+      "loss": 1.0833,
+      "step": 40
+    },
+    {
+      "grad_norm": 0.34996676445007324,
+      "learning_rate": 5.577423184847932e-05,
+      "loss": 1.0655,
+      "step": 50
+    },
+    {
+      "grad_norm": 0.33573195338249207,
+      "learning_rate": 3.933501846281267e-05,
+      "loss": 1.0523,
+      "step": 60
+    },
+    {
+      "grad_norm": 0.36666247248649597,
+      "learning_rate": 2.405152131093926e-05,
+      "loss": 1.0321,
+      "step": 70
+    },
+    {
+      "grad_norm": 0.39881908893585205,
+      "learning_rate": 1.157994445715706e-05,
+      "loss": 1.0118,
+      "step": 80
+    },
+    {
+      "grad_norm": 0.40295055508613586,
+      "learning_rate": 3.271776770026963e-06,
+      "loss": 1.0024,
+      "step": 90
+    },
+    {
+      "grad_norm": 0.4325471818447113,
+      "learning_rate": 2.7337132953697554e-08,
+      "loss": 1.0113,
+      "step": 100
+    },
+    {
+      "grad_norm": 0.39502644538879395,
+      "learning_rate": 4.669547078371504e-05,
+      "loss": 1.0233,
+      "step": 110
+    },
+    {
+      "grad_norm": 0.9289461970329285,
+      "learning_rate": 3.852880399766243e-05,
+      "loss": 0.9895,
+      "step": 120
+    },
+    {
+      "grad_norm": 0.8774065375328064,
+      "learning_rate": 3.0675041535377405e-05,
+      "loss": 0.9202,
+      "step": 130
+    },
+    {
+      "grad_norm": 1.0103175640106201,
+      "learning_rate": 2.3348413563600325e-05,
+      "loss": 0.8666,
+      "step": 140
+    },
+    {
+      "grad_norm": 1.0156214237213135,
+      "learning_rate": 1.6748771394307585e-05,
+      "loss": 0.7999,
+      "step": 150
+    },
+    {
+      "grad_norm": 0.923975944519043,
+      "learning_rate": 1.1056136061894384e-05,
+      "loss": 0.7333,
+      "step": 160
+    },
+    {
+      "grad_norm": 1.110439658164978,
+      "learning_rate": 6.425787818636131e-06,
+      "loss": 0.6969,
+      "step": 170
+    },
+    {
+      "grad_norm": 0.8177175521850586,
+      "learning_rate": 2.9840304941919415e-06,
+      "loss": 0.6798,
+      "step": 180
+    },
+    {
+      "grad_norm": 0.762315571308136,
+      "learning_rate": 8.247462563808817e-07,
+      "loss": 0.6679,
+      "step": 190
+    },
+    {
+      "grad_norm": 0.8166346549987793,
+      "learning_rate": 6.834750376549792e-09,
+      "loss": 0.6646,
+      "step": 200
+    }
+  ],
+  "logging_steps": 10,
+  "max_steps": 200,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 9223372036854775807,
+  "save_steps": 100,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 0.0,
+  "train_batch_size": 48,
+  "trial_name": null,
+  "trial_params": null
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e468f995e72c66264a30c3d2737be2853d368278cba7aeb88c6f4203412707ae
+size 5713

wandb_config.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"project": "finetune-gr00t-n1d6", "run_id": "TestModel4"}