| model: |
| video_backbone: |
| encoder: |
| name: wan_vae |
| model_path: /mnt/data/wangyuran/Wan2.2-TI2V-5B |
| name: wan22_ti2v_5b |
| model_path: /mnt/data/wangyuran/Wan2.2-TI2V-5B |
| from_scratch: false |
| shift_video: 5.0 |
| components: |
| - attr: dit |
| model_class: openwam.model.video_backbone.wan.models.dit.WanModel |
| extra_kwargs: |
| has_image_input: false |
| patch_size: |
| - 1 |
| - 2 |
| - 2 |
| in_dim: 48 |
| dim: 3072 |
| ffn_dim: 14336 |
| freq_dim: 256 |
| text_dim: 4096 |
| out_dim: 48 |
| num_heads: 24 |
| num_layers: 30 |
| eps: 1.0e-06 |
| seperated_timestep: true |
| require_clip_embedding: false |
| require_vae_embedding: false |
| fuse_vae_embedding_in_latents: true |
| - attr: vae |
| model_class: openwam.model.video_backbone.wan.models.vae.WanVideoVAE38 |
| extra_kwargs: {} |
| - attr: text_encoder |
| model_class: openwam.model.video_backbone.wan.models.text_encoder.WanTextEncoder |
| extra_kwargs: {} |
| tokenizer: |
| class: openwam.model.video_backbone.wan.models.text_encoder.HuggingfaceTokenizer |
| attr: tokenizer |
| subdir: tokenizer/google/umt5-xxl |
| path_kwarg: name |
| kwargs: |
| seq_len: 512 |
| clean: whitespace |
| action_backbone: |
| dim: 1024 |
| ffn_dim: 4096 |
| shift_action: 5.0 |
| freeze: |
| - video_backbone.text_encoder |
| - video_backbone.reason1 |
| - video_backbone.vae |
| - video_backbone.image_encoder |
| - video_backbone.video_encoder |
| architecture: |
| framework: dual_system |
| variant: joint_self_attn |
| action_dim: 20 |
| use_proprioception: true |
| state_dim: 20 |
| bridge_layers: null |
| bridge_interval: 1 |
| mot_checkpoint_mixed_attn: true |
| attention_mask_mode: action_sees_video |
| video_attention_mask_mode: first_frame_causal |
| detach_bridge: false |
| idm_video_cond_noise_prob: 0.5 |
| dataloader: |
| type: robotwin |
| dataset_dir: /mnt/data/wangyuran/RoboTwin2.0/dataset |
| task_name: null |
| robot: aloha-agilex |
| variant: both |
| action_mode: eef |
| unify_action: false |
| unify_action_map: |
| - 0-9 |
| - 34-43 |
| num_frames: 33 |
| video_stride: 4 |
| height: 384 |
| width: 320 |
| split: train |
| val_ratio: 0.0 |
| repeat: 1 |
| window_stride: 1 |
| multiview: true |
| camera_layout: |
| - head_camera |
| - left_camera |
| - right_camera |
| target_camera: head_camera |
| normalize_mode: min-max |
| filter_static_segments: true |
| static_segment_threshold: 1.0e-05 |
| max_static_retry: 3 |
| train_tasks: null |
| holdout_tasks: null |
| val_variant: null |
| text_embedding_cache_dir: null |
| text_embedding_dropout: 0.0 |
| seed: 42 |
| training: |
| debug: false |
| learning_rate: 0.0001 |
| adam_betas: |
| - 0.9 |
| - 0.95 |
| weight_decay: 0.01 |
| max_grad_norm: 1.0 |
| num_epochs: 5 |
| max_steps: null |
| batch_size: 24 |
| gradient_accumulation_steps: 1 |
| lr_scheduler: cosine |
| warmup_ratio: 0.05 |
| lr_min_ratio: 0.01 |
| action_lr: null |
| video_lr: null |
| mixed_precision: bf16 |
| zero_stage: 2 |
| use_gradient_checkpointing: true |
| use_gradient_checkpointing_offload: false |
| initialize_model_on_cpu: false |
| offload_optimizer_device: none |
| lambda_video: 1.0 |
| lambda_action: 1.0 |
| max_timestep_boundary: 1.0 |
| min_timestep_boundary: 0.0 |
| output_path: /mnt/data/wangyuran/openwam_checkpoints |
| save_steps: 2000 |
| keep_last_k_ckpts: 1 |
| finetune_ckpt_path: null |
| resume_ckpt_path: null |
| dataset_num_workers: 8 |
| project: |
| name: open-wam |
| output_dir: outputs/${now:%Y-%m-%d_%H-%M-%S} |
| seed: 42 |
| wandb: |
| project: open-wam |
| run_name: null |
| entity: null |
|
|