| model: |
| video_backbone: |
| encoder: |
| name: wan_vae |
| model_path: /mnt/data/wangyuran/Wan2.2-TI2V-5B |
| name: wan22_ti2v_5b |
| model_path: /mnt/data/wangyuran/Wan2.2-TI2V-5B |
| from_scratch: false |
| shift_video: 5.0 |
| components: |
| - attr: dit |
| model_class: openwam.model.video_backbone.wan.models.dit.WanModel |
| extra_kwargs: |
| has_image_input: false |
| patch_size: |
| - 1 |
| - 2 |
| - 2 |
| in_dim: 48 |
| dim: 3072 |
| ffn_dim: 14336 |
| freq_dim: 256 |
| text_dim: 4096 |
| out_dim: 48 |
| num_heads: 24 |
| num_layers: 30 |
| eps: 1.0e-06 |
| seperated_timestep: true |
| require_clip_embedding: false |
| require_vae_embedding: false |
| fuse_vae_embedding_in_latents: true |
| - attr: vae |
| model_class: openwam.model.video_backbone.wan.models.vae.WanVideoVAE38 |
| extra_kwargs: {} |
| - attr: text_encoder |
| model_class: openwam.model.video_backbone.wan.models.text_encoder.WanTextEncoder |
| extra_kwargs: {} |
| tokenizer: |
| class: openwam.model.video_backbone.wan.models.text_encoder.HuggingfaceTokenizer |
| attr: tokenizer |
| subdir: tokenizer/google/umt5-xxl |
| path_kwarg: name |
| kwargs: |
| seq_len: 512 |
| clean: whitespace |
| action_backbone: |
| dim: 1024 |
| ffn_dim: 4096 |
| shift_action: 5.0 |
| freeze: |
| - video_backbone.text_encoder |
| - video_backbone.reason1 |
| - video_backbone.vae |
| - video_backbone.image_encoder |
| - video_backbone.video_encoder |
| architecture: |
| framework: dual_system |
| variant: joint_self_attn |
| action_dim: 80 |
| use_proprioception: true |
| state_dim: 80 |
| bridge_layers: null |
| bridge_interval: 1 |
| mot_checkpoint_mixed_attn: true |
| attention_mask_mode: action_sees_video |
| video_attention_mask_mode: first_frame_causal |
| detach_bridge: false |
| idm_video_cond_noise_prob: 0.5 |
| dataloader: |
| type: wuji_real_task |
| dataset_dir: /mnt/data/liangjiaqi/wuji_sft_data_eef |
| num_frames: 33 |
| video_stride: 4 |
| window_stride: 1 |
| height: 384 |
| width: 320 |
| multiview: true |
| target_camera: observation.images.stereo |
| camera_layout: |
| - observation.images.stereo |
| - observation.images.cam_left_wrist |
| - observation.images.cam_right_wrist |
| normalize_mode: quantile |
| split: train |
| enable_action_supervision: true |
| unify_action: true |
| unify_action_map: |
| - 0-8 |
| - 10-29 |
| - 32-40 |
| - 42-61 |
| seed: 42 |
| training: |
| debug: false |
| learning_rate: 0.0001 |
| adam_betas: |
| - 0.9 |
| - 0.95 |
| weight_decay: 0.01 |
| max_grad_norm: 1.0 |
| num_epochs: null |
| max_steps: 100000 |
| batch_size: 2 |
| gradient_accumulation_steps: 1 |
| lr_scheduler: cosine |
| warmup_ratio: 0.05 |
| lr_min_ratio: 0.01 |
| action_lr: null |
| video_lr: null |
| mixed_precision: bf16 |
| zero_stage: 2 |
| use_gradient_checkpointing: false |
| use_gradient_checkpointing_offload: false |
| initialize_model_on_cpu: false |
| offload_optimizer_device: none |
| lambda_video: 1.0 |
| lambda_action: 1.0 |
| max_timestep_boundary: 1.0 |
| min_timestep_boundary: 0.0 |
| timestep_sampling: default |
| timestep_sampling_lead: video |
| timestep_sampling_alpha: 1.0 |
| output_path: /mnt/data/liangjiaqi/openwam_checkpoints |
| save_steps: 2000 |
| keep_last_k_ckpts: 1 |
| finetune_ckpt_path: /mnt/data/wangyuran/openwam_checkpoints/pretrained_final_dualsystem_jointselfattention |
| resume_ckpt_path: null |
| dataset_num_workers: 8 |
| project: |
| name: open-wam |
| output_dir: outputs/${now:%Y-%m-%d_%H-%M-%S} |
| seed: 42 |
| wandb: |
| project: open-wam |
| run_name: wuji-eef-sft |
| entity: null |
|
|