File size: 3,268 Bytes
9df9fc6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
model:
  video_backbone:
    encoder:
      name: wan22_vae
      model_path: /path/to/Wan2.2-TI2V-5B
    name: wan22_ti2v_5b
    model_path: /path/to/Wan2.2-TI2V-5B
    from_scratch: false
    shift_video: 5.0
    components:
    - attr: dit
      model_class: openwam.model.video_backbone.wan.models.dit.WanModel
      extra_kwargs:
        has_image_input: false
        patch_size:
        - 1
        - 2
        - 2
        in_dim: 48
        dim: 3072
        ffn_dim: 14336
        freq_dim: 256
        text_dim: 4096
        out_dim: 48
        num_heads: 24
        num_layers: 30
        eps: 1.0e-06
        seperated_timestep: true
        require_clip_embedding: false
        require_vae_embedding: false
        fuse_vae_embedding_in_latents: true
    - attr: vae
      model_class: openwam.model.video_backbone.wan.models.vae.WanVideoVAE38
      extra_kwargs: {}
    - attr: text_encoder
      model_class: openwam.model.video_backbone.wan.models.text_encoder.WanTextEncoder
      extra_kwargs: {}
    tokenizer:
      class: openwam.model.video_backbone.wan.models.text_encoder.HuggingfaceTokenizer
      attr: tokenizer
      subdir: tokenizer/google/umt5-xxl
      path_kwarg: name
      kwargs:
        seq_len: 512
        clean: whitespace
  action_backbone:
    dim: 1024
    ffn_dim: 4096
    shift_action: 5.0
  freeze:
  - video_backbone.text_encoder
  - video_backbone.reason1
  - video_backbone.vae
  - video_backbone.image_encoder
  - video_backbone.video_encoder
  architecture:
    framework: dual_system
    variant: joint_self_attn
    action_dim: 80
    use_proprioception: true
    state_dim: 80
    bridge_layers: null
    bridge_interval: 1
    mot_checkpoint_mixed_attn: true
    attention_mask_mode: mutual
    video_attention_mask_mode: first_frame_causal
    detach_bridge: false
    idm_video_cond_noise_prob: 0.5
dataloader:
  type: vlabench
  dataset_dir: /path/to/vlabench_data
  action_mode: eef
  unify_action: true
  unify_action_map:
  - 0-9
  unify_state_map: null
  num_frames: 33
  video_stride: 4
  window_stride: 1
  split: train
  height: 384
  width: 320
  multiview: true
  normalize_mode: min-max
  color_jitter:
    enabled: true
    brightness: 0.2
    contrast: 0.2
    saturation: 0.2
    hue: 0.0
  seed: 42
training:
  debug: false
  learning_rate: 0.0001
  adam_betas:
  - 0.9
  - 0.95
  weight_decay: 0.01
  max_grad_norm: 1.0
  num_epochs: null
  max_steps: 6000
  batch_size: 24
  gradient_accumulation_steps: 1
  lr_scheduler: cosine
  warmup_ratio: 0.05
  lr_min_ratio: 0.01
  action_lr: null
  video_lr: null
  mixed_precision: bf16
  zero_stage: 2
  use_gradient_checkpointing: true
  use_gradient_checkpointing_offload: false
  initialize_model_on_cpu: false
  offload_optimizer_device: none
  lambda_video: 1.0
  lambda_action: 1.0
  max_timestep_boundary: 1.0
  min_timestep_boundary: 0.0
  output_path: outputs/openwam_checkpoints
  save_steps: 2000
  save_full_states_for_resume: false
  keep_last_k_ckpts: 3
  finetune_ckpt_path: /path/to/OpenWAM-Pretrain-Foundation-Model
  resume_ckpt_path: null
  dataset_num_workers: 8
project:
  name: openwam
  output_dir: outputs/${now:%Y-%m-%d_%H-%M-%S}
  seed: 42
  wandb:
    project: openwam
    run_name: vlabench_run2_mutual
    entity: null