File size: 3,418 Bytes
8c1bcd6 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 | model:
video_backbone:
encoder:
name: wan22_vae
model_path: /path/to/Wan2.2-TI2V-5B
name: wan22_ti2v_5b
model_path: /path/to/Wan2.2-TI2V-5B
from_scratch: false
shift_video: 5.0
components:
- attr: dit
model_class: openwam.model.video_backbone.wan.models.dit.WanModel
extra_kwargs:
has_image_input: false
patch_size:
- 1
- 2
- 2
in_dim: 48
dim: 3072
ffn_dim: 14336
freq_dim: 256
text_dim: 4096
out_dim: 48
num_heads: 24
num_layers: 30
eps: 1.0e-06
seperated_timestep: true
require_clip_embedding: false
require_vae_embedding: false
fuse_vae_embedding_in_latents: true
- attr: vae
model_class: openwam.model.video_backbone.wan.models.vae.WanVideoVAE38
extra_kwargs: {}
- attr: text_encoder
model_class: openwam.model.video_backbone.wan.models.text_encoder.WanTextEncoder
extra_kwargs: {}
tokenizer:
class: openwam.model.video_backbone.wan.models.text_encoder.HuggingfaceTokenizer
attr: tokenizer
subdir: tokenizer/google/umt5-xxl
path_kwarg: name
kwargs:
seq_len: 512
clean: whitespace
action_backbone:
dim: 1024
ffn_dim: 4096
shift_action: 5.0
freeze:
- video_backbone.text_encoder
- video_backbone.reason1
- video_backbone.vae
- video_backbone.image_encoder
- video_backbone.video_encoder
architecture:
framework: dual_system
variant: joint_self_attn
action_dim: 80
use_proprioception: true
state_dim: 80
bridge_layers: null
bridge_interval: 1
mot_checkpoint_mixed_attn: true
attention_mask_mode: mutual
video_attention_mask_mode: first_frame_causal
detach_bridge: false
idm_video_cond_noise_prob: 0.5
dataloader:
type: ebench
dataset_dir: /path/to/ebench_data
action_mode: eef
unify_action: true
unify_action_map:
- 0-9
- 34-43
- 68-70
unify_state_map: null
num_frames: 33
video_stride: 4
window_stride: 1
split: train
height: 384
width: 320
multiview: true
camera_layout:
- video.overlook_camera_view
- video.left_camera_view
- video.right_camera_view
target_camera: video.overlook_camera_view
normalize_mode: min-max
color_jitter:
enabled: true
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
seed: 42
training:
debug: false
learning_rate: 0.0001
adam_betas:
- 0.9
- 0.95
weight_decay: 0.01
max_grad_norm: 1.0
num_epochs: null
max_steps: 100000
batch_size: 2
gradient_accumulation_steps: 1
lr_scheduler: cosine
warmup_ratio: 0.05
lr_min_ratio: 0.01
action_lr: null
video_lr: null
mixed_precision: bf16
zero_stage: 2
use_gradient_checkpointing: false
use_gradient_checkpointing_offload: false
initialize_model_on_cpu: false
offload_optimizer_device: none
lambda_video: 1.0
lambda_action: 1.0
max_timestep_boundary: 1.0
min_timestep_boundary: 0.0
output_path: outputs/openwam_checkpoints
save_steps: 2000
save_full_states_for_resume: false
keep_last_k_ckpts: 10
finetune_ckpt_path: /path/to/OpenWAM-Pretrain-Foundation-Model
resume_ckpt_path: null
dataset_num_workers: 8
project:
name: openwam
output_dir: outputs/${now:%Y-%m-%d_%H-%M-%S}
seed: 42
wandb:
project: openwam
run_name: null
entity: null
|