File size: 3,513 Bytes
72054bc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 | model:
video_backbone:
encoder:
name: wan_vae
model_path: /mnt/data/wangyuran/Wan2.2-TI2V-5B
name: wan22_ti2v_5b
model_path: /mnt/data/wangyuran/Wan2.2-TI2V-5B
from_scratch: false
shift_video: 5.0
components:
- attr: dit
model_class: openwam.model.video_backbone.wan.models.dit.WanModel
extra_kwargs:
has_image_input: false
patch_size:
- 1
- 2
- 2
in_dim: 48
dim: 3072
ffn_dim: 14336
freq_dim: 256
text_dim: 4096
out_dim: 48
num_heads: 24
num_layers: 30
eps: 1.0e-06
seperated_timestep: true
require_clip_embedding: false
require_vae_embedding: false
fuse_vae_embedding_in_latents: true
- attr: vae
model_class: openwam.model.video_backbone.wan.models.vae.WanVideoVAE38
extra_kwargs: {}
- attr: text_encoder
model_class: openwam.model.video_backbone.wan.models.text_encoder.WanTextEncoder
extra_kwargs: {}
tokenizer:
class: openwam.model.video_backbone.wan.models.text_encoder.HuggingfaceTokenizer
attr: tokenizer
subdir: tokenizer/google/umt5-xxl
path_kwarg: name
kwargs:
seq_len: 512
clean: whitespace
action_backbone:
dim: 1024
ffn_dim: 4096
shift_action: 5.0
freeze:
- video_backbone.text_encoder
- video_backbone.reason1
- video_backbone.vae
- video_backbone.image_encoder
- video_backbone.video_encoder
architecture:
framework: dual_system
variant: joint_self_attn
action_dim: 80
use_proprioception: true
state_dim: 80
bridge_layers: null
bridge_interval: 1
mot_checkpoint_mixed_attn: true
attention_mask_mode: action_sees_video
video_attention_mask_mode: first_frame_causal
detach_bridge: false
idm_video_cond_noise_prob: 0.5
dataloader:
type: wuji_real_task
dataset_dir: /mnt/data/liangjiaqi/wuji_sft_data_eef
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
multiview: true
target_camera: observation.images.stereo
camera_layout:
- observation.images.stereo
- observation.images.cam_left_wrist
- observation.images.cam_right_wrist
normalize_mode: quantile
split: train
enable_action_supervision: true
unify_action: true
unify_action_map:
- 0-8
- 10-29
- 32-40
- 42-61
seed: 42
training:
debug: false
learning_rate: 0.0001
adam_betas:
- 0.9
- 0.95
weight_decay: 0.01
max_grad_norm: 1.0
num_epochs: null
max_steps: 100000
batch_size: 2
gradient_accumulation_steps: 1
lr_scheduler: cosine
warmup_ratio: 0.05
lr_min_ratio: 0.01
action_lr: null
video_lr: null
mixed_precision: bf16
zero_stage: 2
use_gradient_checkpointing: false
use_gradient_checkpointing_offload: false
initialize_model_on_cpu: false
offload_optimizer_device: none
lambda_video: 1.0
lambda_action: 1.0
max_timestep_boundary: 1.0
min_timestep_boundary: 0.0
timestep_sampling: default
timestep_sampling_lead: video
timestep_sampling_alpha: 1.0
output_path: /mnt/data/liangjiaqi/openwam_checkpoints
save_steps: 2000
keep_last_k_ckpts: 1
finetune_ckpt_path: /mnt/data/wangyuran/openwam_checkpoints/pretrained_final_dualsystem_jointselfattention
resume_ckpt_path: null
dataset_num_workers: 8
project:
name: open-wam
output_dir: outputs/${now:%Y-%m-%d_%H-%M-%S}
seed: 42
wandb:
project: open-wam
run_name: wuji-eef-sft
entity: null
|