Pretrained_OpenWAM / config.yaml
wayrise's picture
Upload folder using huggingface_hub
e699578 verified
Raw
History Blame Contribute Delete
7.63 kB
model:
video_backbone:
encoder:
name: wan_vae
model_path: /mnt/data/wangyuran/Wan2.2-TI2V-5B
name: wan22_ti2v_5b
model_path: /mnt/data/wangyuran/Wan2.2-TI2V-5B
from_scratch: false
shift_video: 5.0
components:
- attr: dit
model_class: openwam.model.video_backbone.wan.models.dit.WanModel
extra_kwargs:
has_image_input: false
patch_size:
- 1
- 2
- 2
in_dim: 48
dim: 3072
ffn_dim: 14336
freq_dim: 256
text_dim: 4096
out_dim: 48
num_heads: 24
num_layers: 30
eps: 1.0e-06
seperated_timestep: true
require_clip_embedding: false
require_vae_embedding: false
fuse_vae_embedding_in_latents: true
- attr: vae
model_class: openwam.model.video_backbone.wan.models.vae.WanVideoVAE38
extra_kwargs: {}
- attr: text_encoder
model_class: openwam.model.video_backbone.wan.models.text_encoder.WanTextEncoder
extra_kwargs: {}
tokenizer:
class: openwam.model.video_backbone.wan.models.text_encoder.HuggingfaceTokenizer
attr: tokenizer
subdir: tokenizer/google/umt5-xxl
path_kwarg: name
kwargs:
seq_len: 512
clean: whitespace
action_backbone:
dim: 1024
ffn_dim: 4096
shift_action: 5.0
freeze:
- video_backbone.text_encoder
- video_backbone.reason1
- video_backbone.vae
- video_backbone.image_encoder
- video_backbone.video_encoder
architecture:
framework: dual_system
variant: joint_self_attn
action_dim: 80
use_proprioception: true
state_dim: 80
bridge_layers: null
bridge_interval: 1
mot_checkpoint_mixed_attn: true
attention_mask_mode: action_sees_video
video_attention_mask_mode: first_frame_causal
detach_bridge: false
idm_video_cond_noise_prob: 0.5
dataloader:
datasets:
robocoin:
type: robocoin
dataset_dir: /mnt/data/wangyuran/RoboCOIN
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
multiview: true
split: train
color_jitter:
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
normalize_mode: quantile
enable_action_supervision: true
unify_action: true
unify_action_map:
- 0-9
- 34-43
total_hours: 621.0
enabled: true
weight: 1.0
oxe_droid:
type: oxe_droid
dataset_dir: /mnt/data/wangyuran/OXE/DROID-Dataset
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
multiview: true
head_camera_choices:
- observation.images.exterior_1_left
- observation.images.exterior_2_left
split: train
color_jitter:
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
normalize_mode: quantile
enable_action_supervision: true
unify_action: true
unify_action_map:
- 0-9
- 34-43
total_hours: 213.4
enabled: true
weight: 1.0
agibotworld:
type: agibotworld
dataset_dir: /mnt/data/wangyuran/AgiBotWorld-Beta
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
multiview: true
split: train
color_jitter:
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
normalize_mode: quantile
enable_action_supervision: true
unify_action: true
total_hours: 1504.6
enabled: true
weight: 1.0
egodex:
type: egodex
dataset_dir: /mnt/data/wangyuran/EgoDex
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
target_camera: observation.images.ego
multiview: true
camera_layout:
- observation.images.ego
- __missing_left__
- __missing_right__
split: train
color_jitter:
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
unify_action: true
total_hours: 536.3
enabled: true
weight: 1.0
ego4d:
type: ego4d
dataset_dir: /mnt/data/wangyuran/Ego4D
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
target_camera: observation.images.ego
multiview: true
camera_layout:
- observation.images.ego
- __missing_left__
- __missing_right__
split: train
color_jitter:
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
unify_action: true
total_hours: 216.3
enabled: true
weight: 1.0
haiyu:
type: haiyu
dataset_dir: /mnt/data/wangyuran/Haiyu
manifest_path: null
rebuild_manifest: false
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
target_camera: observation.images.ego
multiview: true
camera_layout:
- observation.images.ego
- __missing_left__
- __missing_right__
split: train
color_jitter:
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
unify_action: true
total_hours: 688.6
enabled: true
weight: 1.0
lightwheel:
type: lightwheel
dataset_dir: /mnt/data/wangyuran/lightwheel
manifest_path: null
rebuild_manifest: false
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
target_camera: observation.images.ego
multiview: true
camera_layout:
- observation.images.ego
- __missing_left__
- __missing_right__
split: train
color_jitter:
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
unify_action: true
total_hours: 327.1
enabled: true
weight: 1.0
worldengine:
type: worldengine
dataset_dir: /mnt/data/wangyuran/WorldEngine
num_frames: 33
video_stride: 4
window_stride: 1
height: 384
width: 320
target_camera: observation.images.ego
multiview: true
camera_layout:
- observation.images.ego
- __missing_left__
- __missing_right__
split: train
color_jitter:
brightness: 0.2
contrast: 0.2
saturation: 0.2
hue: 0.0
unify_action: true
total_hours: 392.7
enabled: true
weight: 1.0
type: mixture
weight_strategy: proportional
seed: 42
training:
debug: false
learning_rate: 0.0001
adam_betas:
- 0.9
- 0.95
weight_decay: 0.01
max_grad_norm: 1.0
num_epochs: 1
max_steps: null
batch_size: 160
gradient_accumulation_steps: 1
lr_scheduler: cosine
warmup_ratio: 0.05
lr_min_ratio: 0.01
action_lr: null
video_lr: null
mixed_precision: bf16
zero_stage: 2
use_gradient_checkpointing: true
use_gradient_checkpointing_offload: false
initialize_model_on_cpu: false
offload_optimizer_device: none
lambda_video: 1.0
lambda_action: 1.0
max_timestep_boundary: 1.0
min_timestep_boundary: 0.0
timestep_sampling: default
timestep_sampling_lead: video
timestep_sampling_alpha: 1.0
output_path: /mnt/data/wangyuran/openwam_checkpoints
save_steps: 2000
keep_last_k_ckpts: 1
finetune_ckpt_path: null
resume_ckpt_path: null
dataset_num_workers: 16
project:
name: open-wam
output_dir: outputs/${now:%Y-%m-%d_%H-%M-%S}
seed: 42
wandb:
project: open-wam
run_name: null
entity: null