fruit-picking-fastwam / training_code /model_config.yaml
SleepMastger's picture
add model card, conditioning, and training-time processing
7896def verified
Raw
History Blame Contribute Delete
2.36 kB
# Model config matching the ORIGINAL FastWAM LIBERO release checkpoint
# checkpoints/fastwam_release/libero_uncond_2cam224.pt: non-decoupled MoT,
# video expert 30 layers (hidden 3072) + action expert 30 layers (hidden 1024),
# no KV fusion. Architecture fields copied verbatim from the resolved model
# block of runs/libero/Baseline_Frozen/2026-06-29_19-46-04/config.yaml (whose
# checkpoint is key/shape-identical to the release .pt) with two recipe fields
# set to the RELEASE provenance, not the frozen ablation's:
# freeze_video_backbone: false (release video weights deviate from the
# Wan2.2 base by ~0.05-0.065 -> the video expert WAS trained)
# loss.lambda_video: 1.0 (video loss on, as in the paper / M1 run)
# mot_checkpoint_mixed_attn is only a gradient-checkpointing toggle (no effect
# on weights) — set true if the run OOMs.
_target_: fastwam.runtime.create_fastwam
model_id: Wan-AI/Wan2.2-TI2V-5B
tokenizer_model_id: Wan-AI/Wan2.1-T2V-1.3B
tokenizer_max_len: 128
load_text_encoder: false
proprio_dim: ${data.train.processor.proprio_output_dim}
redirect_common_files: true
mot_checkpoint_mixed_attn: false
# harmless init: every mot key is overwritten by resume=<release .pt>
action_dit_pretrained_path: checkpoints/ActionDiT_linear_interp_Wan22_alphascale_1024hdim.pt
skip_dit_load_from_pretrain: false
freeze_video_backbone: false
video_dit_config:
has_image_input: false
patch_size: [1, 2, 2]
in_dim: 48
hidden_dim: 3072
ffn_dim: 14336
freq_dim: 256
text_dim: 4096
out_dim: 48
num_heads: 24
attn_head_dim: 128
num_layers: 30
eps: 1.0e-06
seperated_timestep: true
require_clip_embedding: false
require_vae_embedding: false
fuse_vae_embedding_in_latents: true
use_gradient_checkpointing: false
video_attention_mask_mode: first_frame_causal
action_conditioned: false
action_dim: 7
action_group_causal_mask_mode: group_diagonal
action_dit_config:
action_dim: 7
hidden_dim: 1024
ffn_dim: 4096
num_heads: 24
attn_head_dim: 128
num_layers: 30
text_dim: 4096
freq_dim: 256
eps: 1.0e-06
use_gradient_checkpointing: false
action_text_cross_attn: true
video_scheduler:
train_shift: 5.0
infer_shift: 5.0
num_train_timesteps: 1000
action_scheduler:
train_shift: 5.0
infer_shift: 5.0
num_train_timesteps: 1000
loss:
lambda_action: 1.0
lambda_video: 1.0