# Model config matching the FlashWAM M1_FusedKV_RopeFixed LIBERO run's final # checkpoint runs/libero/M1_FusedKV_RopeFixed/2026-07-03_16-13-56/ # checkpoints/weights/step_021700.pt: FasterWAM decoupled MoT with a 1-layer # action expert, fused_kv KV source (kv_fusing_layer + k/v channel # projections), and the action-aligned video RoPE fix (fixed_rope: true -> # MoTDecoupledActionAlignedVideoRoPE). # Copied VERBATIM from that run's resolved config.yaml model block — the # architecture must match the checkpoint exactly for the finetune leg, and the # scratch leg reuses the identical architecture for a clean comparison. # NOTE action_dit_pretrained_path is null exactly as in the LIBERO run: fused # KV modes cannot use layer-selective init, so with a null path the 1-layer # action expert starts RANDOM (fasterwam_decoupled.py from_wan22_pretrained); # the video expert always initializes from the Wan2.2-TI2V-5B base. _target_: fastwam.models.wan22.fasterwam_decoupled.create_fasterwam_decoupled model_id: Wan-AI/Wan2.2-TI2V-5B tokenizer_model_id: Wan-AI/Wan2.1-T2V-1.3B tokenizer_max_len: 128 load_text_encoder: false proprio_dim: 8 redirect_common_files: true mot_checkpoint_mixed_attn: false action_dit_pretrained_path: null skip_dit_load_from_pretrain: false decoupled: true kv_source_mode: fused_kv fusion_hidden_dim: 64 fusion_use_norm: true fixed_rope: true video_dit_config: has_image_input: false patch_size: - 1 - 2 - 2 in_dim: 48 hidden_dim: 3072 ffn_dim: 14336 freq_dim: 256 text_dim: 4096 out_dim: 48 num_heads: 24 attn_head_dim: 128 num_layers: 30 eps: 1.0e-06 seperated_timestep: true require_clip_embedding: false require_vae_embedding: false fuse_vae_embedding_in_latents: true use_gradient_checkpointing: false video_attention_mask_mode: first_frame_causal action_conditioned: false action_dim: 7 action_group_causal_mask_mode: group_diagonal action_dit_config: action_dim: 7 hidden_dim: 1024 ffn_dim: 4096 num_heads: 24 attn_head_dim: 128 num_layers: 1 layer_selected: null text_dim: 4096 freq_dim: 256 eps: 1.0e-06 use_gradient_checkpointing: false action_text_cross_attn: true video_scheduler: train_shift: 5.0 infer_shift: 5.0 num_train_timesteps: 1000 action_scheduler: train_shift: 5.0 infer_shift: 5.0 num_train_timesteps: 1000 loss: lambda_video: 1.0 lambda_action: 1.0