model: _target_: flexpi.runtime.create_flexpi model_id: Wan-AI/Wan2.2-TI2V-5B tokenizer_model_id: Wan-AI/Wan2.1-T2V-1.3B tokenizer_max_len: 128 load_text_encoder: false proprio_dim: 14 redirect_common_files: true mot_checkpoint_mixed_attn: false action_dit_pretrained_path: checkpoints/ActionDiT_linear_interp_Wan22_alphascale_1024hdim.pt skip_dit_load_from_pretrain: false dino_dim: 768 dino_model_name: vit_base_patch16_dinov3.lvd1689m dino_cam_patches: null dino_cam_regions: null dino_temporal_stride: 1 freeze_dino_encoder: true pointmap_max_depth_m: 2.0 pointmap_norm_bounds: x: - -0.5 - 0.5 'y': - -0.5 - 0.5 z: - 0.0 - 1.5 pointmap_depth_vis_mode: turbo joint_video: true joint_dino: true joint_pointmap: true flex_joint: enabled: true p_present_video: 0.5 p_present_dino: 0.5 p_present_pointmap: 0.5 p_jv: 0.5 p_jd: 0.5 p_jp: 0.5 cross_modal_predict_video: true cross_modal_predict_dino: true cross_modal_predict_pointmap: true video_dit_config: has_image_input: false patch_size: - 1 - 2 - 2 in_dim: 48 hidden_dim: 3072 ffn_dim: 14336 freq_dim: 256 text_dim: 4096 out_dim: 48 num_heads: 24 attn_head_dim: 128 num_layers: 30 eps: 1.0e-06 seperated_timestep: true require_clip_embedding: false require_vae_embedding: false fuse_vae_embedding_in_latents: true use_gradient_checkpointing: false video_attention_mask_mode: first_frame_causal action_conditioned: false action_dim: 14 action_group_causal_mask_mode: group_diagonal action_dit_config: action_dim: 14 hidden_dim: 1024 ffn_dim: 4096 num_heads: 24 attn_head_dim: 128 num_layers: 30 text_dim: 4096 freq_dim: 256 eps: 1.0e-06 use_gradient_checkpointing: false video_scheduler: train_shift: 6.0 infer_shift: 6.0 num_train_timesteps: 1000 action_scheduler: train_shift: 1.0 infer_shift: 1.0 num_train_timesteps: 1000 dino_scheduler: train_shift: 6.0 infer_shift: 6.0 num_train_timesteps: 1000 pointmap_scheduler: train_shift: 6.0 infer_shift: 6.0 num_train_timesteps: 1000 loss: lambda_action: 1.0 lambda_dino: 1.0 lambda_pointmap: 1.0 hbridge: enabled: true bottom_ratio: 0.25 top_ratio: 0.25 data: train: _target_: flexpi.datasets.lerobot.robot_video_dataset.RobotVideoDataset processor: _target_: flexpi.datasets.lerobot.processors.flexpi_processor.FlexPiProcessor shape_meta: images: - key: cam_high raw_shape: - 3 - 240 - 320 shape: - 3 - 240 - 320 - key: cam_left_wrist raw_shape: - 3 - 240 - 320 shape: - 3 - 240 - 320 - key: cam_right_wrist raw_shape: - 3 - 240 - 320 shape: - 3 - 240 - 320 depth: - key: cam_high raw_shape: - 1 - 240 - 320 shape: - 1 - 240 - 320 - key: cam_left_wrist raw_shape: - 1 - 240 - 320 shape: - 1 - 240 - 320 - key: cam_right_wrist raw_shape: - 1 - 240 - 320 shape: - 1 - 240 - 320 action: - key: default raw_shape: 14 shape: 14 state: - key: default raw_shape: 14 shape: 14 num_obs_steps: 33 num_output_cameras: 3 action_output_dim: 14 proprio_output_dim: 14 action_state_transforms: null use_stepwise_action_norm: false norm_default_mode: z-score norm_exception_mode: null action_state_merger: _target_: flexpi.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign train_transforms: - _target_: flexpi.datasets.lerobot.transforms.image.ToTensor - _target_: torchvision.transforms.Resize size: - 240 - 320 val_transforms: - _target_: flexpi.datasets.lerobot.transforms.image.ToTensor - _target_: torchvision.transforms.Resize size: - 240 - 320 num_frames: 33 action_video_freq_ratio: 4 shape_meta: images: - key: cam_high raw_shape: - 3 - 240 - 320 shape: - 3 - 240 - 320 - key: cam_left_wrist raw_shape: - 3 - 240 - 320 shape: - 3 - 240 - 320 - key: cam_right_wrist raw_shape: - 3 - 240 - 320 shape: - 3 - 240 - 320 depth: - key: cam_high raw_shape: - 1 - 240 - 320 shape: - 1 - 240 - 320 - key: cam_left_wrist raw_shape: - 1 - 240 - 320 shape: - 1 - 240 - 320 - key: cam_right_wrist raw_shape: - 1 - 240 - 320 shape: - 1 - 240 - 320 action: - key: default raw_shape: 14 shape: 14 state: - key: default raw_shape: 14 shape: 14 video_size: - 384 - 320 concat_multi_camera: robotwin