| model: |
| _target_: flexpi.runtime.create_flexpi |
| model_id: Wan-AI/Wan2.2-TI2V-5B |
| tokenizer_model_id: Wan-AI/Wan2.1-T2V-1.3B |
| tokenizer_max_len: 128 |
| load_text_encoder: false |
| proprio_dim: 14 |
| redirect_common_files: true |
| mot_checkpoint_mixed_attn: false |
| action_dit_pretrained_path: checkpoints/ActionDiT_linear_interp_Wan22_alphascale_1024hdim.pt |
| skip_dit_load_from_pretrain: false |
| dino_dim: 768 |
| dino_model_name: vit_base_patch16_dinov3.lvd1689m |
| dino_cam_patches: null |
| dino_cam_regions: null |
| dino_temporal_stride: 1 |
| freeze_dino_encoder: true |
| pointmap_max_depth_m: 2.0 |
| pointmap_norm_bounds: |
| x: |
| - -0.5 |
| - 0.5 |
| 'y': |
| - -0.5 |
| - 0.5 |
| z: |
| - 0.0 |
| - 1.5 |
| pointmap_depth_vis_mode: turbo |
| joint_video: true |
| joint_dino: true |
| joint_pointmap: true |
| flex_joint: |
| enabled: true |
| p_present_video: 0.5 |
| p_present_dino: 0.5 |
| p_present_pointmap: 0.5 |
| p_jv: 0.5 |
| p_jd: 0.5 |
| p_jp: 0.5 |
| cross_modal_predict_video: true |
| cross_modal_predict_dino: true |
| cross_modal_predict_pointmap: true |
| video_dit_config: |
| has_image_input: false |
| patch_size: |
| - 1 |
| - 2 |
| - 2 |
| in_dim: 48 |
| hidden_dim: 3072 |
| ffn_dim: 14336 |
| freq_dim: 256 |
| text_dim: 4096 |
| out_dim: 48 |
| num_heads: 24 |
| attn_head_dim: 128 |
| num_layers: 30 |
| eps: 1.0e-06 |
| seperated_timestep: true |
| require_clip_embedding: false |
| require_vae_embedding: false |
| fuse_vae_embedding_in_latents: true |
| use_gradient_checkpointing: false |
| video_attention_mask_mode: first_frame_causal |
| action_conditioned: false |
| action_dim: 14 |
| action_group_causal_mask_mode: group_diagonal |
| action_dit_config: |
| action_dim: 14 |
| hidden_dim: 1024 |
| ffn_dim: 4096 |
| num_heads: 24 |
| attn_head_dim: 128 |
| num_layers: 30 |
| text_dim: 4096 |
| freq_dim: 256 |
| eps: 1.0e-06 |
| use_gradient_checkpointing: false |
| video_scheduler: |
| train_shift: 6.0 |
| infer_shift: 6.0 |
| num_train_timesteps: 1000 |
| action_scheduler: |
| train_shift: 1.0 |
| infer_shift: 1.0 |
| num_train_timesteps: 1000 |
| dino_scheduler: |
| train_shift: 6.0 |
| infer_shift: 6.0 |
| num_train_timesteps: 1000 |
| pointmap_scheduler: |
| train_shift: 6.0 |
| infer_shift: 6.0 |
| num_train_timesteps: 1000 |
| loss: |
| lambda_action: 1.0 |
| lambda_dino: 1.0 |
| lambda_pointmap: 1.0 |
| hbridge: |
| enabled: true |
| bottom_ratio: 0.25 |
| top_ratio: 0.25 |
| data: |
| train: |
| _target_: flexpi.datasets.lerobot.robot_video_dataset.RobotVideoDataset |
| processor: |
| _target_: flexpi.datasets.lerobot.processors.flexpi_processor.FlexPiProcessor |
| shape_meta: |
| images: |
| - key: cam_high |
| raw_shape: |
| - 3 |
| - 240 |
| - 320 |
| shape: |
| - 3 |
| - 240 |
| - 320 |
| - key: cam_left_wrist |
| raw_shape: |
| - 3 |
| - 240 |
| - 320 |
| shape: |
| - 3 |
| - 240 |
| - 320 |
| - key: cam_right_wrist |
| raw_shape: |
| - 3 |
| - 240 |
| - 320 |
| shape: |
| - 3 |
| - 240 |
| - 320 |
| depth: |
| - key: cam_high |
| raw_shape: |
| - 1 |
| - 240 |
| - 320 |
| shape: |
| - 1 |
| - 240 |
| - 320 |
| - key: cam_left_wrist |
| raw_shape: |
| - 1 |
| - 240 |
| - 320 |
| shape: |
| - 1 |
| - 240 |
| - 320 |
| - key: cam_right_wrist |
| raw_shape: |
| - 1 |
| - 240 |
| - 320 |
| shape: |
| - 1 |
| - 240 |
| - 320 |
| action: |
| - key: default |
| raw_shape: 14 |
| shape: 14 |
| state: |
| - key: default |
| raw_shape: 14 |
| shape: 14 |
| num_obs_steps: 33 |
| num_output_cameras: 3 |
| action_output_dim: 14 |
| proprio_output_dim: 14 |
| action_state_transforms: null |
| use_stepwise_action_norm: false |
| norm_default_mode: z-score |
| norm_exception_mode: null |
| action_state_merger: |
| _target_: flexpi.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign |
| train_transforms: |
| - _target_: flexpi.datasets.lerobot.transforms.image.ToTensor |
| - _target_: torchvision.transforms.Resize |
| size: |
| - 240 |
| - 320 |
| val_transforms: |
| - _target_: flexpi.datasets.lerobot.transforms.image.ToTensor |
| - _target_: torchvision.transforms.Resize |
| size: |
| - 240 |
| - 320 |
| num_frames: 33 |
| action_video_freq_ratio: 4 |
| shape_meta: |
| images: |
| - key: cam_high |
| raw_shape: |
| - 3 |
| - 240 |
| - 320 |
| shape: |
| - 3 |
| - 240 |
| - 320 |
| - key: cam_left_wrist |
| raw_shape: |
| - 3 |
| - 240 |
| - 320 |
| shape: |
| - 3 |
| - 240 |
| - 320 |
| - key: cam_right_wrist |
| raw_shape: |
| - 3 |
| - 240 |
| - 320 |
| shape: |
| - 3 |
| - 240 |
| - 320 |
| depth: |
| - key: cam_high |
| raw_shape: |
| - 1 |
| - 240 |
| - 320 |
| shape: |
| - 1 |
| - 240 |
| - 320 |
| - key: cam_left_wrist |
| raw_shape: |
| - 1 |
| - 240 |
| - 320 |
| shape: |
| - 1 |
| - 240 |
| - 320 |
| - key: cam_right_wrist |
| raw_shape: |
| - 1 |
| - 240 |
| - 320 |
| shape: |
| - 1 |
| - 240 |
| - 320 |
| action: |
| - key: default |
| raw_shape: 14 |
| shape: 14 |
| state: |
| - key: default |
| raw_shape: 14 |
| shape: 14 |
| video_size: |
| - 384 |
| - 320 |
| concat_multi_camera: robotwin |
|
|