flexpi-robotwin / config.yaml
lastgeyan's picture
Flex-pi RoboTwin 2.0 checkpoint
87d3833
Raw
History Blame Contribute Delete
5.64 kB
model:
_target_: flexpi.runtime.create_flexpi
model_id: Wan-AI/Wan2.2-TI2V-5B
tokenizer_model_id: Wan-AI/Wan2.1-T2V-1.3B
tokenizer_max_len: 128
load_text_encoder: false
proprio_dim: 14
redirect_common_files: true
mot_checkpoint_mixed_attn: false
action_dit_pretrained_path: checkpoints/ActionDiT_linear_interp_Wan22_alphascale_1024hdim.pt
skip_dit_load_from_pretrain: false
dino_dim: 768
dino_model_name: vit_base_patch16_dinov3.lvd1689m
dino_cam_patches: null
dino_cam_regions: null
dino_temporal_stride: 1
freeze_dino_encoder: true
pointmap_max_depth_m: 2.0
pointmap_norm_bounds:
x:
- -0.5
- 0.5
'y':
- -0.5
- 0.5
z:
- 0.0
- 1.5
pointmap_depth_vis_mode: turbo
joint_video: true
joint_dino: true
joint_pointmap: true
flex_joint:
enabled: true
p_present_video: 0.5
p_present_dino: 0.5
p_present_pointmap: 0.5
p_jv: 0.5
p_jd: 0.5
p_jp: 0.5
cross_modal_predict_video: true
cross_modal_predict_dino: true
cross_modal_predict_pointmap: true
video_dit_config:
has_image_input: false
patch_size:
- 1
- 2
- 2
in_dim: 48
hidden_dim: 3072
ffn_dim: 14336
freq_dim: 256
text_dim: 4096
out_dim: 48
num_heads: 24
attn_head_dim: 128
num_layers: 30
eps: 1.0e-06
seperated_timestep: true
require_clip_embedding: false
require_vae_embedding: false
fuse_vae_embedding_in_latents: true
use_gradient_checkpointing: false
video_attention_mask_mode: first_frame_causal
action_conditioned: false
action_dim: 14
action_group_causal_mask_mode: group_diagonal
action_dit_config:
action_dim: 14
hidden_dim: 1024
ffn_dim: 4096
num_heads: 24
attn_head_dim: 128
num_layers: 30
text_dim: 4096
freq_dim: 256
eps: 1.0e-06
use_gradient_checkpointing: false
video_scheduler:
train_shift: 6.0
infer_shift: 6.0
num_train_timesteps: 1000
action_scheduler:
train_shift: 1.0
infer_shift: 1.0
num_train_timesteps: 1000
dino_scheduler:
train_shift: 6.0
infer_shift: 6.0
num_train_timesteps: 1000
pointmap_scheduler:
train_shift: 6.0
infer_shift: 6.0
num_train_timesteps: 1000
loss:
lambda_action: 1.0
lambda_dino: 1.0
lambda_pointmap: 1.0
hbridge:
enabled: true
bottom_ratio: 0.25
top_ratio: 0.25
data:
train:
_target_: flexpi.datasets.lerobot.robot_video_dataset.RobotVideoDataset
processor:
_target_: flexpi.datasets.lerobot.processors.flexpi_processor.FlexPiProcessor
shape_meta:
images:
- key: cam_high
raw_shape:
- 3
- 240
- 320
shape:
- 3
- 240
- 320
- key: cam_left_wrist
raw_shape:
- 3
- 240
- 320
shape:
- 3
- 240
- 320
- key: cam_right_wrist
raw_shape:
- 3
- 240
- 320
shape:
- 3
- 240
- 320
depth:
- key: cam_high
raw_shape:
- 1
- 240
- 320
shape:
- 1
- 240
- 320
- key: cam_left_wrist
raw_shape:
- 1
- 240
- 320
shape:
- 1
- 240
- 320
- key: cam_right_wrist
raw_shape:
- 1
- 240
- 320
shape:
- 1
- 240
- 320
action:
- key: default
raw_shape: 14
shape: 14
state:
- key: default
raw_shape: 14
shape: 14
num_obs_steps: 33
num_output_cameras: 3
action_output_dim: 14
proprio_output_dim: 14
action_state_transforms: null
use_stepwise_action_norm: false
norm_default_mode: z-score
norm_exception_mode: null
action_state_merger:
_target_: flexpi.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign
train_transforms:
- _target_: flexpi.datasets.lerobot.transforms.image.ToTensor
- _target_: torchvision.transforms.Resize
size:
- 240
- 320
val_transforms:
- _target_: flexpi.datasets.lerobot.transforms.image.ToTensor
- _target_: torchvision.transforms.Resize
size:
- 240
- 320
num_frames: 33
action_video_freq_ratio: 4
shape_meta:
images:
- key: cam_high
raw_shape:
- 3
- 240
- 320
shape:
- 3
- 240
- 320
- key: cam_left_wrist
raw_shape:
- 3
- 240
- 320
shape:
- 3
- 240
- 320
- key: cam_right_wrist
raw_shape:
- 3
- 240
- 320
shape:
- 3
- 240
- 320
depth:
- key: cam_high
raw_shape:
- 1
- 240
- 320
shape:
- 1
- 240
- 320
- key: cam_left_wrist
raw_shape:
- 1
- 240
- 320
shape:
- 1
- 240
- 320
- key: cam_right_wrist
raw_shape:
- 1
- 240
- 320
shape:
- 1
- 240
- 320
action:
- key: default
raw_shape: 14
shape: 14
state:
- key: default
raw_shape: 14
shape: 14
video_size:
- 384
- 320
concat_multi_camera: robotwin