pack3_base / config.yaml
m1ku2's picture
Upload config.yaml with huggingface_hub
1479675 verified
Raw
History Blame Contribute Delete
8.07 kB
output_dir: ./runs/pack_3_objects_plus_base/2026-06-08_23-08-07
batch_size: 8
num_workers: 2
lr_scheduler_type: cosine
learning_rate: 0.0001
num_epochs: 100
max_steps: 20000
log_every: 10
save_every: 2000
eval_every: 500
eval_num_inference_steps: 10
gradient_accumulation_steps: 1
mixed_precision: bf16
seed: 42
max_grad_norm: 1.0
weight_decay: 0.01
resume: null
partial_resume: null
wandb:
enabled: true
workspace: null
project: fast-wam-real
name: pack_3_objects_plus_base
group: null
mode: online
data:
train:
_target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset
dataset_dirs:
- ./data/pack_3_objects_plus/perfect_lerobot
shape_meta:
images:
- key: cam_high
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
- key: cam_left_wrist
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
- key: cam_right_wrist
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
action:
- key: default
raw_shape: 14
shape: 14
state:
- key: default
raw_shape: 14
shape: 14
num_frames: 33
global_sample_stride: 1
action_video_freq_ratio: 4
video_size:
- 384
- 320
camera_key: null
val_set_proportion: 0.01
is_training_set: true
pretrained_norm_stats: null
skip_padding_as_possible: false
concat_multi_camera: robotwin
processor:
_target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor
shape_meta:
images:
- key: cam_high
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
- key: cam_left_wrist
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
- key: cam_right_wrist
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
action:
- key: default
raw_shape: 14
shape: 14
state:
- key: default
raw_shape: 14
shape: 14
num_obs_steps: 33
num_output_cameras: 3
action_output_dim: 14
proprio_output_dim: 14
action_state_transforms: null
use_stepwise_action_norm: false
norm_default_mode: z-score
norm_exception_mode: null
action_state_merger:
_target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign
train_transforms:
- _target_: fastwam.datasets.lerobot.transforms.image.ToTensor
- _target_: torchvision.transforms.Resize
size:
- 240
- 320
val_transforms:
- _target_: fastwam.datasets.lerobot.transforms.image.ToTensor
- _target_: torchvision.transforms.Resize
size:
- 240
- 320
image_augmentation: true
train_aug_transforms:
cam_high:
- _target_: torchvision.transforms.RandomCrop
size:
- 228
- 304
- _target_: torchvision.transforms.Resize
size:
- 240
- 320
- _target_: torchvision.transforms.RandomRotation
degrees: 5
- _target_: torchvision.transforms.ColorJitter
brightness: 0.3
contrast: 0.4
saturation: 0.5
cam_left_wrist:
- _target_: torchvision.transforms.ColorJitter
brightness: 0.3
contrast: 0.4
saturation: 0.5
cam_right_wrist:
- _target_: torchvision.transforms.ColorJitter
brightness: 0.3
contrast: 0.4
saturation: 0.5
text_embedding_cache_dir: ./data/text_embeds_cache/pack_3_objects_plus
context_len: 128
val:
_target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset
dataset_dirs:
- ./data/pack_3_objects_plus/perfect_lerobot
shape_meta:
images:
- key: cam_high
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
- key: cam_left_wrist
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
- key: cam_right_wrist
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
action:
- key: default
raw_shape: 14
shape: 14
state:
- key: default
raw_shape: 14
shape: 14
num_frames: 33
global_sample_stride: 1
action_video_freq_ratio: 4
video_size:
- 384
- 320
camera_key: null
val_set_proportion: 0.01
is_training_set: false
pretrained_norm_stats: null
skip_padding_as_possible: false
concat_multi_camera: robotwin
processor:
_target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor
shape_meta:
images:
- key: cam_high
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
- key: cam_left_wrist
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
- key: cam_right_wrist
raw_shape:
- 3
- 480
- 640
shape:
- 3
- 240
- 320
action:
- key: default
raw_shape: 14
shape: 14
state:
- key: default
raw_shape: 14
shape: 14
num_obs_steps: 33
num_output_cameras: 3
action_output_dim: 14
proprio_output_dim: 14
action_state_transforms: null
use_stepwise_action_norm: false
norm_default_mode: z-score
norm_exception_mode: null
action_state_merger:
_target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign
train_transforms:
- _target_: fastwam.datasets.lerobot.transforms.image.ToTensor
- _target_: torchvision.transforms.Resize
size:
- 240
- 320
val_transforms:
- _target_: fastwam.datasets.lerobot.transforms.image.ToTensor
- _target_: torchvision.transforms.Resize
size:
- 240
- 320
text_embedding_cache_dir: ./data/text_embeds_cache/pack_3_objects_plus
context_len: 128
model:
_target_: fastwam.runtime.create_fastwam
model_id: Wan-AI/Wan2.2-TI2V-5B
tokenizer_model_id: Wan-AI/Wan2.2-TI2V-5B
dit_checkpoint_path: null
tokenizer_max_len: 128
load_text_encoder: false
proprio_dim: 14
redirect_common_files: false
mot_checkpoint_mixed_attn: false
action_dit_pretrained_path: checkpoints/ActionDiT_linear_interp_Wan22_alphascale_1024hdim.pt
skip_dit_load_from_pretrain: false
video_dit_config:
has_image_input: false
patch_size:
- 1
- 2
- 2
in_dim: 48
hidden_dim: 3072
ffn_dim: 14336
freq_dim: 256
text_dim: 4096
out_dim: 48
num_heads: 24
attn_head_dim: 128
num_layers: 30
eps: 1.0e-06
seperated_timestep: true
require_clip_embedding: false
require_vae_embedding: false
fuse_vae_embedding_in_latents: true
use_gradient_checkpointing: false
video_attention_mask_mode: first_frame_causal
action_conditioned: false
action_dim: 14
action_group_causal_mask_mode: group_diagonal
action_dit_config:
action_dim: 14
hidden_dim: 1024
ffn_dim: 4096
num_heads: 24
attn_head_dim: 128
num_layers: 30
text_dim: 4096
freq_dim: 256
eps: 1.0e-06
use_gradient_checkpointing: false
video_scheduler:
train_shift: 5.0
infer_shift: 5.0
num_train_timesteps: 1000
action_scheduler:
train_shift: 5.0
infer_shift: 5.0
num_train_timesteps: 1000
loss:
lambda_action: 1.0