| { |
| "type": "fastwam", |
| "n_obs_steps": 1, |
| "input_features": { |
| "observation.images.base": { |
| "type": "VISUAL", |
| "shape": [ |
| 3, |
| 224, |
| 224 |
| ] |
| }, |
| "observation.images.left_wrist": { |
| "type": "VISUAL", |
| "shape": [ |
| 3, |
| 224, |
| 224 |
| ] |
| }, |
| "observation.images.right_wrist": { |
| "type": "VISUAL", |
| "shape": [ |
| 3, |
| 224, |
| 224 |
| ] |
| }, |
| "observation.state": { |
| "type": "STATE", |
| "shape": [ |
| 16 |
| ] |
| } |
| }, |
| "output_features": { |
| "action": { |
| "type": "ACTION", |
| "shape": [ |
| 16 |
| ] |
| } |
| }, |
| "device": "cuda", |
| "use_amp": false, |
| "use_peft": false, |
| "push_to_hub": true, |
| "repo_id": "nepyope/folding_fastwam", |
| "private": null, |
| "tags": null, |
| "license": null, |
| "pretrained_path": null, |
| "pretrained_revision": null, |
| "action_dim": 16, |
| "proprio_dim": 16, |
| "action_horizon": 32, |
| "n_action_steps": 32, |
| "num_video_frames": 33, |
| "action_video_freq_ratio": 4, |
| "image_size": [ |
| 224, |
| 672 |
| ], |
| "context_len": 128, |
| "use_relative_actions": true, |
| "relative_exclude_joints": [ |
| "gripper" |
| ], |
| "action_feature_names": [ |
| "right_joint_1.pos", |
| "right_joint_2.pos", |
| "right_joint_3.pos", |
| "right_joint_4.pos", |
| "right_joint_5.pos", |
| "right_joint_6.pos", |
| "right_joint_7.pos", |
| "right_gripper.pos", |
| "left_joint_1.pos", |
| "left_joint_2.pos", |
| "left_joint_3.pos", |
| "left_joint_4.pos", |
| "left_joint_5.pos", |
| "left_joint_6.pos", |
| "left_joint_7.pos", |
| "left_gripper.pos" |
| ], |
| "model_id": "Wan-AI/Wan2.2-TI2V-5B", |
| "tokenizer_model_id": "google/umt5-xxl", |
| "text_encoder_model_id": "Wan-AI/Wan2.2-TI2V-5B-Diffusers", |
| "base_model_id": "lerobot/fastwam_base", |
| "tokenizer_max_len": 128, |
| "load_text_encoder": true, |
| "mot_checkpoint_mixed_attn": false, |
| "torch_dtype": "bfloat16", |
| "prompt_template": "A video recorded from a robot's point of view executing the following instruction: {task}", |
| "num_inference_steps": 10, |
| "inference_seed": 42, |
| "rand_device": "cpu", |
| "text_cfg_scale": 1.0, |
| "negative_prompt": "", |
| "sigma_shift": null, |
| "tiled": false, |
| "fp32_attention": false, |
| "use_gradient_checkpointing": true, |
| "freeze_video_expert": false, |
| "toggle_action_dimensions": [], |
| "video_scheduler": { |
| "train_shift": 5.0, |
| "infer_shift": 5.0, |
| "num_train_timesteps": 1000 |
| }, |
| "action_scheduler": { |
| "train_shift": 5.0, |
| "infer_shift": 5.0, |
| "num_train_timesteps": 1000 |
| }, |
| "loss": { |
| "lambda_video": 1.0, |
| "lambda_action": 1.0 |
| }, |
| "video_dit_config": { |
| "patch_size": [ |
| 1, |
| 2, |
| 2 |
| ], |
| "in_dim": 48, |
| "hidden_dim": 3072, |
| "ffn_dim": 14336, |
| "freq_dim": 256, |
| "text_dim": 4096, |
| "out_dim": 48, |
| "num_heads": 24, |
| "attn_head_dim": 128, |
| "num_layers": 30, |
| "eps": 1e-06, |
| "seperated_timestep": true, |
| "use_gradient_checkpointing": true, |
| "video_attention_mask_mode": "first_frame_causal", |
| "action_conditioned": false, |
| "action_dim": 16, |
| "action_group_causal_mask_mode": "group_diagonal", |
| "fp32_attention": false |
| }, |
| "action_dit_config": { |
| "action_dim": 16, |
| "hidden_dim": 1024, |
| "ffn_dim": 4096, |
| "num_heads": 24, |
| "attn_head_dim": 128, |
| "num_layers": 30, |
| "text_dim": 4096, |
| "freq_dim": 256, |
| "eps": 1e-06, |
| "use_gradient_checkpointing": true, |
| "fp32_attention": false |
| }, |
| "normalization_mapping": { |
| "VISUAL": "IDENTITY", |
| "STATE": "MEAN_STD", |
| "ACTION": "MEAN_STD" |
| }, |
| "optimizer_lr": 0.0001, |
| "optimizer_weight_decay": 0.01 |
| } |