{ "dataset": { "repo_id": "lerobot-data-collection/level2_final_quality3_t_0_hil_data_c", "root": "/fsx/martino/.cache/huggingface/hub/datasets--lerobot-data-collection--level2_final_quality3_t_0_hil_data_c/snapshots/2496db53d330c360f910d095e13698d968c56fc6/", "episodes": null, "image_transforms": { "enable": false, "max_num_transforms": 3, "random_order": false, "tfs": { "brightness": { "weight": 1.0, "type": "ColorJitter", "kwargs": { "brightness": [ 0.8, 1.2 ] } }, "contrast": { "weight": 1.0, "type": "ColorJitter", "kwargs": { "contrast": [ 0.8, 1.2 ] } }, "saturation": { "weight": 1.0, "type": "ColorJitter", "kwargs": { "saturation": [ 0.5, 1.5 ] } }, "hue": { "weight": 1.0, "type": "ColorJitter", "kwargs": { "hue": [ -0.05, 0.05 ] } }, "sharpness": { "weight": 1.0, "type": "SharpnessJitter", "kwargs": { "sharpness": [ 0.5, 1.5 ] } }, "affine": { "weight": 1.0, "type": "RandomAffine", "kwargs": { "degrees": [ -5.0, 5.0 ], "translate": [ 0.05, 0.05 ] } } } }, "revision": null, "use_imagenet_stats": false, "video_backend": "pyav", "return_uint8": false, "depth_output_unit": "mm", "streaming": false, "eval_split": 0.0 }, "env": null, "policy": { "type": "fastwam", "n_obs_steps": 1, "input_features": { "observation.images.base": { "type": "VISUAL", "shape": [ 3, 224, 224 ] }, "observation.images.left_wrist": { "type": "VISUAL", "shape": [ 3, 224, 224 ] }, "observation.images.right_wrist": { "type": "VISUAL", "shape": [ 3, 224, 224 ] }, "observation.state": { "type": "STATE", "shape": [ 16 ] } }, "output_features": { "action": { "type": "ACTION", "shape": [ 16 ] } }, "device": "cuda", "use_amp": false, "use_peft": false, "push_to_hub": true, "repo_id": "nepyope/folding_fastwam", "private": null, "tags": null, "license": null, "pretrained_path": "lerobot/fastwam_base", "pretrained_revision": null, "action_dim": 16, "proprio_dim": 16, "action_horizon": 32, "n_action_steps": 32, "num_video_frames": 33, "action_video_freq_ratio": 4, "image_size": [ 224, 672 ], "context_len": 128, "use_relative_actions": true, "relative_exclude_joints": [ "gripper" ], "action_feature_names": [ "right_joint_1.pos", "right_joint_2.pos", "right_joint_3.pos", "right_joint_4.pos", "right_joint_5.pos", "right_joint_6.pos", "right_joint_7.pos", "right_gripper.pos", "left_joint_1.pos", "left_joint_2.pos", "left_joint_3.pos", "left_joint_4.pos", "left_joint_5.pos", "left_joint_6.pos", "left_joint_7.pos", "left_gripper.pos" ], "model_id": "Wan-AI/Wan2.2-TI2V-5B", "tokenizer_model_id": "google/umt5-xxl", "text_encoder_model_id": "Wan-AI/Wan2.2-TI2V-5B-Diffusers", "base_model_id": "lerobot/fastwam_base", "tokenizer_max_len": 128, "load_text_encoder": true, "mot_checkpoint_mixed_attn": false, "torch_dtype": "bfloat16", "prompt_template": "A video recorded from a robot's point of view executing the following instruction: {task}", "num_inference_steps": 10, "inference_seed": 42, "rand_device": "cpu", "text_cfg_scale": 1.0, "negative_prompt": "", "sigma_shift": null, "tiled": false, "fp32_attention": false, "use_gradient_checkpointing": true, "freeze_video_expert": false, "toggle_action_dimensions": [], "video_scheduler": { "train_shift": 5.0, "infer_shift": 5.0, "num_train_timesteps": 1000 }, "action_scheduler": { "train_shift": 5.0, "infer_shift": 5.0, "num_train_timesteps": 1000 }, "loss": { "lambda_video": 1.0, "lambda_action": 1.0 }, "video_dit_config": { "patch_size": [ 1, 2, 2 ], "in_dim": 48, "hidden_dim": 3072, "ffn_dim": 14336, "freq_dim": 256, "text_dim": 4096, "out_dim": 48, "num_heads": 24, "attn_head_dim": 128, "num_layers": 30, "eps": 1e-06, "seperated_timestep": true, "use_gradient_checkpointing": true, "video_attention_mask_mode": "first_frame_causal", "action_conditioned": false, "action_dim": 16, "action_group_causal_mask_mode": "group_diagonal", "fp32_attention": false }, "action_dit_config": { "action_dim": 16, "hidden_dim": 1024, "ffn_dim": 4096, "num_heads": 24, "attn_head_dim": 128, "num_layers": 30, "text_dim": 4096, "freq_dim": 256, "eps": 1e-06, "use_gradient_checkpointing": true, "fp32_attention": false }, "normalization_mapping": { "VISUAL": "IDENTITY", "STATE": "MEAN_STD", "ACTION": "MEAN_STD" }, "optimizer_lr": 0.0001, "optimizer_weight_decay": 0.01 }, "reward_model": null, "output_dir": "/fsx/martino/runs/folding-tshirt/outputs/train/folding_fastwam_multinode_22372363", "job_name": "folding_fastwam_multinode", "resume": false, "seed": 1000, "cudnn_deterministic": false, "num_workers": 4, "batch_size": 16, "prefetch_factor": 4, "persistent_workers": true, "steps": 66686, "env_eval_freq": 20000, "log_freq": 50, "eval_steps": 0, "max_eval_samples": 0, "tolerance_s": 0.0001, "save_checkpoint": true, "save_freq": 10000, "use_policy_training_preset": true, "optimizer": { "type": "adamw", "lr": 0.0001, "weight_decay": 0.01, "grad_clip_norm": 10.0, "betas": [ 0.9, 0.999 ], "eps": 1e-08 }, "scheduler": null, "eval": { "n_episodes": 50, "batch_size": 50, "use_async_envs": true, "recording": false, "recording_repo_id": null, "recording_private": false, "save_predicted_video": false }, "wandb": { "enable": true, "disable_artifact": true, "project": "folding-tshirt", "entity": null, "notes": null, "run_id": "1l3pzcio", "mode": null, "add_tags": true }, "peft": null, "job": { "target": null, "image": "huggingface/lerobot-gpu:latest", "timeout": "2d", "detach": false, "tags": [] }, "save_checkpoint_to_hub": false, "sample_weighting": { "type": "rabc", "progress_path": "/fsx/martino/.cache/huggingface/hub/datasets--lerobot-data-collection--level2_final_quality3_t_0_hil_data_c/snapshots/2496db53d330c360f910d095e13698d968c56fc6/sarm_progress.parquet", "head_mode": "sparse", "kappa": 0.05, "epsilon": 1e-06, "extra_params": {} }, "rename_map": {}, "checkpoint_path": null }