Robotics
LeRobot
Safetensors
lingbot_va
File size: 2,607 Bytes
f610ebd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
{
    "type": "lingbot_va",
    "n_obs_steps": 1,
    "input_features": {
        "observation.images.cam_high": {
            "type": "VISUAL",
            "shape": [
                3,
                256,
                256
            ]
        },
        "observation.images.cam_left_wrist": {
            "type": "VISUAL",
            "shape": [
                3,
                256,
                256
            ]
        },
        "observation.images.cam_right_wrist": {
            "type": "VISUAL",
            "shape": [
                3,
                256,
                256
            ]
        }
    },
    "output_features": {
        "action": {
            "type": "ACTION",
            "shape": [
                6
            ]
        }
    },
    "device": "cuda",
    "use_amp": false,
    "use_peft": false,
    "push_to_hub": true,
    "repo_id": "maximellerbach/omx_multicubes_lingbot_lowres",
    "private": null,
    "tags": null,
    "license": null,
    "pretrained_path": "maximellerbach/omx_multicubes_lingbot_va",
    "pretrained_revision": null,
    "patch_size": [
        1,
        2,
        2
    ],
    "num_attention_heads": 24,
    "attention_head_dim": 128,
    "in_channels": 48,
    "out_channels": 48,
    "action_dim": 30,
    "text_dim": 4096,
    "freq_dim": 256,
    "ffn_dim": 14336,
    "num_layers": 30,
    "cross_attn_norm": true,
    "eps": 1e-06,
    "rope_max_seq_len": 1024,
    "attn_mode": "flex",
    "wan_pretrained_path": "robbyant/lingbot-va-base",
    "dtype": "bfloat16",
    "text_encoder_device": "cuda",
    "obs_cam_keys": [
        "observation.images.cam_high",
        "observation.images.cam_left_wrist"
    ],
    "image_hflip": false,
    "camera_layout": "width_concat",
    "height": 128,
    "width": 160,
    "action_per_frame": 16,
    "frame_chunk_size": 2,
    "attn_window": 72,
    "num_inference_steps": 25,
    "video_exec_step": -1,
    "action_num_inference_steps": 50,
    "guidance_scale": 5.0,
    "action_guidance_scale": 1.0,
    "snr_shift": 5.0,
    "action_snr_shift": 1.0,
    "max_sequence_length": 512,
    "used_action_channel_ids": [
        0,
        1,
        2,
        3,
        4,
        5
    ],
    "save_predicted_video": false,
    "normalization_mapping": {
        "VISUAL": "IDENTITY",
        "STATE": "IDENTITY",
        "ACTION": "QUANTILES"
    },
    "optimizer_lr": 1e-05,
    "optimizer_betas": [
        0.9,
        0.95
    ],
    "optimizer_eps": 1e-08,
    "optimizer_weight_decay": 0.0001,
    "optimizer_grad_clip_norm": 1.0,
    "scheduler_warmup_steps": 1000
}