Robotics
Transformers
Safetensors
English
3d-detection
vision-language-action
pose-estimation
grounding
Instructions to use hetolin/PoseVLA-stage1 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use hetolin/PoseVLA-stage1 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("hetolin/PoseVLA-stage1", device_map="auto") - Notebooks
- Google Colab
- Kaggle
| { | |
| "type": "pi0_ours", | |
| "n_obs_steps": 1, | |
| "normalization_mapping": { | |
| "VISUAL": "IDENTITY", | |
| "STATE": "IDENTITY", | |
| "ACTION": "IDENTITY" | |
| }, | |
| "input_features": {}, | |
| "output_features": {}, | |
| "chunk_size": 50, | |
| "n_action_steps": 50, | |
| "max_state_dim": 32, | |
| "max_action_dim": 32, | |
| "resize_imgs_with_padding": [ | |
| 224, | |
| 224 | |
| ], | |
| "empty_cameras": 0, | |
| "tokenizer_max_length": 512, | |
| "tokenizer_model_path": "google/paligemma-3b-pt-224", | |
| "proj_width": 1024, | |
| "num_steps": 10, | |
| "use_cache": true, | |
| "attention_implementation": "eager", | |
| "pi05": false, | |
| "is_knowledge_insulation": false, | |
| "add_extra_token": true, | |
| "add_image_token": true, | |
| "add_prior": true, | |
| "freeze_vision_encoder": false, | |
| "train_expert_only": false, | |
| "train_state_proj": true, | |
| "optimizer_lr": 5e-05, | |
| "optimizer_betas": [ | |
| 0.9, | |
| 0.95 | |
| ], | |
| "optimizer_eps": 1e-08, | |
| "optimizer_weight_decay": 1e-10, | |
| "scheduler_warmup_steps": 16000, | |
| "scheduler_decay_steps": 1280000, | |
| "scheduler_decay_lr": 5e-06, | |
| "vis_attn": true, | |
| "skip_init_weights": false, | |
| "device": "cpu" | |
| } |