Upload OpenVLA grasp model checkpoint

Browse files

Files changed (16) hide show

action_head--10000_checkpoint.pt +3 -0
added_tokens.json +3 -0
config.json +415 -0
dataset_statistics.json +244 -0
lora_adapter/README.md +202 -0
lora_adapter/adapter_config.json +45 -0
lora_adapter/adapter_model.safetensors +3 -0
preprocessor_config.json +114 -0
processing_prismatic.py +257 -0
processor_config.json +6 -0
proprio_projector--10000_checkpoint.pt +3 -0
special_tokens_map.json +30 -0
tokenizer.json +0 -0
tokenizer.model +3 -0
tokenizer_config.json +53 -0
vision_backbone--10000_checkpoint.pt +3 -0

action_head--10000_checkpoint.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:70fcbbfc3fc492ecb7245c82e61116e6c6e7e76fd0dd435bb8e37c9be1d766a0
+size 604453726

added_tokens.json ADDED Viewed

	@@ -0,0 +1,3 @@

+{
+  "<PAD>": 32000
+}

config.json ADDED Viewed

	@@ -0,0 +1,415 @@

+{
+  "norm_stats": {
+    "gen72_grasp_stacking_baskets": {
+      "action": {
+        "mean": [
+          71.42466735839844,
+          84.38833618164062,
+          -82.9383316040039,
+          -85.52547454833984,
+          -4.3938679695129395,
+          19.08216094970703,
+          0.308868408203125,
+          67.15239715576172,
+          -80.31741333007812,
+          84.10924530029297,
+          86.23428344726562,
+          -86.4471206665039,
+          3.9559240341186523,
+          9.594415664672852,
+          4.726510524749756,
+          82.21428680419922
+        ],
+        "std": [
+          25.26982879638672,
+          15.808586120605469,
+          20.259897232055664,
+          21.52882957458496,
+          20.819337844848633,
+          20.531246185302734,
+          26.965940475463867,
+          45.217315673828125,
+          27.104999542236328,
+          12.850805282592773,
+          14.046908378601074,
+          17.79497718811035,
+          13.857259750366211,
+          17.137197494506836,
+          17.71614646911621,
+          36.83033752441406
+        ],
+        "max": [
+          124.62999725341797,
+          102.0,
+          12.920000076293945,
+          52.0,
+          119.18000030517578,
+          89.77999877929688,
+          169.0,
+          100.0,
+          3.9600000381469727,
+          102.0,
+          123.30999755859375,
+          52.0,
+          73.56500244140625,
+          66.08999633789062,
+          86.5250015258789,
+          100.0
+        ],
+        "min": [
+          5.84499979019165,
+          -38.275001525878906,
+          -169.0,
+          -97.47000122070312,
+          -113.19999694824219,
+          -65.30000305175781,
+          -139.75,
+          0.0,
+          -121.68499755859375,
+          5.84499979019165,
+          -8.654999732971191,
+          -98.16999816894531,
+          -55.459999084472656,
+          -78.44499969482422,
+          -92.29000091552734,
+          0.0
+        ],
+        "q01": [
+          11.34000015258789,
+          35.595001220703125,
+          -127.60875129699707,
+          -96.5,
+          -68.33499908447266,
+          -37.86875057220459,
+          -82.83499908447266,
+          0.0,
+          -109.64374732971191,
+          33.56999969482422,
+          33.582499504089355,
+          -96.94000244140625,
+          -22.323750495910645,
+          -32.78499984741211,
+          -46.05500030517578,
+          0.0
+        ],
+        "q99": [
+          103.84500122070312,
+          102.0,
+          -23.510000228881836,
+          24.047500133514404,
+          58.91999912261963,
+          69.83000183105469,
+          63.060001373291016,
+          100.0,
+          -10.866249561309814,
+          102.0,
+          108.41500091552734,
+          2.14000004529953,
+          54.625,
+          51.900001525878906,
+          61.040000915527344,
+          100.0
+        ],
+        "mask": [
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true
+        ]
+      },
+      "proprio": {
+        "mean": [
+          71.39665222167969,
+          84.42801666259766,
+          -82.94181823730469,
+          -85.53479766845703,
+          -4.377756118774414,
+          19.067413330078125,
+          0.31575945019721985,
+          67.15303802490234,
+          -80.27192687988281,
+          84.11940002441406,
+          86.19857788085938,
+          -86.43647766113281,
+          3.944101572036743,
+          9.555686950683594,
+          4.720884323120117,
+          82.21434783935547
+        ],
+        "std": [
+          24.659215927124023,
+          15.453649520874023,
+          19.847105026245117,
+          21.23726463317871,
+          20.293472290039062,
+          19.495216369628906,
+          26.316822052001953,
+          45.21790313720703,
+          26.583589553833008,
+          12.41045093536377,
+          13.648248672485352,
+          17.389333724975586,
+          13.373165130615234,
+          16.093711853027344,
+          17.135786056518555,
+          36.83036804199219
+        ],
+        "max": [
+          124.08000183105469,
+          102.05000305175781,
+          10.850000381469727,
+          52.0099983215332,
+          109.37000274658203,
+          87.58000183105469,
+          168.66000366210938,
+          100.0,
+          3.8399999141693115,
+          102.01000213623047,
+          119.8499984741211,
+          41.31999969482422,
+          68.77999877929688,
+          65.19999694824219,
+          82.41999816894531,
+          100.0
+        ],
+        "min": [
+          6.940000057220459,
+          -35.029998779296875,
+          -169.00999450683594,
+          -97.37999725341797,
+          -111.45999908447266,
+          -62.58000183105469,
+          -136.9199981689453,
+          0.0,
+          -120.04000091552734,
+          10.140000343322754,
+          -3.2699999809265137,
+          -97.66000366210938,
+          -54.2599983215332,
+          -64.5,
+          -89.81999969482422,
+          0.0
+        ],
+        "q01": [
+          11.512500286102295,
+          36.560001373291016,
+          -126.5099983215332,
+          -96.41000366210938,
+          -66.58749771118164,
+          -35.62750053405762,
+          -81.84500122070312,
+          0.0,
+          -109.31999969482422,
+          37.10499858856201,
+          34.77500057220459,
+          -96.91999816894531,
+          -21.260000228881836,
+          -29.440000534057617,
+          -45.08750057220459,
+          0.0
+        ],
+        "q99": [
+          102.95499992370605,
+          102.01000213623047,
+          -25.047500610351562,
+          23.78499937057495,
+          56.52000045776367,
+          67.02999877929688,
+          61.95750045776367,
+          100.0,
+          -11.315000295639038,
+          102.01000213623047,
+          107.51000213623047,
+          0.6450000107288361,
+          53.470001220703125,
+          49.709999084472656,
+          59.13999938964844,
+          100.0
+        ]
+      },
+      "num_transitions": 74826,
+      "num_trajectories": 200
+    }
+  },
+  "n_action_bins": 256,
+  "vision_backbone_id": "dinosiglip-vit-so-224px",
+  "llm_backbone_id": "llama2-7b-pure",
+  "arch_specifier": "no-align+fused-gelu-mlp",
+  "output_projector_states": false,
+  "use_fused_vision_backbone": true,
+  "timm_model_ids": [
+    "vit_large_patch14_reg4_dinov2.lvd142m",
+    "vit_so400m_patch14_siglip_224"
+  ],
+  "timm_override_act_layers": [
+    null,
+    null
+  ],
+  "image_sizes": [
+    224,
+    224
+  ],
+  "image_resize_strategy": "resize-naive",
+  "hf_llm_id": "meta-llama/Llama-2-7b-hf",
+  "llm_max_length": 2048,
+  "pad_token_id": 32000,
+  "pad_to_multiple_of": 64,
+  "text_config": {
+    "vocab_size": 32064,
+    "max_position_embeddings": 2048,
+    "hidden_size": 4096,
+    "intermediate_size": 11008,
+    "num_hidden_layers": 32,
+    "num_attention_heads": 32,
+    "num_key_value_heads": 32,
+    "hidden_act": "silu",
+    "initializer_range": 0.02,
+    "rms_norm_eps": 1e-06,
+    "pretraining_tp": 1,
+    "use_cache": true,
+    "rope_theta": 10000.0,
+    "rope_scaling": null,
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "return_dict": true,
+    "output_hidden_states": false,
+    "output_attentions": false,
+    "torchscript": false,
+    "torch_dtype": "bfloat16",
+    "use_bfloat16": false,
+    "tf_legacy_loss": false,
+    "pruned_heads": {},
+    "tie_word_embeddings": false,
+    "chunk_size_feed_forward": 0,
+    "is_encoder_decoder": false,
+    "is_decoder": false,
+    "cross_attention_hidden_size": null,
+    "add_cross_attention": false,
+    "tie_encoder_decoder": false,
+    "max_length": 20,
+    "min_length": 0,
+    "do_sample": false,
+    "early_stopping": false,
+    "num_beams": 1,
+    "num_beam_groups": 1,
+    "diversity_penalty": 0.0,
+    "temperature": 1.0,
+    "top_k": 50,
+    "top_p": 1.0,
+    "typical_p": 1.0,
+    "repetition_penalty": 1.0,
+    "length_penalty": 1.0,
+    "no_repeat_ngram_size": 0,
+    "encoder_no_repeat_ngram_size": 0,
+    "bad_words_ids": null,
+    "num_return_sequences": 1,
+    "output_scores": false,
+    "return_dict_in_generate": false,
+    "forced_bos_token_id": null,
+    "forced_eos_token_id": null,
+    "remove_invalid_values": false,
+    "exponential_decay_length_penalty": null,
+    "suppress_tokens": null,
+    "begin_suppress_tokens": null,
+    "architectures": null,
+    "finetuning_task": null,
+    "id2label": {
+      "0": "LABEL_0",
+      "1": "LABEL_1"
+    },
+    "label2id": {
+      "LABEL_0": 0,
+      "LABEL_1": 1
+    },
+    "tokenizer_class": null,
+    "prefix": null,
+    "bos_token_id": 1,
+    "pad_token_id": 32000,
+    "eos_token_id": 2,
+    "sep_token_id": null,
+    "decoder_start_token_id": null,
+    "task_specific_params": null,
+    "problem_type": null,
+    "_name_or_path": "",
+    "model_type": "llama"
+  },
+  "return_dict": true,
+  "output_hidden_states": false,
+  "output_attentions": false,
+  "torchscript": false,
+  "torch_dtype": "bfloat16",
+  "use_bfloat16": false,
+  "tf_legacy_loss": false,
+  "pruned_heads": {},
+  "tie_word_embeddings": true,
+  "chunk_size_feed_forward": 0,
+  "is_encoder_decoder": false,
+  "is_decoder": false,
+  "cross_attention_hidden_size": null,
+  "add_cross_attention": false,
+  "tie_encoder_decoder": false,
+  "max_length": 20,
+  "min_length": 0,
+  "do_sample": false,
+  "early_stopping": false,
+  "num_beams": 1,
+  "num_beam_groups": 1,
+  "diversity_penalty": 0.0,
+  "temperature": 1.0,
+  "top_k": 50,
+  "top_p": 1.0,
+  "typical_p": 1.0,
+  "repetition_penalty": 1.0,
+  "length_penalty": 1.0,
+  "no_repeat_ngram_size": 0,
+  "encoder_no_repeat_ngram_size": 0,
+  "bad_words_ids": null,
+  "num_return_sequences": 1,
+  "output_scores": false,
+  "return_dict_in_generate": false,
+  "forced_bos_token_id": null,
+  "forced_eos_token_id": null,
+  "remove_invalid_values": false,
+  "exponential_decay_length_penalty": null,
+  "suppress_tokens": null,
+  "begin_suppress_tokens": null,
+  "architectures": [
+    "OpenVLAForActionPrediction"
+  ],
+  "finetuning_task": null,
+  "id2label": {
+    "0": "LABEL_0",
+    "1": "LABEL_1"
+  },
+  "label2id": {
+    "LABEL_0": 0,
+    "LABEL_1": 1
+  },
+  "tokenizer_class": null,
+  "prefix": null,
+  "bos_token_id": null,
+  "eos_token_id": null,
+  "sep_token_id": null,
+  "decoder_start_token_id": null,
+  "task_specific_params": null,
+  "problem_type": null,
+  "_name_or_path": "/home/guangyu/.cache/huggingface/hub/models--openvla--openvla-7b/snapshots/31f090d05236101ebfc381b61c674dd4746d4ce0",
+  "transformers_version": "4.40.1",
+  "auto_map": {
+    "AutoConfig": "configuration_prismatic.OpenVLAConfig",
+    "AutoModelForVision2Seq": "modeling_prismatic.OpenVLAForActionPrediction"
+  },
+  "model_type": "openvla"
+}

dataset_statistics.json ADDED Viewed

	@@ -0,0 +1,244 @@

+{
+  "gen72_grasp_stacking_baskets": {
+    "action": {
+      "mean": [
+        71.42466735839844,
+        84.38833618164062,
+        -82.9383316040039,
+        -85.52547454833984,
+        -4.3938679695129395,
+        19.08216094970703,
+        0.308868408203125,
+        67.15239715576172,
+        -80.31741333007812,
+        84.10924530029297,
+        86.23428344726562,
+        -86.4471206665039,
+        3.9559240341186523,
+        9.594415664672852,
+        4.726510524749756,
+        82.21428680419922
+      ],
+      "std": [
+        25.26982879638672,
+        15.808586120605469,
+        20.259897232055664,
+        21.52882957458496,
+        20.819337844848633,
+        20.531246185302734,
+        26.965940475463867,
+        45.217315673828125,
+        27.104999542236328,
+        12.850805282592773,
+        14.046908378601074,
+        17.79497718811035,
+        13.857259750366211,
+        17.137197494506836,
+        17.71614646911621,
+        36.83033752441406
+      ],
+      "max": [
+        124.62999725341797,
+        102.0,
+        12.920000076293945,
+        52.0,
+        119.18000030517578,
+        89.77999877929688,
+        169.0,
+        100.0,
+        3.9600000381469727,
+        102.0,
+        123.30999755859375,
+        52.0,
+        73.56500244140625,
+        66.08999633789062,
+        86.5250015258789,
+        100.0
+      ],
+      "min": [
+        5.84499979019165,
+        -38.275001525878906,
+        -169.0,
+        -97.47000122070312,
+        -113.19999694824219,
+        -65.30000305175781,
+        -139.75,
+        0.0,
+        -121.68499755859375,
+        5.84499979019165,
+        -8.654999732971191,
+        -98.16999816894531,
+        -55.459999084472656,
+        -78.44499969482422,
+        -92.29000091552734,
+        0.0
+      ],
+      "q01": [
+        11.34000015258789,
+        35.595001220703125,
+        -127.60875129699707,
+        -96.5,
+        -68.33499908447266,
+        -37.86875057220459,
+        -82.83499908447266,
+        0.0,
+        -109.64374732971191,
+        33.56999969482422,
+        33.582499504089355,
+        -96.94000244140625,
+        -22.323750495910645,
+        -32.78499984741211,
+        -46.05500030517578,
+        0.0
+      ],
+      "q99": [
+        103.84500122070312,
+        102.0,
+        -23.510000228881836,
+        24.047500133514404,
+        58.91999912261963,
+        69.83000183105469,
+        63.060001373291016,
+        100.0,
+        -10.866249561309814,
+        102.0,
+        108.41500091552734,
+        2.14000004529953,
+        54.625,
+        51.900001525878906,
+        61.040000915527344,
+        100.0
+      ],
+      "mask": [
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true
+      ]
+    },
+    "proprio": {
+      "mean": [
+        71.39665222167969,
+        84.42801666259766,
+        -82.94181823730469,
+        -85.53479766845703,
+        -4.377756118774414,
+        19.067413330078125,
+        0.31575945019721985,
+        67.15303802490234,
+        -80.27192687988281,
+        84.11940002441406,
+        86.19857788085938,
+        -86.43647766113281,
+        3.944101572036743,
+        9.555686950683594,
+        4.720884323120117,
+        82.21434783935547
+      ],
+      "std": [
+        24.659215927124023,
+        15.453649520874023,
+        19.847105026245117,
+        21.23726463317871,
+        20.293472290039062,
+        19.495216369628906,
+        26.316822052001953,
+        45.21790313720703,
+        26.583589553833008,
+        12.41045093536377,
+        13.648248672485352,
+        17.389333724975586,
+        13.373165130615234,
+        16.093711853027344,
+        17.135786056518555,
+        36.83036804199219
+      ],
+      "max": [
+        124.08000183105469,
+        102.05000305175781,
+        10.850000381469727,
+        52.0099983215332,
+        109.37000274658203,
+        87.58000183105469,
+        168.66000366210938,
+        100.0,
+        3.8399999141693115,
+        102.01000213623047,
+        119.8499984741211,
+        41.31999969482422,
+        68.77999877929688,
+        65.19999694824219,
+        82.41999816894531,
+        100.0
+      ],
+      "min": [
+        6.940000057220459,
+        -35.029998779296875,
+        -169.00999450683594,
+        -97.37999725341797,
+        -111.45999908447266,
+        -62.58000183105469,
+        -136.9199981689453,
+        0.0,
+        -120.04000091552734,
+        10.140000343322754,
+        -3.2699999809265137,
+        -97.66000366210938,
+        -54.2599983215332,
+        -64.5,
+        -89.81999969482422,
+        0.0
+      ],
+      "q01": [
+        11.512500286102295,
+        36.560001373291016,
+        -126.5099983215332,
+        -96.41000366210938,
+        -66.58749771118164,
+        -35.62750053405762,
+        -81.84500122070312,
+        0.0,
+        -109.31999969482422,
+        37.10499858856201,
+        34.77500057220459,
+        -96.91999816894531,
+        -21.260000228881836,
+        -29.440000534057617,
+        -45.08750057220459,
+        0.0
+      ],
+      "q99": [
+        102.95499992370605,
+        102.01000213623047,
+        -25.047500610351562,
+        23.78499937057495,
+        56.52000045776367,
+        67.02999877929688,
+        61.95750045776367,
+        100.0,
+        -11.315000295639038,
+        102.01000213623047,
+        107.51000213623047,
+        0.6450000107288361,
+        53.470001220703125,
+        49.709999084472656,
+        59.13999938964844,
+        100.0
+      ]
+    },
+    "num_transitions": 74826,
+    "num_trajectories": 200
+  }
+}

lora_adapter/README.md ADDED Viewed

	@@ -0,0 +1,202 @@

+---
+base_model: /home/guangyu/.cache/huggingface/hub/models--openvla--openvla-7b/snapshots/31f090d05236101ebfc381b61c674dd4746d4ce0
+library_name: peft
+---
+# Model Card for Model ID
+<!-- Provide a quick summary of what the model is/does. -->
+## Model Details
+### Model Description
+<!-- Provide a longer summary of what this model is. -->
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+### Model Sources [optional]
+<!-- Provide the basic links for the model. -->
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+## Uses
+<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
+### Direct Use
+<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
+[More Information Needed]
+### Downstream Use [optional]
+<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
+[More Information Needed]
+### Out-of-Scope Use
+<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
+[More Information Needed]
+## Bias, Risks, and Limitations
+<!-- This section is meant to convey both technical and sociotechnical limitations. -->
+[More Information Needed]
+### Recommendations
+<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+## How to Get Started with the Model
+Use the code below to get started with the model.
+[More Information Needed]
+## Training Details
+### Training Data
+<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
+[More Information Needed]
+### Training Procedure
+<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
+#### Preprocessing [optional]
+[More Information Needed]
+#### Training Hyperparameters
+- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
+#### Speeds, Sizes, Times [optional]
+<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
+[More Information Needed]
+## Evaluation
+<!-- This section describes the evaluation protocols and provides the results. -->
+### Testing Data, Factors & Metrics
+#### Testing Data
+<!-- This should link to a Dataset Card if possible. -->
+[More Information Needed]
+#### Factors
+<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
+[More Information Needed]
+#### Metrics
+<!-- These are the evaluation metrics being used, ideally with a description of why. -->
+[More Information Needed]
+### Results
+[More Information Needed]
+#### Summary
+## Model Examination [optional]
+<!-- Relevant interpretability work for the model goes here -->
+[More Information Needed]
+## Environmental Impact
+<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+## Technical Specifications [optional]
+### Model Architecture and Objective
+[More Information Needed]
+### Compute Infrastructure
+[More Information Needed]
+#### Hardware
+[More Information Needed]
+#### Software
+[More Information Needed]
+## Citation [optional]
+<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
+**BibTeX:**
+[More Information Needed]
+**APA:**
+[More Information Needed]
+## Glossary [optional]
+<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
+[More Information Needed]
+## More Information [optional]
+[More Information Needed]
+## Model Card Authors [optional]
+[More Information Needed]
+## Model Card Contact
+[More Information Needed]
+### Framework versions
+- PEFT 0.11.1

lora_adapter/adapter_config.json ADDED Viewed

	@@ -0,0 +1,45 @@

+{
+  "alpha_pattern": {},
+  "auto_mapping": {
+    "base_model_class": "OpenVLAForActionPrediction",
+    "parent_library": "transformers_modules.31f090d05236101ebfc381b61c674dd4746d4ce0.modeling_prismatic"
+  },
+  "base_model_name_or_path": "/home/guangyu/.cache/huggingface/hub/models--openvla--openvla-7b/snapshots/31f090d05236101ebfc381b61c674dd4746d4ce0",
+  "bias": "none",
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": "gaussian",
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 16,
+  "lora_dropout": 0.0,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "r": 32,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": [
+    "k_proj",
+    "qkv",
+    "fc1",
+    "up_proj",
+    "q_proj",
+    "lm_head",
+    "fc2",
+    "o_proj",
+    "down_proj",
+    "kv",
+    "proj",
+    "gate_proj",
+    "v_proj",
+    "q",
+    "fc3"
+  ],
+  "task_type": null,
+  "use_dora": false,
+  "use_rslora": false
+}

lora_adapter/adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d10387bc396d7897bdb06b0bd45ee62d2d7204cc122e51e1271b18eefa55e1aa
+size 484467800

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,114 @@

+{
+  "auto_map": {
+    "AutoImageProcessor": "processing_prismatic.PrismaticImageProcessor",
+    "AutoProcessor": "processing_prismatic.PrismaticProcessor"
+  },
+  "image_processor_type": "PrismaticImageProcessor",
+  "image_resize_strategy": "resize-naive",
+  "input_sizes": [
+    [
+      3,
+      224,
+      224
+    ],
+    [
+      3,
+      224,
+      224
+    ]
+  ],
+  "interpolations": [
+    "bicubic",
+    "bicubic"
+  ],
+  "means": [
+    [
+      0.485,
+      0.456,
+      0.406
+    ],
+    [
+      0.5,
+      0.5,
+      0.5
+    ]
+  ],
+  "processor_class": "PrismaticProcessor",
+  "stds": [
+    [
+      0.229,
+      0.224,
+      0.225
+    ],
+    [
+      0.5,
+      0.5,
+      0.5
+    ]
+  ],
+  "tvf_crop_params": [
+    {
+      "output_size": [
+        224,
+        224
+      ]
+    },
+    {
+      "output_size": [
+        224,
+        224
+      ]
+    }
+  ],
+  "tvf_do_letterbox": false,
+  "tvf_letterbox_fill": null,
+  "tvf_normalize_params": [
+    {
+      "inplace": false,
+      "mean": [
+        0.484375,
+        0.455078125,
+        0.40625
+      ],
+      "std": [
+        0.228515625,
+        0.2236328125,
+        0.224609375
+      ]
+    },
+    {
+      "inplace": false,
+      "mean": [
+        0.5,
+        0.5,
+        0.5
+      ],
+      "std": [
+        0.5,
+        0.5,
+        0.5
+      ]
+    }
+  ],
+  "tvf_resize_params": [
+    {
+      "antialias": true,
+      "interpolation": 3,
+      "max_size": null,
+      "size": [
+        224,
+        224
+      ]
+    },
+    {
+      "antialias": true,
+      "interpolation": 3,
+      "max_size": null,
+      "size": [
+        224,
+        224
+      ]
+    }
+  ],
+  "use_fused_vision_backbone": true
+}

processing_prismatic.py ADDED Viewed

	@@ -0,0 +1,257 @@

+"""
+processing_prismatic.py
+HuggingFace-style preprocessor definitions for Prismatic VLMs, inheriting from `ProcessorMixin`. Default configuration
+specifies `siglip-224px+7b`.
+"""
+from typing import Any, ClassVar, List, Optional, Tuple, Union
+import timm.data
+import torch
+import torchvision.transforms.functional as TVF
+from PIL import Image
+from torchvision.transforms import CenterCrop, Compose, Normalize, Resize, ToTensor
+from transformers import PreTrainedTokenizerBase
+from transformers.image_processing_utils import BatchFeature, ImageProcessingMixin
+from transformers.processing_utils import ProcessorMixin
+from transformers.tokenization_utils import PaddingStrategy, PreTokenizedInput, TextInput, TruncationStrategy
+from transformers.utils import TensorType
+# === Image Processing ===
+def letterbox_pad_transform(image: Image.Image, padding_fill_value: Tuple[int, int, int]) -> Image.Image:
+    """Given a PIL.Image, pad to square by adding a symmetric border around the height/width."""
+    (w, h), max_wh = image.size, max(image.size)
+    horizontal_pad, vertical_pad = int((max_wh - w) / 2), int((max_wh - h) / 2)
+    padding = (horizontal_pad, vertical_pad, horizontal_pad, vertical_pad)
+    return TVF.pad(image, padding, fill=padding_fill_value, padding_mode="constant")
+class PrismaticImageProcessor(ImageProcessingMixin):
+    model_input_names: ClassVar[List[str]] = ["pixel_values"]
+    def __init__(
+        self,
+        use_fused_vision_backbone: bool = False,
+        image_resize_strategy: str = "letterbox",
+        input_sizes: Optional[List[Tuple[int, int, int]]] = None,
+        interpolations: Optional[List[str]] = None,
+        means: Optional[List[Tuple[float, float, float]]] = None,
+        stds: Optional[List[Tuple[float, float, float]]] = None,
+        **kwargs: str,
+    ) -> None:
+        """
+        Initialize a PrismaticImageProcessor as a wrapper around a torchvision transform; this transform will be
+        created by TIMM, and edited to follow our custom `image_resize_strategy` logic.
+        @param use_fused_vision_backbone: Boolean indicating single or fused (dual) vision backbone
+        @param image_resize_strategy: Prismatic image resize strategy in < resize-naive | resize-crop | letterbox >
+        @param input_size: [TIMM :: `data_cfg`] Input image size as tuple (channels, width, height)
+        @param interpolation: [TIMM :: `data_cfg`] Interpolation as string (default: "bicubic")
+        @param mean: [TIMM :: `data_cfg`] Normalization mean as float tuple (or two-tuple if `fused_backbone`)
+        @param std: [TIMM :: `data_cfg`] Normalization std as float tuple (or two-tuple if `fused_backbone`)
+        """
+        self.use_fused_vision_backbone = use_fused_vision_backbone
+        self.image_resize_strategy = image_resize_strategy
+        # Handle `None` default values
+        input_sizes = [(3, 224, 224)] if input_sizes is None else input_sizes
+        means = [(0.5, 0.5, 0.5)] if means is None else means
+        stds = [(0.5, 0.5, 0.5)] if stds is None else stds
+        # TIMM `data_cfg` Parameters
+        self.input_sizes, self.interpolations, self.means, self.stds = input_sizes, interpolations, means, stds
+        # Grab torchvision transforms via TIMM =>> need to parse for specific "functional" transform values!
+        self.tvf_resize_params, self.tvf_crop_params, self.tvf_normalize_params = [], [], []
+        self.tvf_do_letterbox, self.tvf_letterbox_fill = False, None
+        for idx in range(len(input_sizes)):
+            transform = timm.data.create_transform(
+                input_size=self.input_sizes[idx],
+                interpolation=self.interpolations[idx],
+                mean=self.means[idx],
+                std=self.stds[idx],
+                crop_pct=1.0,  # Set to 1.0 to ignore cropping (initial Resize sets `input_size`)
+                crop_mode="center",  # Default crop mode -- no-op when `crop_pct == 1.0`
+                is_training=False,  # No image augmentations when loading the transform!
+            )
+            # [Validation] Ensure appropriate transform structure, expected sizes
+            if not (
+                isinstance(transform, Compose)
+                and (len(transform.transforms) == 4)
+                and isinstance(transform.transforms[0], Resize)
+                and isinstance(transform.transforms[1], CenterCrop)
+                and isinstance(transform.transforms[2], ToTensor)
+                and isinstance(transform.transforms[3], Normalize)
+                and (transform.transforms[0].size == self.input_sizes[idx][-1])
+                and (transform.transforms[1].size == self.input_sizes[idx][-2:])
+            ):
+                raise ValueError(f"Unexpected TIMM image transformation structure/sizes: `{transform}`")
+            # HF Image Processors *must* be JSON-serializable; as such, cannot have torchvision. as an attribute.
+            #   => Instead, we're going to parse the transform and call "torchvision.transforms.functional" (`tvf`)
+            resize_t, crop_t, norm_t = transform.transforms[0], transform.transforms[1], transform.transforms[3]
+            self.tvf_resize_params.append(
+                {
+                    "size": resize_t.size,
+                    "interpolation": TVF.pil_modes_mapping[resize_t.interpolation],
+                    "max_size": None,
+                    "antialias": True,
+                }
+            )
+            self.tvf_crop_params.append({"output_size": crop_t.size})
+            self.tvf_normalize_params.append(
+                {
+                    "mean": norm_t.mean.float().numpy().tolist(),
+                    "std": norm_t.std.float().numpy().tolist(),
+                    "inplace": False,
+                }
+            )
+            self.tvf_do_letterbox, self.tvf_letterbox_fill = False, None
+            # Handle Prismatic `image_resize_strategy`
+            if self.image_resize_strategy == "resize-naive":
+                self.tvf_resize_params[idx]["size"] = (resize_t.size, resize_t.size)
+            elif self.image_resize_strategy == "letterbox":
+                self.tvf_do_letterbox, self.tvf_letterbox_fill = True, tuple([int(x * 255) for x in self.means[idx]])
+            elif self.image_resize_strategy == "resize-crop":
+                pass
+            else:
+                raise ValueError(f"Image resize strategy `{self.image_resize_strategy}` is not supported!")
+        # Dispatch **kwargs to super()
+        super().__init__(**kwargs)
+    def apply_transform(self, img: Image.Image) -> torch.Tensor:
+        """Apply `functional` variant of TIMM's Transform = Compose([Resize -> CenterCrop -> ToTensor -> Normalize])"""
+        if self.tvf_do_letterbox:
+            img = letterbox_pad_transform(img, self.tvf_letterbox_fill)
+        # [Contract] Fused Backbones expect "channel-stacked" inputs; we'll unpack on the model side!
+        imgs_t = []
+        for idx in range(len(self.input_sizes)):
+            img_idx = TVF.resize(img, **self.tvf_resize_params[idx])
+            img_idx = TVF.center_crop(img_idx, **self.tvf_crop_params[idx])
+            img_idx_t = TVF.to_tensor(img_idx)
+            img_idx_t = TVF.normalize(img_idx_t, **self.tvf_normalize_params[idx])
+            imgs_t.append(img_idx_t)
+        # [Contract] `imgs_t` is a list of Tensors of shape [3, input_size, input_size]; stack along dim = 0
+        img_t = torch.vstack(imgs_t)
+        return img_t
+    def preprocess(
+        self,
+        images: Union[Image.Image, List[Image.Image]],
+        return_tensors: Optional[Union[str, TensorType]] = None,
+        **_: str,
+    ) -> BatchFeature:
+        """
+        Preprocess an image (or batch of images); note that unlike the `transformers :: BaseImageProcessor` we
+        explicitly only handle PIL.Image.Image instances for simplicity.
+        @param images: A (batch of) PIL.Image.Image instance(s) to preprocess.
+        @param return_tensors: BatchFeature default Tensor format (e.g., "pt" for torch); if None, returns np.ndarray
+        @return: Instance of `transformers :: BatchFeature` with a single key "pixel_values"
+        """
+        if not isinstance(images, list):
+            images = [images]
+        # Apply `self.img_transform` to each image (will return list of torch.Tensors); stack into "batched" Tensor
+        pixel_values = torch.stack([self.apply_transform(img.convert("RGB")) for img in images])
+        # Return BatchFeature =>> note that for compatibility, constructor expects Dict[str, np.ndarray], so we convert
+        return BatchFeature(data={"pixel_values": pixel_values.float().numpy()}, tensor_type=return_tensors)
+    def __call__(self, images: Union[Image.Image, List[Image.Image]], **kwargs) -> BatchFeature:
+        return self.preprocess(images, **kwargs)
+# === PrismaticProcessor =>> Wraps both ImageProcessor and Tokenizer ===
+#   =>> https://github.com/huggingface/transformers/blob/main/src/transformers/models/llava/processing_llava.py
+class PrismaticProcessor(ProcessorMixin):
+    attributes: ClassVar[List[str]] = ["image_processor", "tokenizer"]
+    image_processor_class: str = "AutoImageProcessor"
+    tokenizer_class: str = "AutoTokenizer"
+    def __init__(
+        self,
+        image_processor: Optional[ImageProcessingMixin] = None,
+        tokenizer: Optional[PreTrainedTokenizerBase] = None,
+    ) -> None:
+        super().__init__(image_processor, tokenizer)
+    def __call__(
+        self,
+        text: Union[TextInput, PreTokenizedInput, List[TextInput], List[PreTokenizedInput]],
+        images: Union[Image.Image, List[Image.Image]],
+        padding: Union[bool, str, PaddingStrategy] = False,
+        truncation: Optional[Union[bool, str, TruncationStrategy]] = None,
+        max_length: Optional[int] = None,
+        return_tensors: Optional[Union[str, TensorType]] = TensorType.PYTORCH,
+    ) -> BatchFeature:
+        """
+        Preprocess a given (batch) of text/images for a Prismatic VLM; forwards text to the underlying LLM's tokenizer,
+        forwards images to PrismaticImageProcessor.
+        @param text: The (batch) of text to encode; must be a string or list of strings.
+        @param images: A (batch of) PIL.Image.Image instance(s) to preprocess.
+        @param padding: Sequence padding strategy (if multiple specified) in < True = "longest" | "max_length" | False >
+        @param truncation: Truncation strategy for the output sequences; requires `max_length` to be specified
+        @param max_length: Maximum length (in tokens) to truncate
+        @param return_tensors: Type of return tensors (usually "pt" or TensorType.PYTORCH)
+        @return: BatchFeature with keys for `input_ids`, `attention_mask` and `pixel_values`.
+        """
+        pixel_values = self.image_processor(images, return_tensors=return_tensors)["pixel_values"]
+        text_inputs = self.tokenizer(
+            text, return_tensors=return_tensors, padding=padding, truncation=truncation, max_length=max_length
+        )
+        # [Validate] Need same number of images and text inputs!
+        if pixel_values.shape[0] != text_inputs.input_ids.shape[0]:
+            raise ValueError("Batch is malformed; expected same number of images and text inputs!")
+        return BatchFeature(data={**text_inputs, "pixel_values": pixel_values})
+    # === Tokenizer Dispatch Utilities =>> check `PreTrainedTokenizerBase` for documentation ===
+    def batch_decode(
+        self,
+        sequences: Union[List[int], List[List[int]], torch.Tensor, Any],  # `Any` = np.ndarray | tf.Tensor
+        skip_special_tokens: bool = False,
+        clean_up_tokenization_spaces: Optional[bool] = None,
+        **kwargs: str,
+    ) -> List[str]:
+        return self.tokenizer.batch_decode(
+            sequences=sequences,
+            skip_special_tokens=skip_special_tokens,
+            clean_up_tokenization_spaces=clean_up_tokenization_spaces,
+            **kwargs,
+        )
+    def decode(
+        self,
+        token_ids: Union[int, List[int], torch.Tensor, Any],  # `Any` = np.ndarray | tf.Tensor
+        skip_special_tokens: bool = False,
+        clean_up_tokenization_spaces: Optional[bool] = None,
+        **kwargs: str,
+    ) -> str:
+        return self.tokenizer.decode(
+            token_ids=token_ids,
+            skip_special_tokens=skip_special_tokens,
+            clean_up_tokenization_spaces=clean_up_tokenization_spaces,
+            **kwargs,
+        )
+    @property
+    def model_input_names(self) -> List[str]:
+        tokenizer_input_names = self.tokenizer.model_input_names
+        image_processor_input_names = self.image_processor.model_input_names
+        return list(dict.fromkeys(tokenizer_input_names + image_processor_input_names))

processor_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "auto_map": {
+    "AutoProcessor": "processing_prismatic.PrismaticProcessor"
+  },
+  "processor_class": "PrismaticProcessor"
+}

proprio_projector--10000_checkpoint.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:99a7ecce003d63423fecdf228c69c048975400245c462cc0928d9eee69f1508a
+size 67406320

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<PAD>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
+size 499723

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,53 @@

+{
+  "add_bos_token": true,
+  "add_eos_token": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "32000": {
+      "content": "<PAD>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "auto_map": {
+    "AutoProcessor": "processing_prismatic.PrismaticProcessor"
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "legacy": false,
+  "model_max_length": 2048,
+  "pad_token": "<PAD>",
+  "padding_side": "right",
+  "processor_class": "PrismaticProcessor",
+  "sp_model_kwargs": {},
+  "tokenizer_class": "LlamaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

vision_backbone--10000_checkpoint.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bcf3f0c6c17b8d55ea97db293e408131d043acc1594bb7700ca2179f1411194c
+size 3344957817