Upload OpenVLA grasp model checkpoint

Browse files

Files changed (16) hide show

action_head--22000_checkpoint.pt +3 -0
added_tokens.json +3 -0
config.json +415 -0
dataset_statistics.json +244 -0
lora_adapter/README.md +202 -0
lora_adapter/adapter_config.json +45 -0
lora_adapter/adapter_model.safetensors +3 -0
preprocessor_config.json +114 -0
processing_prismatic.py +257 -0
processor_config.json +6 -0
proprio_projector--22000_checkpoint.pt +3 -0
special_tokens_map.json +30 -0
tokenizer.json +0 -0
tokenizer.model +3 -0
tokenizer_config.json +53 -0
vision_backbone--22000_checkpoint.pt +3 -0

action_head--22000_checkpoint.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f685846e462b77f5b950aa72b6e1262b1b486ca73e3445543bfd0796c8bce598
+size 604453726

added_tokens.json ADDED Viewed

	@@ -0,0 +1,3 @@

+{
+  "<PAD>": 32000
+}

config.json ADDED Viewed

	@@ -0,0 +1,415 @@

+{
+  "norm_stats": {
+    "gen72_grasp_best2": {
+      "action": {
+        "mean": [
+          79.11602783203125,
+          94.15535736083984,
+          -80.4166259765625,
+          -93.67428588867188,
+          -3.3462841510772705,
+          17.82291603088379,
+          6.387060165405273,
+          78.9906997680664,
+          -82.27678680419922,
+          88.60293579101562,
+          82.36312103271484,
+          -95.4980239868164,
+          3.2222084999084473,
+          15.378264427185059,
+          1.2210661172866821,
+          80.26712799072266
+        ],
+        "std": [
+          21.024213790893555,
+          5.583713531494141,
+          10.143651008605957,
+          2.8190293312072754,
+          6.1579060554504395,
+          12.958017349243164,
+          9.206350326538086,
+          38.70134735107422,
+          24.093891143798828,
+          10.70143985748291,
+          13.960134506225586,
+          3.1140949726104736,
+          12.975412368774414,
+          19.39619255065918,
+          11.416078567504883,
+          38.01255416870117
+        ],
+        "max": [
+          112.63500213623047,
+          102.0,
+          -43.02000045776367,
+          -68.55000305175781,
+          20.434999465942383,
+          61.125,
+          46.845001220703125,
+          100.0,
+          -18.19499969482422,
+          102.0,
+          110.38999938964844,
+          -72.7699966430664,
+          71.93499755859375,
+          67.31999969482422,
+          42.23500061035156,
+          100.0
+        ],
+        "min": [
+          31.375,
+          66.01000213623047,
+          -99.53500366210938,
+          -95.75499725341797,
+          -38.540000915527344,
+          -22.940000534057617,
+          -37.529998779296875,
+          0.0,
+          -115.30999755859375,
+          14.984999656677246,
+          18.020000457763672,
+          -97.47000122070312,
+          -22.719999313354492,
+          -23.639999389648438,
+          -41.84000015258789,
+          0.0
+        ],
+        "q01": [
+          34.29199905395508,
+          78.30999755859375,
+          -93.43000030517578,
+          -95.44999694824219,
+          -20.040000915527344,
+          -3.5509999990463235,
+          -15.819999694824219,
+          0.0,
+          -109.94999694824219,
+          46.845001220703125,
+          35.15999984741211,
+          -97.29000091552734,
+          -15.1479998588562,
+          -12.739999771118164,
+          -30.791500282287597,
+          0.0
+        ],
+        "q99": [
+          105.26450119018561,
+          102.0,
+          -49.8484996795654,
+          -80.68149948120117,
+          8.350000381469727,
+          47.290000915527344,
+          32.16999816894531,
+          100.0,
+          -30.818500328063934,
+          102.0,
+          100.80999755859375,
+          -81.88200302124021,
+          52.8650016784668,
+          55.61800079345706,
+          30.233500289916996,
+          100.0
+        ],
+        "mask": [
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true,
+          true
+        ]
+      },
+      "proprio": {
+        "mean": [
+          79.11497497558594,
+          94.18975830078125,
+          -80.38821411132812,
+          -93.6719970703125,
+          -3.3557398319244385,
+          17.77706527709961,
+          6.375845909118652,
+          78.99101257324219,
+          -82.22480773925781,
+          88.610107421875,
+          82.32734680175781,
+          -95.49593353271484,
+          3.2511658668518066,
+          15.379179000854492,
+          1.202197551727295,
+          80.26780700683594
+        ],
+        "std": [
+          20.358299255371094,
+          5.266684055328369,
+          9.759184837341309,
+          2.6425156593322754,
+          5.872856140136719,
+          12.372116088867188,
+          8.679393768310547,
+          38.70150375366211,
+          23.474021911621094,
+          10.264312744140625,
+          13.466100692749023,
+          2.960315227508545,
+          12.600417137145996,
+          18.766708374023438,
+          11.020885467529297,
+          38.01284408569336
+        ],
+        "max": [
+          108.94000244140625,
+          102.0199966430664,
+          -45.0099983215332,
+          -72.97000122070312,
+          14.890000343322754,
+          55.84000015258789,
+          46.47999954223633,
+          100.0,
+          -21.459999084472656,
+          102.01000213623047,
+          104.38999938964844,
+          -73.69999694824219,
+          62.630001068115234,
+          64.4000015258789,
+          40.790000915527344,
+          100.0
+        ],
+        "min": [
+          32.54999923706055,
+          73.81999969482422,
+          -98.31999969482422,
+          -95.58000183105469,
+          -36.43000030517578,
+          -16.389999389648438,
+          -26.139999389648438,
+          0.0,
+          -111.0999984741211,
+          28.559999465942383,
+          19.8700008392334,
+          -97.37000274658203,
+          -16.200000762939453,
+          -18.959999084472656,
+          -39.83000183105469,
+          0.0
+        ],
+        "q01": [
+          36.040000915527344,
+          79.12000274658203,
+          -92.80000305175781,
+          -95.37999725341797,
+          -19.156999778747558,
+          -0.11400000452995164,
+          -12.917000007629394,
+          0.0,
+          -109.78100128173828,
+          49.36499938964844,
+          36.233000946044925,
+          -97.2300033569336,
+          -14.706999969482421,
+          -12.258000183105468,
+          -30.653999710083006,
+          0.0
+        ],
+        "q99": [
+          102.68300018310553,
+          102.01000213623047,
+          -51.226001358032214,
+          -81.30999755859375,
+          8.239999771118164,
+          45.500998687744165,
+          31.277000617980963,
+          100.0,
+          -31.829999351501392,
+          101.95999908447266,
+          100.08000183105469,
+          -82.35599975585936,
+          51.74799919128421,
+          53.9569995880127,
+          29.15399971008302,
+          100.0
+        ]
+      },
+      "num_transitions": 16131,
+      "num_trajectories": 49
+    }
+  },
+  "n_action_bins": 256,
+  "vision_backbone_id": "dinosiglip-vit-so-224px",
+  "llm_backbone_id": "llama2-7b-pure",
+  "arch_specifier": "no-align+fused-gelu-mlp",
+  "output_projector_states": false,
+  "use_fused_vision_backbone": true,
+  "timm_model_ids": [
+    "vit_large_patch14_reg4_dinov2.lvd142m",
+    "vit_so400m_patch14_siglip_224"
+  ],
+  "timm_override_act_layers": [
+    null,
+    null
+  ],
+  "image_sizes": [
+    224,
+    224
+  ],
+  "image_resize_strategy": "resize-naive",
+  "hf_llm_id": "meta-llama/Llama-2-7b-hf",
+  "llm_max_length": 2048,
+  "pad_token_id": 32000,
+  "pad_to_multiple_of": 64,
+  "text_config": {
+    "vocab_size": 32064,
+    "max_position_embeddings": 2048,
+    "hidden_size": 4096,
+    "intermediate_size": 11008,
+    "num_hidden_layers": 32,
+    "num_attention_heads": 32,
+    "num_key_value_heads": 32,
+    "hidden_act": "silu",
+    "initializer_range": 0.02,
+    "rms_norm_eps": 1e-06,
+    "pretraining_tp": 1,
+    "use_cache": true,
+    "rope_theta": 10000.0,
+    "rope_scaling": null,
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "return_dict": true,
+    "output_hidden_states": false,
+    "output_attentions": false,
+    "torchscript": false,
+    "torch_dtype": "bfloat16",
+    "use_bfloat16": false,
+    "tf_legacy_loss": false,
+    "pruned_heads": {},
+    "tie_word_embeddings": false,
+    "chunk_size_feed_forward": 0,
+    "is_encoder_decoder": false,
+    "is_decoder": false,
+    "cross_attention_hidden_size": null,
+    "add_cross_attention": false,
+    "tie_encoder_decoder": false,
+    "max_length": 20,
+    "min_length": 0,
+    "do_sample": false,
+    "early_stopping": false,
+    "num_beams": 1,
+    "num_beam_groups": 1,
+    "diversity_penalty": 0.0,
+    "temperature": 1.0,
+    "top_k": 50,
+    "top_p": 1.0,
+    "typical_p": 1.0,
+    "repetition_penalty": 1.0,
+    "length_penalty": 1.0,
+    "no_repeat_ngram_size": 0,
+    "encoder_no_repeat_ngram_size": 0,
+    "bad_words_ids": null,
+    "num_return_sequences": 1,
+    "output_scores": false,
+    "return_dict_in_generate": false,
+    "forced_bos_token_id": null,
+    "forced_eos_token_id": null,
+    "remove_invalid_values": false,
+    "exponential_decay_length_penalty": null,
+    "suppress_tokens": null,
+    "begin_suppress_tokens": null,
+    "architectures": null,
+    "finetuning_task": null,
+    "id2label": {
+      "0": "LABEL_0",
+      "1": "LABEL_1"
+    },
+    "label2id": {
+      "LABEL_0": 0,
+      "LABEL_1": 1
+    },
+    "tokenizer_class": null,
+    "prefix": null,
+    "bos_token_id": 1,
+    "pad_token_id": 32000,
+    "eos_token_id": 2,
+    "sep_token_id": null,
+    "decoder_start_token_id": null,
+    "task_specific_params": null,
+    "problem_type": null,
+    "_name_or_path": "",
+    "model_type": "llama"
+  },
+  "return_dict": true,
+  "output_hidden_states": false,
+  "output_attentions": false,
+  "torchscript": false,
+  "torch_dtype": "bfloat16",
+  "use_bfloat16": false,
+  "tf_legacy_loss": false,
+  "pruned_heads": {},
+  "tie_word_embeddings": true,
+  "chunk_size_feed_forward": 0,
+  "is_encoder_decoder": false,
+  "is_decoder": false,
+  "cross_attention_hidden_size": null,
+  "add_cross_attention": false,
+  "tie_encoder_decoder": false,
+  "max_length": 20,
+  "min_length": 0,
+  "do_sample": false,
+  "early_stopping": false,
+  "num_beams": 1,
+  "num_beam_groups": 1,
+  "diversity_penalty": 0.0,
+  "temperature": 1.0,
+  "top_k": 50,
+  "top_p": 1.0,
+  "typical_p": 1.0,
+  "repetition_penalty": 1.0,
+  "length_penalty": 1.0,
+  "no_repeat_ngram_size": 0,
+  "encoder_no_repeat_ngram_size": 0,
+  "bad_words_ids": null,
+  "num_return_sequences": 1,
+  "output_scores": false,
+  "return_dict_in_generate": false,
+  "forced_bos_token_id": null,
+  "forced_eos_token_id": null,
+  "remove_invalid_values": false,
+  "exponential_decay_length_penalty": null,
+  "suppress_tokens": null,
+  "begin_suppress_tokens": null,
+  "architectures": [
+    "OpenVLAForActionPrediction"
+  ],
+  "finetuning_task": null,
+  "id2label": {
+    "0": "LABEL_0",
+    "1": "LABEL_1"
+  },
+  "label2id": {
+    "LABEL_0": 0,
+    "LABEL_1": 1
+  },
+  "tokenizer_class": null,
+  "prefix": null,
+  "bos_token_id": null,
+  "eos_token_id": null,
+  "sep_token_id": null,
+  "decoder_start_token_id": null,
+  "task_specific_params": null,
+  "problem_type": null,
+  "_name_or_path": "/home/guangyu/.cache/huggingface/hub/models--openvla--openvla-7b/snapshots/31f090d05236101ebfc381b61c674dd4746d4ce0",
+  "transformers_version": "4.40.1",
+  "auto_map": {
+    "AutoConfig": "configuration_prismatic.OpenVLAConfig",
+    "AutoModelForVision2Seq": "modeling_prismatic.OpenVLAForActionPrediction"
+  },
+  "model_type": "openvla"
+}

dataset_statistics.json ADDED Viewed

	@@ -0,0 +1,244 @@

+{
+  "gen72_grasp_best2": {
+    "action": {
+      "mean": [
+        79.11602783203125,
+        94.15535736083984,
+        -80.4166259765625,
+        -93.67428588867188,
+        -3.3462841510772705,
+        17.82291603088379,
+        6.387060165405273,
+        78.9906997680664,
+        -82.27678680419922,
+        88.60293579101562,
+        82.36312103271484,
+        -95.4980239868164,
+        3.2222084999084473,
+        15.378264427185059,
+        1.2210661172866821,
+        80.26712799072266
+      ],
+      "std": [
+        21.024213790893555,
+        5.583713531494141,
+        10.143651008605957,
+        2.8190293312072754,
+        6.1579060554504395,
+        12.958017349243164,
+        9.206350326538086,
+        38.70134735107422,
+        24.093891143798828,
+        10.70143985748291,
+        13.960134506225586,
+        3.1140949726104736,
+        12.975412368774414,
+        19.39619255065918,
+        11.416078567504883,
+        38.01255416870117
+      ],
+      "max": [
+        112.63500213623047,
+        102.0,
+        -43.02000045776367,
+        -68.55000305175781,
+        20.434999465942383,
+        61.125,
+        46.845001220703125,
+        100.0,
+        -18.19499969482422,
+        102.0,
+        110.38999938964844,
+        -72.7699966430664,
+        71.93499755859375,
+        67.31999969482422,
+        42.23500061035156,
+        100.0
+      ],
+      "min": [
+        31.375,
+        66.01000213623047,
+        -99.53500366210938,
+        -95.75499725341797,
+        -38.540000915527344,
+        -22.940000534057617,
+        -37.529998779296875,
+        0.0,
+        -115.30999755859375,
+        14.984999656677246,
+        18.020000457763672,
+        -97.47000122070312,
+        -22.719999313354492,
+        -23.639999389648438,
+        -41.84000015258789,
+        0.0
+      ],
+      "q01": [
+        34.29199905395508,
+        78.30999755859375,
+        -93.43000030517578,
+        -95.44999694824219,
+        -20.040000915527344,
+        -3.5509999990463235,
+        -15.819999694824219,
+        0.0,
+        -109.94999694824219,
+        46.845001220703125,
+        35.15999984741211,
+        -97.29000091552734,
+        -15.1479998588562,
+        -12.739999771118164,
+        -30.791500282287597,
+        0.0
+      ],
+      "q99": [
+        105.26450119018561,
+        102.0,
+        -49.8484996795654,
+        -80.68149948120117,
+        8.350000381469727,
+        47.290000915527344,
+        32.16999816894531,
+        100.0,
+        -30.818500328063934,
+        102.0,
+        100.80999755859375,
+        -81.88200302124021,
+        52.8650016784668,
+        55.61800079345706,
+        30.233500289916996,
+        100.0
+      ],
+      "mask": [
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true,
+        true
+      ]
+    },
+    "proprio": {
+      "mean": [
+        79.11497497558594,
+        94.18975830078125,
+        -80.38821411132812,
+        -93.6719970703125,
+        -3.3557398319244385,
+        17.77706527709961,
+        6.375845909118652,
+        78.99101257324219,
+        -82.22480773925781,
+        88.610107421875,
+        82.32734680175781,
+        -95.49593353271484,
+        3.2511658668518066,
+        15.379179000854492,
+        1.202197551727295,
+        80.26780700683594
+      ],
+      "std": [
+        20.358299255371094,
+        5.266684055328369,
+        9.759184837341309,
+        2.6425156593322754,
+        5.872856140136719,
+        12.372116088867188,
+        8.679393768310547,
+        38.70150375366211,
+        23.474021911621094,
+        10.264312744140625,
+        13.466100692749023,
+        2.960315227508545,
+        12.600417137145996,
+        18.766708374023438,
+        11.020885467529297,
+        38.01284408569336
+      ],
+      "max": [
+        108.94000244140625,
+        102.0199966430664,
+        -45.0099983215332,
+        -72.97000122070312,
+        14.890000343322754,
+        55.84000015258789,
+        46.47999954223633,
+        100.0,
+        -21.459999084472656,
+        102.01000213623047,
+        104.38999938964844,
+        -73.69999694824219,
+        62.630001068115234,
+        64.4000015258789,
+        40.790000915527344,
+        100.0
+      ],
+      "min": [
+        32.54999923706055,
+        73.81999969482422,
+        -98.31999969482422,
+        -95.58000183105469,
+        -36.43000030517578,
+        -16.389999389648438,
+        -26.139999389648438,
+        0.0,
+        -111.0999984741211,
+        28.559999465942383,
+        19.8700008392334,
+        -97.37000274658203,
+        -16.200000762939453,
+        -18.959999084472656,
+        -39.83000183105469,
+        0.0
+      ],
+      "q01": [
+        36.040000915527344,
+        79.12000274658203,
+        -92.80000305175781,
+        -95.37999725341797,
+        -19.156999778747558,
+        -0.11400000452995164,
+        -12.917000007629394,
+        0.0,
+        -109.78100128173828,
+        49.36499938964844,
+        36.233000946044925,
+        -97.2300033569336,
+        -14.706999969482421,
+        -12.258000183105468,
+        -30.653999710083006,
+        0.0
+      ],
+      "q99": [
+        102.68300018310553,
+        102.01000213623047,
+        -51.226001358032214,
+        -81.30999755859375,
+        8.239999771118164,
+        45.500998687744165,
+        31.277000617980963,
+        100.0,
+        -31.829999351501392,
+        101.95999908447266,
+        100.08000183105469,
+        -82.35599975585936,
+        51.74799919128421,
+        53.9569995880127,
+        29.15399971008302,
+        100.0
+      ]
+    },
+    "num_transitions": 16131,
+    "num_trajectories": 49
+  }
+}

lora_adapter/README.md ADDED Viewed

	@@ -0,0 +1,202 @@

+---
+base_model: /home/guangyu/.cache/huggingface/hub/models--openvla--openvla-7b/snapshots/31f090d05236101ebfc381b61c674dd4746d4ce0
+library_name: peft
+---
+# Model Card for Model ID
+<!-- Provide a quick summary of what the model is/does. -->
+## Model Details
+### Model Description
+<!-- Provide a longer summary of what this model is. -->
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+### Model Sources [optional]
+<!-- Provide the basic links for the model. -->
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+## Uses
+<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
+### Direct Use
+<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
+[More Information Needed]
+### Downstream Use [optional]
+<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
+[More Information Needed]
+### Out-of-Scope Use
+<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
+[More Information Needed]
+## Bias, Risks, and Limitations
+<!-- This section is meant to convey both technical and sociotechnical limitations. -->
+[More Information Needed]
+### Recommendations
+<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+## How to Get Started with the Model
+Use the code below to get started with the model.
+[More Information Needed]
+## Training Details
+### Training Data
+<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
+[More Information Needed]
+### Training Procedure
+<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
+#### Preprocessing [optional]
+[More Information Needed]
+#### Training Hyperparameters
+- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
+#### Speeds, Sizes, Times [optional]
+<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
+[More Information Needed]
+## Evaluation
+<!-- This section describes the evaluation protocols and provides the results. -->
+### Testing Data, Factors & Metrics
+#### Testing Data
+<!-- This should link to a Dataset Card if possible. -->
+[More Information Needed]
+#### Factors
+<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
+[More Information Needed]
+#### Metrics
+<!-- These are the evaluation metrics being used, ideally with a description of why. -->
+[More Information Needed]
+### Results
+[More Information Needed]
+#### Summary
+## Model Examination [optional]
+<!-- Relevant interpretability work for the model goes here -->
+[More Information Needed]
+## Environmental Impact
+<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+## Technical Specifications [optional]
+### Model Architecture and Objective
+[More Information Needed]
+### Compute Infrastructure
+[More Information Needed]
+#### Hardware
+[More Information Needed]
+#### Software
+[More Information Needed]
+## Citation [optional]
+<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
+**BibTeX:**
+[More Information Needed]
+**APA:**
+[More Information Needed]
+## Glossary [optional]
+<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
+[More Information Needed]
+## More Information [optional]
+[More Information Needed]
+## Model Card Authors [optional]
+[More Information Needed]
+## Model Card Contact
+[More Information Needed]
+### Framework versions
+- PEFT 0.11.1

lora_adapter/adapter_config.json ADDED Viewed

	@@ -0,0 +1,45 @@

+{
+  "alpha_pattern": {},
+  "auto_mapping": {
+    "base_model_class": "OpenVLAForActionPrediction",
+    "parent_library": "transformers_modules.31f090d05236101ebfc381b61c674dd4746d4ce0.modeling_prismatic"
+  },
+  "base_model_name_or_path": "/home/guangyu/.cache/huggingface/hub/models--openvla--openvla-7b/snapshots/31f090d05236101ebfc381b61c674dd4746d4ce0",
+  "bias": "none",
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": "gaussian",
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 16,
+  "lora_dropout": 0.0,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "r": 32,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": [
+    "fc1",
+    "qkv",
+    "k_proj",
+    "q_proj",
+    "kv",
+    "o_proj",
+    "v_proj",
+    "q",
+    "proj",
+    "fc2",
+    "down_proj",
+    "up_proj",
+    "lm_head",
+    "gate_proj",
+    "fc3"
+  ],
+  "task_type": null,
+  "use_dora": false,
+  "use_rslora": false
+}

lora_adapter/adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2e422139a3ade0bf21f18b3c5b69ccef1cdf28efa10fe4bd17e8b1e5692b5349
+size 484467800

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,114 @@

+{
+  "auto_map": {
+    "AutoImageProcessor": "processing_prismatic.PrismaticImageProcessor",
+    "AutoProcessor": "processing_prismatic.PrismaticProcessor"
+  },
+  "image_processor_type": "PrismaticImageProcessor",
+  "image_resize_strategy": "resize-naive",
+  "input_sizes": [
+    [
+      3,
+      224,
+      224
+    ],
+    [
+      3,
+      224,
+      224
+    ]
+  ],
+  "interpolations": [
+    "bicubic",
+    "bicubic"
+  ],
+  "means": [
+    [
+      0.485,
+      0.456,
+      0.406
+    ],
+    [
+      0.5,
+      0.5,
+      0.5
+    ]
+  ],
+  "processor_class": "PrismaticProcessor",
+  "stds": [
+    [
+      0.229,
+      0.224,
+      0.225
+    ],
+    [
+      0.5,
+      0.5,
+      0.5
+    ]
+  ],
+  "tvf_crop_params": [
+    {
+      "output_size": [
+        224,
+        224
+      ]
+    },
+    {
+      "output_size": [
+        224,
+        224
+      ]
+    }
+  ],
+  "tvf_do_letterbox": false,
+  "tvf_letterbox_fill": null,
+  "tvf_normalize_params": [
+    {
+      "inplace": false,
+      "mean": [
+        0.484375,
+        0.455078125,
+        0.40625
+      ],
+      "std": [
+        0.228515625,
+        0.2236328125,
+        0.224609375
+      ]
+    },
+    {
+      "inplace": false,
+      "mean": [
+        0.5,
+        0.5,
+        0.5
+      ],
+      "std": [
+        0.5,
+        0.5,
+        0.5
+      ]
+    }
+  ],
+  "tvf_resize_params": [
+    {
+      "antialias": true,
+      "interpolation": 3,
+      "max_size": null,
+      "size": [
+        224,
+        224
+      ]
+    },
+    {
+      "antialias": true,
+      "interpolation": 3,
+      "max_size": null,
+      "size": [
+        224,
+        224
+      ]
+    }
+  ],
+  "use_fused_vision_backbone": true
+}

processing_prismatic.py ADDED Viewed

	@@ -0,0 +1,257 @@

+"""
+processing_prismatic.py
+HuggingFace-style preprocessor definitions for Prismatic VLMs, inheriting from `ProcessorMixin`. Default configuration
+specifies `siglip-224px+7b`.
+"""
+from typing import Any, ClassVar, List, Optional, Tuple, Union
+import timm.data
+import torch
+import torchvision.transforms.functional as TVF
+from PIL import Image
+from torchvision.transforms import CenterCrop, Compose, Normalize, Resize, ToTensor
+from transformers import PreTrainedTokenizerBase
+from transformers.image_processing_utils import BatchFeature, ImageProcessingMixin
+from transformers.processing_utils import ProcessorMixin
+from transformers.tokenization_utils import PaddingStrategy, PreTokenizedInput, TextInput, TruncationStrategy
+from transformers.utils import TensorType
+# === Image Processing ===
+def letterbox_pad_transform(image: Image.Image, padding_fill_value: Tuple[int, int, int]) -> Image.Image:
+    """Given a PIL.Image, pad to square by adding a symmetric border around the height/width."""
+    (w, h), max_wh = image.size, max(image.size)
+    horizontal_pad, vertical_pad = int((max_wh - w) / 2), int((max_wh - h) / 2)
+    padding = (horizontal_pad, vertical_pad, horizontal_pad, vertical_pad)
+    return TVF.pad(image, padding, fill=padding_fill_value, padding_mode="constant")
+class PrismaticImageProcessor(ImageProcessingMixin):
+    model_input_names: ClassVar[List[str]] = ["pixel_values"]
+    def __init__(
+        self,
+        use_fused_vision_backbone: bool = False,
+        image_resize_strategy: str = "letterbox",
+        input_sizes: Optional[List[Tuple[int, int, int]]] = None,
+        interpolations: Optional[List[str]] = None,
+        means: Optional[List[Tuple[float, float, float]]] = None,
+        stds: Optional[List[Tuple[float, float, float]]] = None,
+        **kwargs: str,
+    ) -> None:
+        """
+        Initialize a PrismaticImageProcessor as a wrapper around a torchvision transform; this transform will be
+        created by TIMM, and edited to follow our custom `image_resize_strategy` logic.
+        @param use_fused_vision_backbone: Boolean indicating single or fused (dual) vision backbone
+        @param image_resize_strategy: Prismatic image resize strategy in < resize-naive | resize-crop | letterbox >
+        @param input_size: [TIMM :: `data_cfg`] Input image size as tuple (channels, width, height)
+        @param interpolation: [TIMM :: `data_cfg`] Interpolation as string (default: "bicubic")
+        @param mean: [TIMM :: `data_cfg`] Normalization mean as float tuple (or two-tuple if `fused_backbone`)
+        @param std: [TIMM :: `data_cfg`] Normalization std as float tuple (or two-tuple if `fused_backbone`)
+        """
+        self.use_fused_vision_backbone = use_fused_vision_backbone
+        self.image_resize_strategy = image_resize_strategy
+        # Handle `None` default values
+        input_sizes = [(3, 224, 224)] if input_sizes is None else input_sizes
+        means = [(0.5, 0.5, 0.5)] if means is None else means
+        stds = [(0.5, 0.5, 0.5)] if stds is None else stds
+        # TIMM `data_cfg` Parameters
+        self.input_sizes, self.interpolations, self.means, self.stds = input_sizes, interpolations, means, stds
+        # Grab torchvision transforms via TIMM =>> need to parse for specific "functional" transform values!
+        self.tvf_resize_params, self.tvf_crop_params, self.tvf_normalize_params = [], [], []
+        self.tvf_do_letterbox, self.tvf_letterbox_fill = False, None
+        for idx in range(len(input_sizes)):
+            transform = timm.data.create_transform(
+                input_size=self.input_sizes[idx],
+                interpolation=self.interpolations[idx],
+                mean=self.means[idx],
+                std=self.stds[idx],
+                crop_pct=1.0,  # Set to 1.0 to ignore cropping (initial Resize sets `input_size`)
+                crop_mode="center",  # Default crop mode -- no-op when `crop_pct == 1.0`
+                is_training=False,  # No image augmentations when loading the transform!
+            )
+            # [Validation] Ensure appropriate transform structure, expected sizes
+            if not (
+                isinstance(transform, Compose)
+                and (len(transform.transforms) == 4)
+                and isinstance(transform.transforms[0], Resize)
+                and isinstance(transform.transforms[1], CenterCrop)
+                and isinstance(transform.transforms[2], ToTensor)
+                and isinstance(transform.transforms[3], Normalize)
+                and (transform.transforms[0].size == self.input_sizes[idx][-1])
+                and (transform.transforms[1].size == self.input_sizes[idx][-2:])
+            ):
+                raise ValueError(f"Unexpected TIMM image transformation structure/sizes: `{transform}`")
+            # HF Image Processors *must* be JSON-serializable; as such, cannot have torchvision. as an attribute.
+            #   => Instead, we're going to parse the transform and call "torchvision.transforms.functional" (`tvf`)
+            resize_t, crop_t, norm_t = transform.transforms[0], transform.transforms[1], transform.transforms[3]
+            self.tvf_resize_params.append(
+                {
+                    "size": resize_t.size,
+                    "interpolation": TVF.pil_modes_mapping[resize_t.interpolation],
+                    "max_size": None,
+                    "antialias": True,
+                }
+            )
+            self.tvf_crop_params.append({"output_size": crop_t.size})
+            self.tvf_normalize_params.append(
+                {
+                    "mean": norm_t.mean.float().numpy().tolist(),
+                    "std": norm_t.std.float().numpy().tolist(),
+                    "inplace": False,
+                }
+            )
+            self.tvf_do_letterbox, self.tvf_letterbox_fill = False, None
+            # Handle Prismatic `image_resize_strategy`
+            if self.image_resize_strategy == "resize-naive":
+                self.tvf_resize_params[idx]["size"] = (resize_t.size, resize_t.size)
+            elif self.image_resize_strategy == "letterbox":
+                self.tvf_do_letterbox, self.tvf_letterbox_fill = True, tuple([int(x * 255) for x in self.means[idx]])
+            elif self.image_resize_strategy == "resize-crop":
+                pass
+            else:
+                raise ValueError(f"Image resize strategy `{self.image_resize_strategy}` is not supported!")
+        # Dispatch **kwargs to super()
+        super().__init__(**kwargs)
+    def apply_transform(self, img: Image.Image) -> torch.Tensor:
+        """Apply `functional` variant of TIMM's Transform = Compose([Resize -> CenterCrop -> ToTensor -> Normalize])"""
+        if self.tvf_do_letterbox:
+            img = letterbox_pad_transform(img, self.tvf_letterbox_fill)
+        # [Contract] Fused Backbones expect "channel-stacked" inputs; we'll unpack on the model side!
+        imgs_t = []
+        for idx in range(len(self.input_sizes)):
+            img_idx = TVF.resize(img, **self.tvf_resize_params[idx])
+            img_idx = TVF.center_crop(img_idx, **self.tvf_crop_params[idx])
+            img_idx_t = TVF.to_tensor(img_idx)
+            img_idx_t = TVF.normalize(img_idx_t, **self.tvf_normalize_params[idx])
+            imgs_t.append(img_idx_t)
+        # [Contract] `imgs_t` is a list of Tensors of shape [3, input_size, input_size]; stack along dim = 0
+        img_t = torch.vstack(imgs_t)
+        return img_t
+    def preprocess(
+        self,
+        images: Union[Image.Image, List[Image.Image]],
+        return_tensors: Optional[Union[str, TensorType]] = None,
+        **_: str,
+    ) -> BatchFeature:
+        """
+        Preprocess an image (or batch of images); note that unlike the `transformers :: BaseImageProcessor` we
+        explicitly only handle PIL.Image.Image instances for simplicity.
+        @param images: A (batch of) PIL.Image.Image instance(s) to preprocess.
+        @param return_tensors: BatchFeature default Tensor format (e.g., "pt" for torch); if None, returns np.ndarray
+        @return: Instance of `transformers :: BatchFeature` with a single key "pixel_values"
+        """
+        if not isinstance(images, list):
+            images = [images]
+        # Apply `self.img_transform` to each image (will return list of torch.Tensors); stack into "batched" Tensor
+        pixel_values = torch.stack([self.apply_transform(img.convert("RGB")) for img in images])
+        # Return BatchFeature =>> note that for compatibility, constructor expects Dict[str, np.ndarray], so we convert
+        return BatchFeature(data={"pixel_values": pixel_values.float().numpy()}, tensor_type=return_tensors)
+    def __call__(self, images: Union[Image.Image, List[Image.Image]], **kwargs) -> BatchFeature:
+        return self.preprocess(images, **kwargs)
+# === PrismaticProcessor =>> Wraps both ImageProcessor and Tokenizer ===
+#   =>> https://github.com/huggingface/transformers/blob/main/src/transformers/models/llava/processing_llava.py
+class PrismaticProcessor(ProcessorMixin):
+    attributes: ClassVar[List[str]] = ["image_processor", "tokenizer"]
+    image_processor_class: str = "AutoImageProcessor"
+    tokenizer_class: str = "AutoTokenizer"
+    def __init__(
+        self,
+        image_processor: Optional[ImageProcessingMixin] = None,
+        tokenizer: Optional[PreTrainedTokenizerBase] = None,
+    ) -> None:
+        super().__init__(image_processor, tokenizer)
+    def __call__(
+        self,
+        text: Union[TextInput, PreTokenizedInput, List[TextInput], List[PreTokenizedInput]],
+        images: Union[Image.Image, List[Image.Image]],
+        padding: Union[bool, str, PaddingStrategy] = False,
+        truncation: Optional[Union[bool, str, TruncationStrategy]] = None,
+        max_length: Optional[int] = None,
+        return_tensors: Optional[Union[str, TensorType]] = TensorType.PYTORCH,
+    ) -> BatchFeature:
+        """
+        Preprocess a given (batch) of text/images for a Prismatic VLM; forwards text to the underlying LLM's tokenizer,
+        forwards images to PrismaticImageProcessor.
+        @param text: The (batch) of text to encode; must be a string or list of strings.
+        @param images: A (batch of) PIL.Image.Image instance(s) to preprocess.
+        @param padding: Sequence padding strategy (if multiple specified) in < True = "longest" | "max_length" | False >
+        @param truncation: Truncation strategy for the output sequences; requires `max_length` to be specified
+        @param max_length: Maximum length (in tokens) to truncate
+        @param return_tensors: Type of return tensors (usually "pt" or TensorType.PYTORCH)
+        @return: BatchFeature with keys for `input_ids`, `attention_mask` and `pixel_values`.
+        """
+        pixel_values = self.image_processor(images, return_tensors=return_tensors)["pixel_values"]
+        text_inputs = self.tokenizer(
+            text, return_tensors=return_tensors, padding=padding, truncation=truncation, max_length=max_length
+        )
+        # [Validate] Need same number of images and text inputs!
+        if pixel_values.shape[0] != text_inputs.input_ids.shape[0]:
+            raise ValueError("Batch is malformed; expected same number of images and text inputs!")
+        return BatchFeature(data={**text_inputs, "pixel_values": pixel_values})
+    # === Tokenizer Dispatch Utilities =>> check `PreTrainedTokenizerBase` for documentation ===
+    def batch_decode(
+        self,
+        sequences: Union[List[int], List[List[int]], torch.Tensor, Any],  # `Any` = np.ndarray | tf.Tensor
+        skip_special_tokens: bool = False,
+        clean_up_tokenization_spaces: Optional[bool] = None,
+        **kwargs: str,
+    ) -> List[str]:
+        return self.tokenizer.batch_decode(
+            sequences=sequences,
+            skip_special_tokens=skip_special_tokens,
+            clean_up_tokenization_spaces=clean_up_tokenization_spaces,
+            **kwargs,
+        )
+    def decode(
+        self,
+        token_ids: Union[int, List[int], torch.Tensor, Any],  # `Any` = np.ndarray | tf.Tensor
+        skip_special_tokens: bool = False,
+        clean_up_tokenization_spaces: Optional[bool] = None,
+        **kwargs: str,
+    ) -> str:
+        return self.tokenizer.decode(
+            token_ids=token_ids,
+            skip_special_tokens=skip_special_tokens,
+            clean_up_tokenization_spaces=clean_up_tokenization_spaces,
+            **kwargs,
+        )
+    @property
+    def model_input_names(self) -> List[str]:
+        tokenizer_input_names = self.tokenizer.model_input_names
+        image_processor_input_names = self.image_processor.model_input_names
+        return list(dict.fromkeys(tokenizer_input_names + image_processor_input_names))

processor_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "auto_map": {
+    "AutoProcessor": "processing_prismatic.PrismaticProcessor"
+  },
+  "processor_class": "PrismaticProcessor"
+}

proprio_projector--22000_checkpoint.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8f2e1dba8b2aaf8033ac044a441f26b36bcd241d613590016714bfba916d88bd
+size 67406320

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<PAD>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
+size 499723

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,53 @@

+{
+  "add_bos_token": true,
+  "add_eos_token": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "32000": {
+      "content": "<PAD>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "auto_map": {
+    "AutoProcessor": "processing_prismatic.PrismaticProcessor"
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "legacy": false,
+  "model_max_length": 2048,
+  "pad_token": "<PAD>",
+  "padding_side": "right",
+  "processor_class": "PrismaticProcessor",
+  "sp_model_kwargs": {},
+  "tokenizer_class": "LlamaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

vision_backbone--22000_checkpoint.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2e38522c6094d87902453341de9e0a227fc9cc24b5c1f1f0d6b254552f8dce20
+size 3344957817