ModaVerse-7b-v0

Browse files

Files changed (7) hide show

added_tokens.json +8 -0
config.json +29 -0
config.py +83 -0
pytorch_model.pt +3 -0
special_tokens_map.json +24 -0
tokenizer.model +3 -0
tokenizer_config.json +90 -0

added_tokens.json ADDED Viewed

	@@ -0,0 +1,8 @@

+{
+  "</Media>": 32001,
+  "<Media>": 32000,
+  "[AUDIO]": 32004,
+  "[IMAGE]": 32003,
+  "[TEXT]": 32002,
+  "[VIDEO]": 32005
+}

config.json ADDED Viewed

	@@ -0,0 +1,29 @@

+{
+  "_name_or_path": ".checkpoints/7b_v0",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "hidden_act": "silu",
+  "hidden_size": 4096,
+  "initializer_range": 0.02,
+  "intermediate_size": 11008,
+  "max_position_embeddings": 2048,
+  "model_type": "llama",
+  "num_attention_heads": 32,
+  "num_hidden_layers": 32,
+  "num_key_value_heads": 32,
+  "pad_token_id": 0,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "torch_dtype": "float16",
+  "transformers_version": "4.39.2",
+  "use_cache": true,
+  "vocab_size": 32006
+}

config.py ADDED Viewed

	@@ -0,0 +1,83 @@

+from typing import List
+import torch
+from peft import LoraConfig, TaskType
+from transformers import StoppingCriteria, StoppingCriteriaList
+class StoppingCriteriaSub(StoppingCriteria):
+    def __init__(self, stops: List = None, encounters: int = 1):
+        super().__init__()
+        self.stops = stops
+        self.ENCOUNTERS = encounters
+    def __call__(self, input_ids: torch.LongTensor, scores: torch.FloatTensor):
+        stop_count = 0
+        for stop in self.stops:
+            _stop = torch.tensor(stop).to(input_ids[0].device)
+            indices = torch.where(_stop[0] == input_ids)
+            for i in indices:
+                if len(i) > 0:
+                    if torch.all(input_ids[0][i:i + len(_stop)] == _stop):
+                        stop_count += 1
+        if stop_count >= self.ENCOUNTERS:
+            return True
+        return False
+prompt_configs = dict(path='assets/prompts/prompt_template.txt',
+                      media_placeholder='{media}',
+                      instruction_placeholder='{instruction}')
+model_configs = dict(
+    name='ModaVerse-7b',
+    imagebind=dict(hidden_size=1024),
+    foundation_llm=dict(type='vicuna-7b', checkpoint='.checkpoints/7b_v0'),
+    modaverse=dict(
+        max_length=512,
+        modality_begin_token='<Media>',
+        modality_end_token='</Media>',
+        modality_flags=['[TEXT]', '[IMAGE]', '[AUDIO]', '[VIDEO]'],
+        target_padding=-100,
+        top_p=0.01,
+        temperature=1,
+        max_new_tokens=246,
+        do_sample=True,
+        use_cache=True,
+        stopping_token=835,
+        stopping_criteria=StoppingCriteriaList(
+            [StoppingCriteriaSub(stops=[[835]], encounters=1)], ),
+        generator=dict(
+            image_diffuser=dict(
+                type='stable_diffusion',
+                # preload=False,
+                cfgs=dict(model='runwayml/stable-diffusion-v1-5')),
+            video_diffuser=dict(
+                type='damo_vilab',
+                # preload=False,
+                cfgs=dict(model='damo-vilab/text-to-video-ms-1.7b')),
+            audio_diffuser=dict(type='audio_ldm',
+                                cfgs=dict(model='cvssp/audioldm-l-full')),
+        ),
+    ))
+training_configs = dict(
+    lora_config=LoraConfig(
+        task_type=TaskType.CAUSAL_LM,
+        inference_mode=False,
+        r=32,
+        lora_alpha=32,
+        lora_dropout=0.1,
+        target_modules=['q_proj', 'k_proj', 'v_proj', 'o_proj']),
+    deepspeed_cfg=dict(path='configs/dscfg.json', backend='nccl'),
+    saving_root='./experiments',
+    epochs=1,
+    warmup_rate=0.1,
+    force_training_layers=['embed_tokens.weight', 'lm_head.weight'],
+    report_backend=dict(type='wandb', iterval=10),
+    print_prediction=dict(turn_on=True, interval=1000),
+    checkpointer=dict(type='iteration', interval=5000))
+dataset_configs = dict(train=dict(instruction_path='dataset/instructions.json',
+                                  media_root='dataset/'))

pytorch_model.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3ce807340b62e3622cf1e76a2a975ec5388b892be26dae52ecac1b4219179796
+size 599942525

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,24 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": "</s>",
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
+size 499723

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,90 @@

+{
+  "add_bos_token": true,
+  "add_eos_token": false,
+  "add_prefix_space": true,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "32000": {
+      "content": "<Media>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "32001": {
+      "content": "</Media>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "32002": {
+      "content": "[TEXT]",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "32003": {
+      "content": "[IMAGE]",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "32004": {
+      "content": "[AUDIO]",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "32005": {
+      "content": "[VIDEO]",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "legacy": true,
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "</s>",
+  "sp_model_kwargs": {},
+  "spaces_between_special_tokens": false,
+  "tokenizer_class": "LlamaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}