Upload Exaone4ForCausalLM

Browse files

Files changed (13) hide show

config.json +116 -0
generation_config.json +8 -0
pytorch_model-00001-of-00010.bin +3 -0
pytorch_model-00002-of-00010.bin +3 -0
pytorch_model-00003-of-00010.bin +3 -0
pytorch_model-00004-of-00010.bin +3 -0
pytorch_model-00005-of-00010.bin +3 -0
pytorch_model-00006-of-00010.bin +3 -0
pytorch_model-00007-of-00010.bin +3 -0
pytorch_model-00008-of-00010.bin +3 -0
pytorch_model-00009-of-00010.bin +3 -0
pytorch_model-00010-of-00010.bin +3 -0
pytorch_model.bin.index.json +0 -0

config.json ADDED Viewed

	@@ -0,0 +1,116 @@

+{
+  "architectures": [
+    "Exaone4ForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "bos_token_id": 1,
+  "dtype": "float16",
+  "eos_token_id": 361,
+  "head_dim": 128,
+  "hidden_act": "silu",
+  "hidden_size": 5120,
+  "initializer_range": 0.02,
+  "intermediate_size": 27392,
+  "layer_types": [
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 131072,
+  "model_type": "exaone4",
+  "num_attention_heads": 40,
+  "num_hidden_layers": 64,
+  "num_key_value_heads": 8,
+  "pad_token_id": 0,
+  "quantization_config": {
+    "_load_in_4bit": true,
+    "_load_in_8bit": false,
+    "bnb_4bit_compute_dtype": "bfloat16",
+    "bnb_4bit_quant_storage": "uint8",
+    "bnb_4bit_quant_type": "nf4",
+    "bnb_4bit_use_double_quant": false,
+    "llm_int8_enable_fp32_cpu_offload": false,
+    "llm_int8_has_fp16_weight": false,
+    "llm_int8_skip_modules": null,
+    "llm_int8_threshold": 6.0,
+    "load_in_4bit": true,
+    "load_in_8bit": false,
+    "quant_method": "bitsandbytes"
+  },
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": {
+    "factor": 16.0,
+    "high_freq_factor": 4.0,
+    "low_freq_factor": 1.0,
+    "original_max_position_embeddings": 8192,
+    "rope_type": "llama3"
+  },
+  "rope_theta": 1000000,
+  "sliding_window": 4096,
+  "sliding_window_pattern": "LLLG",
+  "tie_word_embeddings": false,
+  "transformers_version": "4.57.3",
+  "use_cache": true,
+  "vocab_size": 102400
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,8 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 1,
+  "cache_implementation": "hybrid",
+  "eos_token_id": 361,
+  "pad_token_id": 0,
+  "transformers_version": "4.57.3"
+}

pytorch_model-00001-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:60a1a6dfa8647e94b266208edab763e92d396b79164778ab249fb9a05bb01d0d
+size 1979133074

pytorch_model-00002-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:401e58245de7455954df2339c0346f4e207880029acb1ce4772f3aff6e6cb703
+size 1983518121

pytorch_model-00003-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a4ead56587630ed3c043dd9980de04dcabe1f91bf5068edf94a7e5a9a4f99269
+size 1998286683

pytorch_model-00004-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6bcd75d001a69cef7c436aa38392dcd34c0fa26a2ba0a396d76b4d1e81e19f55
+size 1995401466

pytorch_model-00005-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ad788db14104b7aa4410ef96419e0785f3995755dcedcd9de063b4d6a6b8b20d
+size 1992284962

pytorch_model-00006-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f1e1cf3b98d8f507598587bf1d07050e974bfa57fc8c6b76c2924c47270122b6
+size 1998286683

pytorch_model-00007-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6e0ab144bb15023596130d85333e06ed1b393ac824ad0a01ee6126f159063298
+size 1995401466

pytorch_model-00008-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f22d5bec1440c0b8ca6f34fc5f952880d6e6e11ed73ea495e99e2441be0d6008
+size 1992284962

pytorch_model-00009-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e89b4bb35cbfc5ac7f8ba3e61e60f063043f9e8bcf7ca0a343e113e6ea049c9a
+size 1998286683

pytorch_model-00010-of-00010.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0334cd5f0d03cd71baa98923bd5c492e71fce9f7ce66e32cf51c74f924f6db65
+size 1578019645

pytorch_model.bin.index.json ADDED Viewed

The diff for this file is too large to render. See raw diff