Update weights

Files changed (6) hide show

model-00001-of-00002.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:341ade02e4366270f99f2cdf3cf684b55821a4893aefb28c03422539fb0cf1dc
-size 4915467032

 version https://git-lfs.github.com/spec/v1
+oid sha256:18f6240e255c381876ddb8f3eec068f5e737e2184df8f6f48d0b05ee08ba3d9a
+size 4965798912

model-00002-of-00002.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:b1c26408c3dcf391d20077d954f05cb224deeb3272fb6f0bf016581a1d7928dd
-size 3120871248

 version https://git-lfs.github.com/spec/v1
+oid sha256:31bcbcbfe00248bab2b283e021997e56ba5aa06c1e09277506e7af80bdf0eb11
+size 2265183848

model.safetensors.index.json CHANGED Viewed

@@ -1,7 +1,7 @@
 {
   "metadata": {
     "total_parameters": 3212774400,
-    "total_size": 8036308992
   },
   "weight_map": {
     "model.embed_tokens.weight": "model-00001-of-00002.safetensors",
@@ -124,8 +124,8 @@
     "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.20.input_layernorm.weight": "model-00002-of-00002.safetensors",
     "model.layers.20.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
-    "model.layers.20.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
-    "model.layers.20.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
     "model.layers.20.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
     "model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",

 {
   "metadata": {
     "total_parameters": 3212774400,
+    "total_size": 7230953472
   },
   "weight_map": {
     "model.embed_tokens.weight": "model-00001-of-00002.safetensors",
     "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.20.input_layernorm.weight": "model-00002-of-00002.safetensors",
     "model.layers.20.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
+    "model.layers.20.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
+    "model.layers.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.20.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
     "model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",

tokenizer.json CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:a9d4fd2d4afa82d8a7dadae3490fdc20b26f06e32cec78a8dc96521b4dc79038
-size 17210200

 version https://git-lfs.github.com/spec/v1
+oid sha256:6b9e4e7fb171f92fd137b777cc2714bf87d11576700a1dcd7a399e7bbe39537b
+size 17209920

tokenizer_config.json CHANGED Viewed

@@ -2053,18 +2053,11 @@
   "clean_up_tokenization_spaces": true,
   "eos_token": "<|end_of_text|>",
   "extra_special_tokens": {},
-  "max_length": 512,
   "model_input_names": [
     "input_ids",
     "attention_mask"
   ],
   "model_max_length": 131072,
-  "pad_to_multiple_of": null,
   "pad_token": "<|end_of_text|>",
-  "pad_token_type_id": 0,
-  "padding_side": "right",
-  "stride": 0,
-  "tokenizer_class": "PreTrainedTokenizerFast",
-  "truncation_side": "right",
-  "truncation_strategy": "longest_first"
 }

   "clean_up_tokenization_spaces": true,
   "eos_token": "<|end_of_text|>",
   "extra_special_tokens": {},
   "model_input_names": [
     "input_ids",
     "attention_mask"
   ],
   "model_max_length": 131072,
   "pad_token": "<|end_of_text|>",
+  "tokenizer_class": "PreTrainedTokenizerFast"
 }

training_metadata.json CHANGED Viewed

@@ -1,6 +1,6 @@
 {
-  "epoch": 12,
-  "train_acc": 0.995978915348876,
-  "val_acc": 0.9951678746979922,
-  "val_loss": 0.054994360926616276
 }

 {
+  "epoch": 10,
+  "train_acc": 0.9972498281142571,
+  "val_acc": 0.9814213113388319,
+  "val_loss": 0.11411569089561124
 }