diff --git a/README.md b/README.md new file mode 100644 index 0000000000000000000000000000000000000000..bc5f30d6632ac0efdc7be2e9095e9e9579af2e33 --- /dev/null +++ b/README.md @@ -0,0 +1,199 @@ +--- +library_name: transformers +tags: [] +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + +This is the model card of a 🤗 transformers model that has been pushed on the Hub. This model card has been automatically generated. + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000000000000000000000000000000000000..682915665788bcb9a01276087568182666e6c5ab --- /dev/null +++ b/config.json @@ -0,0 +1,40 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "dtype": "bfloat16", + "eos_token_id": [ + 128001, + 128008, + 128009 + ], + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 8192, + "initializer_range": 0.02, + "intermediate_size": 28672, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 64, + "num_hidden_layers": 80, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 8.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": false, + "torch_dtype": "float32", + "transformers_version": "4.54.1", + "use_cache": true, + "vocab_size": 128256 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fb922075a9bf1e542ba674d5b5ed52a4c17fe46b --- /dev/null +++ b/generation_config.json @@ -0,0 +1,12 @@ +{ + "bos_token_id": 128000, + "do_sample": true, + "eos_token_id": [ + 128001, + 128008, + 128009 + ], + "temperature": 0.6, + "top_p": 0.9, + "transformers_version": "4.54.1" +} diff --git a/model-00001-of-00062.safetensors b/model-00001-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ce7ca916a714ef7624f96670a087a08632e8c6f2 --- /dev/null +++ b/model-00001-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:69ef2d3d31864b3ae4824353fab4e5ecf60acda594c66ab50ca0dfb23398286b +size 4806672984 diff --git a/model-00002-of-00062.safetensors b/model-00002-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d1dd4fbbb7c2a0e9622585f1aba39f8a059eb235 --- /dev/null +++ b/model-00002-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a77e02f8d92f3458eeb9fbbee695093adaf1a209f6d019650e90541bcd355fd8 +size 4362142864 diff --git a/model-00003-of-00062.safetensors b/model-00003-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..912cca0c5a06048f629ea96143f1b19635f07515 --- /dev/null +++ b/model-00003-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6bbcf365dce372cf6edcc9f3121364116b440f3060693a345c97a052c08772b2 +size 4362142864 diff --git a/model-00004-of-00062.safetensors b/model-00004-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0da7abaf68f8bccb4cd62ba49179aa82b4c177d0 --- /dev/null +++ b/model-00004-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e88cee1b5cc7128d5325707be57b8b0ff0ad5608fc39749c7594ff9507476f0a +size 4966188864 diff --git a/model-00005-of-00062.safetensors b/model-00005-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..8212909ac6592f177be93d12f4fd7b865ed5d092 --- /dev/null +++ b/model-00005-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:01657058dc41348b98f07ae1dab53c1f8153b932de64ff2da812a12ba3658a67 +size 4362142864 diff --git a/model-00006-of-00062.safetensors b/model-00006-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..91146da6b6c552fb47800280a2553b44026f458a --- /dev/null +++ b/model-00006-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:14e042f45f9ec888694c1defdeaff92970f2cc8edd7f4a7f82b2c18cf4262e4f +size 4362142864 diff --git a/model-00007-of-00062.safetensors b/model-00007-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..82e9cf6e59fd0d6d35960f31e5ab6de77af80ac1 --- /dev/null +++ b/model-00007-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f5cb037a05b5531d2df1f0e7d58f59f2f35008900f11f8e2c67affb149ab7826 +size 4966188864 diff --git a/model-00008-of-00062.safetensors b/model-00008-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..dc894584cd688a493d6a8dc61249bf0f0b73f145 --- /dev/null +++ b/model-00008-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8d4dea338f08f67a21f80cad3bad6e51637c339d2ff27ff085136b1d0b2d2d1a +size 4362142864 diff --git a/model-00009-of-00062.safetensors b/model-00009-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0b90f51f85aef5b18923cc40c5b6fcff1edce8ec --- /dev/null +++ b/model-00009-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:065577f7c67e872c4d7e1cca033abf9677daaf664b6a2955c364a2791b68803c +size 4362142880 diff --git a/model-00010-of-00062.safetensors b/model-00010-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..b94591d781a714a8b89068be1463c81f16c360d5 --- /dev/null +++ b/model-00010-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:86e22348ed2355e09aa77a04cf97aeb060163e5b010057d37bf5af50b5e4391f +size 4966188880 diff --git a/model-00011-of-00062.safetensors b/model-00011-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5bffd0f2862a291d3232c94ba8b23ab1b2ca60c4 --- /dev/null +++ b/model-00011-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5d9aeef52376e5a70526d230eac6f220d71a7e84bfcfe4fbdec709902dd8dea5 +size 4362142872 diff --git a/model-00012-of-00062.safetensors b/model-00012-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..8a0021ca46774424247f91e83d5324aa3d0b1b52 --- /dev/null +++ b/model-00012-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6f119505a1ee14b0dfad904594a8d192f2b4000a0c22c977d1023db8fc4e75c6 +size 4362142872 diff --git a/model-00013-of-00062.safetensors b/model-00013-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..27dbd99e0d979cffd49996a15497533f3d273f88 --- /dev/null +++ b/model-00013-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:52a31946391a36c0defbcc27b8609f0421a1d9712b35fdeb475b5b565958f42c +size 4966188880 diff --git a/model-00014-of-00062.safetensors b/model-00014-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..785bf9a8180aef4fce6070c21bac31441a59f652 --- /dev/null +++ b/model-00014-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bb3c27abce843bf5e5d4e2ff73445a240770d49b12b58e8bfb0898fe0cec560a +size 4362142872 diff --git a/model-00015-of-00062.safetensors b/model-00015-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d815b4876a4d592628abf46046b5dc85f1d785c9 --- /dev/null +++ b/model-00015-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3b6f9edfd8451834bb1216f559c76ef916f3316207b9c9b00c23759992f47103 +size 4362142872 diff --git a/model-00016-of-00062.safetensors b/model-00016-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..530fb6679d91d99ce988e8d0849a0ed45a41bbe4 --- /dev/null +++ b/model-00016-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:77408fcbc5d89bf5e1f5e93d47fb6c6dd9d4206c05352b63bb930210901ec31f +size 4966188880 diff --git a/model-00017-of-00062.safetensors b/model-00017-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..859098043f8efb2fbedf406480ca92cbc896c415 --- /dev/null +++ b/model-00017-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ef2883286692f0abd9cbdc154d2b64757dead466308e7dac7f338f164d6d0f37 +size 4362142872 diff --git a/model-00018-of-00062.safetensors b/model-00018-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4b46ee4277dc2f778dc3451d37cdcd1cbd10d8c7 --- /dev/null +++ b/model-00018-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:96b78c8c23b4d287ca49c913ff0424bbdab3719ab4daa382cfb269c4ac2f9a71 +size 4362142872 diff --git a/model-00019-of-00062.safetensors b/model-00019-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..96af3c214b83b8d39a071fb8b4156fc1ba832654 --- /dev/null +++ b/model-00019-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b1485cb1dcafe48a60a6ad46f55e8be93af5ae29338e338d0343134c3e8787df +size 4966188880 diff --git a/model-00020-of-00062.safetensors b/model-00020-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..30666726e23f61a46715590bda8a7263fa01d517 --- /dev/null +++ b/model-00020-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daff6d2cf0ac5b9bf19b7b3a78c9a0a05b92adea2a0bd92514aec493d1c5a33d +size 4362142872 diff --git a/model-00021-of-00062.safetensors b/model-00021-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6a7cf03e99cf242b3d13f01a3cc092705db2803f --- /dev/null +++ b/model-00021-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5258982ff63d9661df789ffa81d4a6311ac4323b05da826d979e717b19178d3d +size 4362142872 diff --git a/model-00022-of-00062.safetensors b/model-00022-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e02026efc7e0ae0066b2041892e2e409897868bd --- /dev/null +++ b/model-00022-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f1ea9bcd5d50103d06b73cdd12ad62bd250f33ca5c0b1b690e279ff4dc2505e +size 4966188880 diff --git a/model-00023-of-00062.safetensors b/model-00023-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..dc181f3f49e167e5522644276ecf894548eb654c --- /dev/null +++ b/model-00023-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:85020a3a9db9d80bc91e8212dc4a49a2068818fe9835565f91153c19137f95d6 +size 4362142872 diff --git a/model-00024-of-00062.safetensors b/model-00024-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..513dbad8aa33690a549e42688a9c36544da65f11 --- /dev/null +++ b/model-00024-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:edb276cb5aa8ade16fa1238516973e9d28257e516c0b1fdd9227ef7069b6f150 +size 4362142872 diff --git a/model-00025-of-00062.safetensors b/model-00025-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ffa9ec84ddbf0a4db2b4fb5261c518176f5c6ff6 --- /dev/null +++ b/model-00025-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2eaf4594be67b9765df0c537911670bf4f55b15a3f51ebb774a9a60c60d75f13 +size 4966188880 diff --git a/model-00026-of-00062.safetensors b/model-00026-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..b6c7f22bf5f77d2460d89a2100e28d306245150d --- /dev/null +++ b/model-00026-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d2fba91f67b9f26eff5dfd50f7acea689fb0d0381335ef069b70a8a52023b97f +size 4362142872 diff --git a/model-00027-of-00062.safetensors b/model-00027-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..116062385188ea3e4079d059f8333e7da3f3e002 --- /dev/null +++ b/model-00027-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:63dcddd79bb1066d81c3bef6bd2cb3e21c48605e2dd9d89b6242baedd2108a37 +size 4362142872 diff --git a/model-00028-of-00062.safetensors b/model-00028-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..43cdc9af52097ff17da141c34e13e65f97a525cf --- /dev/null +++ b/model-00028-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8d3f5b69f86d96a0071ef746bc920be520870efab4295a344c1abe4f37158d1d +size 4966188880 diff --git a/model-00029-of-00062.safetensors b/model-00029-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a4951448c3ab98e8b589e7417207657b463e2e64 --- /dev/null +++ b/model-00029-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:498320047e76a9fdfd48ed9e494160c42195d0e5e7ea9e3aeedac80a04cd3aa9 +size 4362142872 diff --git a/model-00030-of-00062.safetensors b/model-00030-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..496d9528638fb09ea88a7317ac34b2e66dd53a61 --- /dev/null +++ b/model-00030-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2b35abeff6781df47bba883c21677a99cabfbefcf1e2eaba04979936f22a9cc4 +size 4362142872 diff --git a/model-00031-of-00062.safetensors b/model-00031-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d0d496e713132c0e4cbbac1f981e7aae549bfe8c --- /dev/null +++ b/model-00031-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9a4c7d73c4aa85f7ff3580921b5a1be25310847100add692df70c7b2066d08b6 +size 4966188880 diff --git a/model-00032-of-00062.safetensors b/model-00032-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..032a08f24df032fc958b8a2558c8470287c6b77d --- /dev/null +++ b/model-00032-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d14efdc532bf24f04df7c349b1f1d5ec21c6f15809bc7b4337a5fb0f24fa3d3e +size 4362142872 diff --git a/model-00033-of-00062.safetensors b/model-00033-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..272209e1576dffe1cd37d15f08997e5fdb2083bf --- /dev/null +++ b/model-00033-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8e4433a40930b44fb9562a3ff31b73088465bcf1fe9622b39fbedb8e4849240b +size 4362142872 diff --git a/model-00034-of-00062.safetensors b/model-00034-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..fec8e9488e70d6f46cc0e2e32c20c60a6164d754 --- /dev/null +++ b/model-00034-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ca1341aadded58036ce44df0fb427eee088a03a368e50554f7ba89c3acf82bd2 +size 4966188880 diff --git a/model-00035-of-00062.safetensors b/model-00035-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..c1f4900af8b74258e1b4ada1eee8a153056a67e6 --- /dev/null +++ b/model-00035-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:373e55793dc2cc7208c81adcf601354acc5f2965e3426feaf99f464c3f73823f +size 4362142872 diff --git a/model-00036-of-00062.safetensors b/model-00036-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..89c0c633182ae3a89608be6b75709077c73cb46e --- /dev/null +++ b/model-00036-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8057a051ebb98bea94faf3dfc110358dfb5740ab9640823b0c0c6de3ded75c63 +size 4362142872 diff --git a/model-00037-of-00062.safetensors b/model-00037-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e9c3f541d6354bbb6723e8b4bcf4c069c5ec0e73 --- /dev/null +++ b/model-00037-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:74c0d223bbc02f709fe729e20ea1541f5e850925b6ededdef3835794c088c0fc +size 4966188880 diff --git a/model-00038-of-00062.safetensors b/model-00038-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..c1c3a6444b234bb3c708bd63a3e749ef5cbbf8e5 --- /dev/null +++ b/model-00038-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:91b7bc674fc6abdeeac591270a757939bc1df91d982fea55698e169fa787ecd5 +size 4362142872 diff --git a/model-00039-of-00062.safetensors b/model-00039-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..1e64b5b924b70753bb3969cbd191c2d138c6b8ed --- /dev/null +++ b/model-00039-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:90e22d7d5f29db536ee30219691ef8d219516e29dfb001504b2a23991a53e83a +size 4362142872 diff --git a/model-00040-of-00062.safetensors b/model-00040-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a3eabf65bc1cc59aa2caf1f7d1fe9d9ea457f776 --- /dev/null +++ b/model-00040-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f08e57a58d4c14363263d44f96c2f678427df52200553a4396a107119ce67236 +size 4966188880 diff --git a/model-00041-of-00062.safetensors b/model-00041-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2dfae58ebf8422b31083770d3339a9ed5b618288 --- /dev/null +++ b/model-00041-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c8f0f9f321b81e193674eed0a0c656e97ec67bb0a51ebd4c881d73e83149e262 +size 4362142872 diff --git a/model-00042-of-00062.safetensors b/model-00042-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5c46eee19d3e0388798d4eef1177b4fef09e5294 --- /dev/null +++ b/model-00042-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9adbb1adc7f0e3b48e60924f48045b060e7ab1244c8280cdd3984230acbf03e2 +size 4362142872 diff --git a/model-00043-of-00062.safetensors b/model-00043-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5766565cb976e13b3f899187eaa4e6a507f8739c --- /dev/null +++ b/model-00043-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c23a4b9af5bd87a980a09ee014c628ab84fd05a3e5662c3d37f3cc3bf966b2f +size 4966188880 diff --git a/model-00044-of-00062.safetensors b/model-00044-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..cc328e91f02932096b5363cc0aee5bdd3864ec75 --- /dev/null +++ b/model-00044-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f4f49a7b2d74c2fef89f4a1c33e1284cf8bdb90b2bdc1816bb32f36b8fbedd3e +size 4362142872 diff --git a/model-00045-of-00062.safetensors b/model-00045-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9598939972f06c45e43dc4eab2ce576f3a1a8cc3 --- /dev/null +++ b/model-00045-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8cca93aeb685af05785258998bc5037af69e0704905902b549b5cbd5b896c052 +size 4362142872 diff --git a/model-00046-of-00062.safetensors b/model-00046-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..974f3d3e849dcd9c90c8c05a39268755da5aeafc --- /dev/null +++ b/model-00046-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b20a18abb5d715f9f702ad3ee1bb07e9a8792c14165c254fea1837b8770a1709 +size 4966188880 diff --git a/model-00047-of-00062.safetensors b/model-00047-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..55bbfbb26491fb609c48bc27a5f10a4eb2026624 --- /dev/null +++ b/model-00047-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:49b38b43e1f38eaabd7ba5514f2b406d4245fb856aff09428f5802cae9a16974 +size 4362142872 diff --git a/model-00048-of-00062.safetensors b/model-00048-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..819fa9847d297671780b156aa796e90fa6ee65ad --- /dev/null +++ b/model-00048-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e60496cc700150636a5c59c76debfa287bfd69becb8ec7b9bde37326f37f6607 +size 4362142872 diff --git a/model-00049-of-00062.safetensors b/model-00049-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7339ab52a16697a107462ed400dafff23b0d2ff4 --- /dev/null +++ b/model-00049-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:86d3500450c4184fbf62474ec2f346fd020e5f302d8d4ecb68b44b1bb7829dd8 +size 4966188880 diff --git a/model-00050-of-00062.safetensors b/model-00050-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..c9cd3cf318f4c4015b8108dcedaed71b27efc70d --- /dev/null +++ b/model-00050-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b609ed2777879e28ccee7c63881f74d5e3fb3dda9ee3468875f8ce722d12346d +size 4362142872 diff --git a/model-00051-of-00062.safetensors b/model-00051-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a1b4c3eafa1c0dfa709523ed4b4fe69ea691cadc --- /dev/null +++ b/model-00051-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f169f141b1989d5e7ca2c017c5f7663f99d86ebb4ab5a9d21bd5bf7b5cf5294 +size 4362142872 diff --git a/model-00052-of-00062.safetensors b/model-00052-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..39dacbf892bba354728a0ba78dbddee49716e260 --- /dev/null +++ b/model-00052-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bd21fa0f796253457f2e6341d75b372fb77c600c6a54b5c115c5a8ec89b4195a +size 4966188880 diff --git a/model-00053-of-00062.safetensors b/model-00053-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..db77c61e3aec02bf16b91137860ba15828e40933 --- /dev/null +++ b/model-00053-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d5d86c5359e39376de08b7c25867bb809aa7dab6d1a7b26c85b60713a215340b +size 4362142872 diff --git a/model-00054-of-00062.safetensors b/model-00054-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d190e94305c62a9f079200d0a1072b964b8dbf1c --- /dev/null +++ b/model-00054-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:325fe307134b9f6a8da3266edc01d5f1b57dafa08ea6bb132e9b180ee3ca5e32 +size 4362142872 diff --git a/model-00055-of-00062.safetensors b/model-00055-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2952bc49bf01cc384e962b46bb66767a8d59e2fd --- /dev/null +++ b/model-00055-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:463ffa6e3ddd30d2519d604be01fddc5b2fea879656042136e6706c45629188a +size 4966188880 diff --git a/model-00056-of-00062.safetensors b/model-00056-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..96d10e224df3fd25790cdab9386fc5e1ea591e3d --- /dev/null +++ b/model-00056-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0d9205ca9fda5aa41247b0db2baae82880a42443ff547ff12c0ff27ed9ad5520 +size 4362142872 diff --git a/model-00057-of-00062.safetensors b/model-00057-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..87e5c665d599c92938a63ffb4b6bbb1632f4e209 --- /dev/null +++ b/model-00057-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c1c4526cb3588d6f5fddef160fe5266a5e8f6d1f81f94b079f7663f9b662d4d +size 4362142872 diff --git a/model-00058-of-00062.safetensors b/model-00058-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..097dfd4355fd74d71f41037b14338d010e0fb857 --- /dev/null +++ b/model-00058-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:44e9053986ecfd33aa44b07c3b5cce08851ba3e900b37963975f3182eedefb04 +size 4966188880 diff --git a/model-00059-of-00062.safetensors b/model-00059-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..58f8358bd5f05f7a1e729042762efb76f545776b --- /dev/null +++ b/model-00059-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fc1061d315604ee06156ecdb88a592c145d5a6c5a266658c449ab2378109fb57 +size 4362142872 diff --git a/model-00060-of-00062.safetensors b/model-00060-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6f11ceb3a5befc3c0ef2905ef849d2fd6dd33595 --- /dev/null +++ b/model-00060-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1aace8e144b51ca8e5723b244189a602b66b07578757ba506211f82e60e030ee +size 4362142872 diff --git a/model-00061-of-00062.safetensors b/model-00061-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e0fff5f1fe61543d12d1e8bd4ee9c99f8bedb1b1 --- /dev/null +++ b/model-00061-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f4692111b8e60073c9ed7ca417a09e8624514e9ea097fa302e2d53faec96140a +size 4362241496 diff --git a/model-00062-of-00062.safetensors b/model-00062-of-00062.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..908a6bd1df871d81d23f41bd765c1d5c2abf525a --- /dev/null +++ b/model-00062-of-00062.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:37adba49b605e4b59b81ef2930475973ff37b20bd4ddd0fbc20c302fc3c112d5 +size 4202692736 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000000000000000000000000000000000000..ef0c642b145f0661c97c573578ea5f5c868b5c19 --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,731 @@ +{ + "metadata": { + "total_parameters": 70553706496, + "total_size": 282214825984 + }, + "weight_map": { + "lm_head.weight": "model-00062-of-00062.safetensors", + "model.embed_tokens.weight": "model-00001-of-00062.safetensors", + "model.layers.0.input_layernorm.weight": "model-00002-of-00062.safetensors", + "model.layers.0.mlp.down_proj.weight": "model-00002-of-00062.safetensors", + "model.layers.0.mlp.gate_proj.weight": "model-00002-of-00062.safetensors", + "model.layers.0.mlp.up_proj.weight": "model-00002-of-00062.safetensors", + "model.layers.0.post_attention_layernorm.weight": "model-00002-of-00062.safetensors", + "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00062.safetensors", + "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00062.safetensors", + "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00062.safetensors", + "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00062.safetensors", + "model.layers.1.input_layernorm.weight": "model-00003-of-00062.safetensors", + "model.layers.1.mlp.down_proj.weight": "model-00003-of-00062.safetensors", + "model.layers.1.mlp.gate_proj.weight": "model-00002-of-00062.safetensors", + "model.layers.1.mlp.up_proj.weight": "model-00003-of-00062.safetensors", + "model.layers.1.post_attention_layernorm.weight": "model-00003-of-00062.safetensors", + "model.layers.1.self_attn.k_proj.weight": "model-00002-of-00062.safetensors", + "model.layers.1.self_attn.o_proj.weight": "model-00002-of-00062.safetensors", + "model.layers.1.self_attn.q_proj.weight": "model-00002-of-00062.safetensors", + "model.layers.1.self_attn.v_proj.weight": "model-00002-of-00062.safetensors", + "model.layers.10.input_layernorm.weight": "model-00010-of-00062.safetensors", + "model.layers.10.mlp.down_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.10.mlp.gate_proj.weight": "model-00009-of-00062.safetensors", + "model.layers.10.mlp.up_proj.weight": "model-00009-of-00062.safetensors", + "model.layers.10.post_attention_layernorm.weight": "model-00010-of-00062.safetensors", + "model.layers.10.self_attn.k_proj.weight": "model-00009-of-00062.safetensors", + "model.layers.10.self_attn.o_proj.weight": "model-00009-of-00062.safetensors", + "model.layers.10.self_attn.q_proj.weight": "model-00009-of-00062.safetensors", + "model.layers.10.self_attn.v_proj.weight": "model-00009-of-00062.safetensors", + "model.layers.11.input_layernorm.weight": "model-00010-of-00062.safetensors", + "model.layers.11.mlp.down_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.11.mlp.gate_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.11.mlp.up_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.11.post_attention_layernorm.weight": "model-00010-of-00062.safetensors", + "model.layers.11.self_attn.k_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.11.self_attn.o_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.11.self_attn.q_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.11.self_attn.v_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.12.input_layernorm.weight": "model-00011-of-00062.safetensors", + "model.layers.12.mlp.down_proj.weight": "model-00011-of-00062.safetensors", + "model.layers.12.mlp.gate_proj.weight": "model-00011-of-00062.safetensors", + "model.layers.12.mlp.up_proj.weight": "model-00011-of-00062.safetensors", + "model.layers.12.post_attention_layernorm.weight": "model-00011-of-00062.safetensors", + "model.layers.12.self_attn.k_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.12.self_attn.o_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.12.self_attn.q_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.12.self_attn.v_proj.weight": "model-00010-of-00062.safetensors", + "model.layers.13.input_layernorm.weight": "model-00012-of-00062.safetensors", + "model.layers.13.mlp.down_proj.weight": "model-00012-of-00062.safetensors", + "model.layers.13.mlp.gate_proj.weight": "model-00011-of-00062.safetensors", + "model.layers.13.mlp.up_proj.weight": "model-00012-of-00062.safetensors", + "model.layers.13.post_attention_layernorm.weight": "model-00012-of-00062.safetensors", + "model.layers.13.self_attn.k_proj.weight": "model-00011-of-00062.safetensors", + "model.layers.13.self_attn.o_proj.weight": "model-00011-of-00062.safetensors", + "model.layers.13.self_attn.q_proj.weight": "model-00011-of-00062.safetensors", + "model.layers.13.self_attn.v_proj.weight": "model-00011-of-00062.safetensors", + "model.layers.14.input_layernorm.weight": "model-00013-of-00062.safetensors", + "model.layers.14.mlp.down_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.14.mlp.gate_proj.weight": "model-00012-of-00062.safetensors", + "model.layers.14.mlp.up_proj.weight": "model-00012-of-00062.safetensors", + "model.layers.14.post_attention_layernorm.weight": "model-00013-of-00062.safetensors", + "model.layers.14.self_attn.k_proj.weight": "model-00012-of-00062.safetensors", + "model.layers.14.self_attn.o_proj.weight": "model-00012-of-00062.safetensors", + "model.layers.14.self_attn.q_proj.weight": "model-00012-of-00062.safetensors", + "model.layers.14.self_attn.v_proj.weight": "model-00012-of-00062.safetensors", + "model.layers.15.input_layernorm.weight": "model-00013-of-00062.safetensors", + "model.layers.15.mlp.down_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.15.mlp.gate_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.15.mlp.up_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.15.post_attention_layernorm.weight": "model-00013-of-00062.safetensors", + "model.layers.15.self_attn.k_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.15.self_attn.o_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.15.self_attn.q_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.15.self_attn.v_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.16.input_layernorm.weight": "model-00014-of-00062.safetensors", + "model.layers.16.mlp.down_proj.weight": "model-00014-of-00062.safetensors", + "model.layers.16.mlp.gate_proj.weight": "model-00014-of-00062.safetensors", + "model.layers.16.mlp.up_proj.weight": "model-00014-of-00062.safetensors", + "model.layers.16.post_attention_layernorm.weight": "model-00014-of-00062.safetensors", + "model.layers.16.self_attn.k_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.16.self_attn.o_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.16.self_attn.q_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.16.self_attn.v_proj.weight": "model-00013-of-00062.safetensors", + "model.layers.17.input_layernorm.weight": "model-00015-of-00062.safetensors", + "model.layers.17.mlp.down_proj.weight": "model-00015-of-00062.safetensors", + "model.layers.17.mlp.gate_proj.weight": "model-00014-of-00062.safetensors", + "model.layers.17.mlp.up_proj.weight": "model-00015-of-00062.safetensors", + "model.layers.17.post_attention_layernorm.weight": "model-00015-of-00062.safetensors", + "model.layers.17.self_attn.k_proj.weight": "model-00014-of-00062.safetensors", + "model.layers.17.self_attn.o_proj.weight": "model-00014-of-00062.safetensors", + "model.layers.17.self_attn.q_proj.weight": "model-00014-of-00062.safetensors", + "model.layers.17.self_attn.v_proj.weight": "model-00014-of-00062.safetensors", + "model.layers.18.input_layernorm.weight": "model-00016-of-00062.safetensors", + "model.layers.18.mlp.down_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.18.mlp.gate_proj.weight": "model-00015-of-00062.safetensors", + "model.layers.18.mlp.up_proj.weight": "model-00015-of-00062.safetensors", + "model.layers.18.post_attention_layernorm.weight": "model-00016-of-00062.safetensors", + "model.layers.18.self_attn.k_proj.weight": "model-00015-of-00062.safetensors", + "model.layers.18.self_attn.o_proj.weight": "model-00015-of-00062.safetensors", + "model.layers.18.self_attn.q_proj.weight": "model-00015-of-00062.safetensors", + "model.layers.18.self_attn.v_proj.weight": "model-00015-of-00062.safetensors", + "model.layers.19.input_layernorm.weight": "model-00016-of-00062.safetensors", + "model.layers.19.mlp.down_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.19.mlp.gate_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.19.mlp.up_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.19.post_attention_layernorm.weight": "model-00016-of-00062.safetensors", + "model.layers.19.self_attn.k_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.19.self_attn.o_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.19.self_attn.q_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.19.self_attn.v_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.2.input_layernorm.weight": "model-00004-of-00062.safetensors", + "model.layers.2.mlp.down_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.2.mlp.gate_proj.weight": "model-00003-of-00062.safetensors", + "model.layers.2.mlp.up_proj.weight": "model-00003-of-00062.safetensors", + "model.layers.2.post_attention_layernorm.weight": "model-00004-of-00062.safetensors", + "model.layers.2.self_attn.k_proj.weight": "model-00003-of-00062.safetensors", + "model.layers.2.self_attn.o_proj.weight": "model-00003-of-00062.safetensors", + "model.layers.2.self_attn.q_proj.weight": "model-00003-of-00062.safetensors", + "model.layers.2.self_attn.v_proj.weight": "model-00003-of-00062.safetensors", + "model.layers.20.input_layernorm.weight": "model-00017-of-00062.safetensors", + "model.layers.20.mlp.down_proj.weight": "model-00017-of-00062.safetensors", + "model.layers.20.mlp.gate_proj.weight": "model-00017-of-00062.safetensors", + "model.layers.20.mlp.up_proj.weight": "model-00017-of-00062.safetensors", + "model.layers.20.post_attention_layernorm.weight": "model-00017-of-00062.safetensors", + "model.layers.20.self_attn.k_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.20.self_attn.o_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.20.self_attn.q_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.20.self_attn.v_proj.weight": "model-00016-of-00062.safetensors", + "model.layers.21.input_layernorm.weight": "model-00018-of-00062.safetensors", + "model.layers.21.mlp.down_proj.weight": "model-00018-of-00062.safetensors", + "model.layers.21.mlp.gate_proj.weight": "model-00017-of-00062.safetensors", + "model.layers.21.mlp.up_proj.weight": "model-00018-of-00062.safetensors", + "model.layers.21.post_attention_layernorm.weight": "model-00018-of-00062.safetensors", + "model.layers.21.self_attn.k_proj.weight": "model-00017-of-00062.safetensors", + "model.layers.21.self_attn.o_proj.weight": "model-00017-of-00062.safetensors", + "model.layers.21.self_attn.q_proj.weight": "model-00017-of-00062.safetensors", + "model.layers.21.self_attn.v_proj.weight": "model-00017-of-00062.safetensors", + "model.layers.22.input_layernorm.weight": "model-00019-of-00062.safetensors", + "model.layers.22.mlp.down_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.22.mlp.gate_proj.weight": "model-00018-of-00062.safetensors", + "model.layers.22.mlp.up_proj.weight": "model-00018-of-00062.safetensors", + "model.layers.22.post_attention_layernorm.weight": "model-00019-of-00062.safetensors", + "model.layers.22.self_attn.k_proj.weight": "model-00018-of-00062.safetensors", + "model.layers.22.self_attn.o_proj.weight": "model-00018-of-00062.safetensors", + "model.layers.22.self_attn.q_proj.weight": "model-00018-of-00062.safetensors", + "model.layers.22.self_attn.v_proj.weight": "model-00018-of-00062.safetensors", + "model.layers.23.input_layernorm.weight": "model-00019-of-00062.safetensors", + "model.layers.23.mlp.down_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.23.mlp.gate_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.23.mlp.up_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.23.post_attention_layernorm.weight": "model-00019-of-00062.safetensors", + "model.layers.23.self_attn.k_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.23.self_attn.o_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.23.self_attn.q_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.23.self_attn.v_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.24.input_layernorm.weight": "model-00020-of-00062.safetensors", + "model.layers.24.mlp.down_proj.weight": "model-00020-of-00062.safetensors", + "model.layers.24.mlp.gate_proj.weight": "model-00020-of-00062.safetensors", + "model.layers.24.mlp.up_proj.weight": "model-00020-of-00062.safetensors", + "model.layers.24.post_attention_layernorm.weight": "model-00020-of-00062.safetensors", + "model.layers.24.self_attn.k_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.24.self_attn.o_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.24.self_attn.q_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.24.self_attn.v_proj.weight": "model-00019-of-00062.safetensors", + "model.layers.25.input_layernorm.weight": "model-00021-of-00062.safetensors", + "model.layers.25.mlp.down_proj.weight": "model-00021-of-00062.safetensors", + "model.layers.25.mlp.gate_proj.weight": "model-00020-of-00062.safetensors", + "model.layers.25.mlp.up_proj.weight": "model-00021-of-00062.safetensors", + "model.layers.25.post_attention_layernorm.weight": "model-00021-of-00062.safetensors", + "model.layers.25.self_attn.k_proj.weight": "model-00020-of-00062.safetensors", + "model.layers.25.self_attn.o_proj.weight": "model-00020-of-00062.safetensors", + "model.layers.25.self_attn.q_proj.weight": "model-00020-of-00062.safetensors", + "model.layers.25.self_attn.v_proj.weight": "model-00020-of-00062.safetensors", + "model.layers.26.input_layernorm.weight": "model-00022-of-00062.safetensors", + "model.layers.26.mlp.down_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.26.mlp.gate_proj.weight": "model-00021-of-00062.safetensors", + "model.layers.26.mlp.up_proj.weight": "model-00021-of-00062.safetensors", + "model.layers.26.post_attention_layernorm.weight": "model-00022-of-00062.safetensors", + "model.layers.26.self_attn.k_proj.weight": "model-00021-of-00062.safetensors", + "model.layers.26.self_attn.o_proj.weight": "model-00021-of-00062.safetensors", + "model.layers.26.self_attn.q_proj.weight": "model-00021-of-00062.safetensors", + "model.layers.26.self_attn.v_proj.weight": "model-00021-of-00062.safetensors", + "model.layers.27.input_layernorm.weight": "model-00022-of-00062.safetensors", + "model.layers.27.mlp.down_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.27.mlp.gate_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.27.mlp.up_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.27.post_attention_layernorm.weight": "model-00022-of-00062.safetensors", + "model.layers.27.self_attn.k_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.27.self_attn.o_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.27.self_attn.q_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.27.self_attn.v_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.28.input_layernorm.weight": "model-00023-of-00062.safetensors", + "model.layers.28.mlp.down_proj.weight": "model-00023-of-00062.safetensors", + "model.layers.28.mlp.gate_proj.weight": "model-00023-of-00062.safetensors", + "model.layers.28.mlp.up_proj.weight": "model-00023-of-00062.safetensors", + "model.layers.28.post_attention_layernorm.weight": "model-00023-of-00062.safetensors", + "model.layers.28.self_attn.k_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.28.self_attn.o_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.28.self_attn.q_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.28.self_attn.v_proj.weight": "model-00022-of-00062.safetensors", + "model.layers.29.input_layernorm.weight": "model-00024-of-00062.safetensors", + "model.layers.29.mlp.down_proj.weight": "model-00024-of-00062.safetensors", + "model.layers.29.mlp.gate_proj.weight": "model-00023-of-00062.safetensors", + "model.layers.29.mlp.up_proj.weight": "model-00024-of-00062.safetensors", + "model.layers.29.post_attention_layernorm.weight": "model-00024-of-00062.safetensors", + "model.layers.29.self_attn.k_proj.weight": "model-00023-of-00062.safetensors", + "model.layers.29.self_attn.o_proj.weight": "model-00023-of-00062.safetensors", + "model.layers.29.self_attn.q_proj.weight": "model-00023-of-00062.safetensors", + "model.layers.29.self_attn.v_proj.weight": "model-00023-of-00062.safetensors", + "model.layers.3.input_layernorm.weight": "model-00004-of-00062.safetensors", + "model.layers.3.mlp.down_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.3.mlp.gate_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.3.mlp.up_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.3.post_attention_layernorm.weight": "model-00004-of-00062.safetensors", + "model.layers.3.self_attn.k_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.3.self_attn.o_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.3.self_attn.q_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.3.self_attn.v_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.30.input_layernorm.weight": "model-00025-of-00062.safetensors", + "model.layers.30.mlp.down_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.30.mlp.gate_proj.weight": "model-00024-of-00062.safetensors", + "model.layers.30.mlp.up_proj.weight": "model-00024-of-00062.safetensors", + "model.layers.30.post_attention_layernorm.weight": "model-00025-of-00062.safetensors", + "model.layers.30.self_attn.k_proj.weight": "model-00024-of-00062.safetensors", + "model.layers.30.self_attn.o_proj.weight": "model-00024-of-00062.safetensors", + "model.layers.30.self_attn.q_proj.weight": "model-00024-of-00062.safetensors", + "model.layers.30.self_attn.v_proj.weight": "model-00024-of-00062.safetensors", + "model.layers.31.input_layernorm.weight": "model-00025-of-00062.safetensors", + "model.layers.31.mlp.down_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.31.mlp.gate_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.31.mlp.up_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.31.post_attention_layernorm.weight": "model-00025-of-00062.safetensors", + "model.layers.31.self_attn.k_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.31.self_attn.o_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.31.self_attn.q_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.31.self_attn.v_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.32.input_layernorm.weight": "model-00026-of-00062.safetensors", + "model.layers.32.mlp.down_proj.weight": "model-00026-of-00062.safetensors", + "model.layers.32.mlp.gate_proj.weight": "model-00026-of-00062.safetensors", + "model.layers.32.mlp.up_proj.weight": "model-00026-of-00062.safetensors", + "model.layers.32.post_attention_layernorm.weight": "model-00026-of-00062.safetensors", + "model.layers.32.self_attn.k_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.32.self_attn.o_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.32.self_attn.q_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.32.self_attn.v_proj.weight": "model-00025-of-00062.safetensors", + "model.layers.33.input_layernorm.weight": "model-00027-of-00062.safetensors", + "model.layers.33.mlp.down_proj.weight": "model-00027-of-00062.safetensors", + "model.layers.33.mlp.gate_proj.weight": "model-00026-of-00062.safetensors", + "model.layers.33.mlp.up_proj.weight": "model-00027-of-00062.safetensors", + "model.layers.33.post_attention_layernorm.weight": "model-00027-of-00062.safetensors", + "model.layers.33.self_attn.k_proj.weight": "model-00026-of-00062.safetensors", + "model.layers.33.self_attn.o_proj.weight": "model-00026-of-00062.safetensors", + "model.layers.33.self_attn.q_proj.weight": "model-00026-of-00062.safetensors", + "model.layers.33.self_attn.v_proj.weight": "model-00026-of-00062.safetensors", + "model.layers.34.input_layernorm.weight": "model-00028-of-00062.safetensors", + "model.layers.34.mlp.down_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.34.mlp.gate_proj.weight": "model-00027-of-00062.safetensors", + "model.layers.34.mlp.up_proj.weight": "model-00027-of-00062.safetensors", + "model.layers.34.post_attention_layernorm.weight": "model-00028-of-00062.safetensors", + "model.layers.34.self_attn.k_proj.weight": "model-00027-of-00062.safetensors", + "model.layers.34.self_attn.o_proj.weight": "model-00027-of-00062.safetensors", + "model.layers.34.self_attn.q_proj.weight": "model-00027-of-00062.safetensors", + "model.layers.34.self_attn.v_proj.weight": "model-00027-of-00062.safetensors", + "model.layers.35.input_layernorm.weight": "model-00028-of-00062.safetensors", + "model.layers.35.mlp.down_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.35.mlp.gate_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.35.mlp.up_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.35.post_attention_layernorm.weight": "model-00028-of-00062.safetensors", + "model.layers.35.self_attn.k_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.35.self_attn.o_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.35.self_attn.q_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.35.self_attn.v_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.36.input_layernorm.weight": "model-00029-of-00062.safetensors", + "model.layers.36.mlp.down_proj.weight": "model-00029-of-00062.safetensors", + "model.layers.36.mlp.gate_proj.weight": "model-00029-of-00062.safetensors", + "model.layers.36.mlp.up_proj.weight": "model-00029-of-00062.safetensors", + "model.layers.36.post_attention_layernorm.weight": "model-00029-of-00062.safetensors", + "model.layers.36.self_attn.k_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.36.self_attn.o_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.36.self_attn.q_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.36.self_attn.v_proj.weight": "model-00028-of-00062.safetensors", + "model.layers.37.input_layernorm.weight": "model-00030-of-00062.safetensors", + "model.layers.37.mlp.down_proj.weight": "model-00030-of-00062.safetensors", + "model.layers.37.mlp.gate_proj.weight": "model-00029-of-00062.safetensors", + "model.layers.37.mlp.up_proj.weight": "model-00030-of-00062.safetensors", + "model.layers.37.post_attention_layernorm.weight": "model-00030-of-00062.safetensors", + "model.layers.37.self_attn.k_proj.weight": "model-00029-of-00062.safetensors", + "model.layers.37.self_attn.o_proj.weight": "model-00029-of-00062.safetensors", + "model.layers.37.self_attn.q_proj.weight": "model-00029-of-00062.safetensors", + "model.layers.37.self_attn.v_proj.weight": "model-00029-of-00062.safetensors", + "model.layers.38.input_layernorm.weight": "model-00031-of-00062.safetensors", + "model.layers.38.mlp.down_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.38.mlp.gate_proj.weight": "model-00030-of-00062.safetensors", + "model.layers.38.mlp.up_proj.weight": "model-00030-of-00062.safetensors", + "model.layers.38.post_attention_layernorm.weight": "model-00031-of-00062.safetensors", + "model.layers.38.self_attn.k_proj.weight": "model-00030-of-00062.safetensors", + "model.layers.38.self_attn.o_proj.weight": "model-00030-of-00062.safetensors", + "model.layers.38.self_attn.q_proj.weight": "model-00030-of-00062.safetensors", + "model.layers.38.self_attn.v_proj.weight": "model-00030-of-00062.safetensors", + "model.layers.39.input_layernorm.weight": "model-00031-of-00062.safetensors", + "model.layers.39.mlp.down_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.39.mlp.gate_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.39.mlp.up_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.39.post_attention_layernorm.weight": "model-00031-of-00062.safetensors", + "model.layers.39.self_attn.k_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.39.self_attn.o_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.39.self_attn.q_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.39.self_attn.v_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.4.input_layernorm.weight": "model-00005-of-00062.safetensors", + "model.layers.4.mlp.down_proj.weight": "model-00005-of-00062.safetensors", + "model.layers.4.mlp.gate_proj.weight": "model-00005-of-00062.safetensors", + "model.layers.4.mlp.up_proj.weight": "model-00005-of-00062.safetensors", + "model.layers.4.post_attention_layernorm.weight": "model-00005-of-00062.safetensors", + "model.layers.4.self_attn.k_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.4.self_attn.o_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.4.self_attn.q_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.4.self_attn.v_proj.weight": "model-00004-of-00062.safetensors", + "model.layers.40.input_layernorm.weight": "model-00032-of-00062.safetensors", + "model.layers.40.mlp.down_proj.weight": "model-00032-of-00062.safetensors", + "model.layers.40.mlp.gate_proj.weight": "model-00032-of-00062.safetensors", + "model.layers.40.mlp.up_proj.weight": "model-00032-of-00062.safetensors", + "model.layers.40.post_attention_layernorm.weight": "model-00032-of-00062.safetensors", + "model.layers.40.self_attn.k_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.40.self_attn.o_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.40.self_attn.q_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.40.self_attn.v_proj.weight": "model-00031-of-00062.safetensors", + "model.layers.41.input_layernorm.weight": "model-00033-of-00062.safetensors", + "model.layers.41.mlp.down_proj.weight": "model-00033-of-00062.safetensors", + "model.layers.41.mlp.gate_proj.weight": "model-00032-of-00062.safetensors", + "model.layers.41.mlp.up_proj.weight": "model-00033-of-00062.safetensors", + "model.layers.41.post_attention_layernorm.weight": "model-00033-of-00062.safetensors", + "model.layers.41.self_attn.k_proj.weight": "model-00032-of-00062.safetensors", + "model.layers.41.self_attn.o_proj.weight": "model-00032-of-00062.safetensors", + "model.layers.41.self_attn.q_proj.weight": "model-00032-of-00062.safetensors", + "model.layers.41.self_attn.v_proj.weight": "model-00032-of-00062.safetensors", + "model.layers.42.input_layernorm.weight": "model-00034-of-00062.safetensors", + "model.layers.42.mlp.down_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.42.mlp.gate_proj.weight": "model-00033-of-00062.safetensors", + "model.layers.42.mlp.up_proj.weight": "model-00033-of-00062.safetensors", + "model.layers.42.post_attention_layernorm.weight": "model-00034-of-00062.safetensors", + "model.layers.42.self_attn.k_proj.weight": "model-00033-of-00062.safetensors", + "model.layers.42.self_attn.o_proj.weight": "model-00033-of-00062.safetensors", + "model.layers.42.self_attn.q_proj.weight": "model-00033-of-00062.safetensors", + "model.layers.42.self_attn.v_proj.weight": "model-00033-of-00062.safetensors", + "model.layers.43.input_layernorm.weight": "model-00034-of-00062.safetensors", + "model.layers.43.mlp.down_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.43.mlp.gate_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.43.mlp.up_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.43.post_attention_layernorm.weight": "model-00034-of-00062.safetensors", + "model.layers.43.self_attn.k_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.43.self_attn.o_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.43.self_attn.q_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.43.self_attn.v_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.44.input_layernorm.weight": "model-00035-of-00062.safetensors", + "model.layers.44.mlp.down_proj.weight": "model-00035-of-00062.safetensors", + "model.layers.44.mlp.gate_proj.weight": "model-00035-of-00062.safetensors", + "model.layers.44.mlp.up_proj.weight": "model-00035-of-00062.safetensors", + "model.layers.44.post_attention_layernorm.weight": "model-00035-of-00062.safetensors", + "model.layers.44.self_attn.k_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.44.self_attn.o_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.44.self_attn.q_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.44.self_attn.v_proj.weight": "model-00034-of-00062.safetensors", + "model.layers.45.input_layernorm.weight": "model-00036-of-00062.safetensors", + "model.layers.45.mlp.down_proj.weight": "model-00036-of-00062.safetensors", + "model.layers.45.mlp.gate_proj.weight": "model-00035-of-00062.safetensors", + "model.layers.45.mlp.up_proj.weight": "model-00036-of-00062.safetensors", + "model.layers.45.post_attention_layernorm.weight": "model-00036-of-00062.safetensors", + "model.layers.45.self_attn.k_proj.weight": "model-00035-of-00062.safetensors", + "model.layers.45.self_attn.o_proj.weight": "model-00035-of-00062.safetensors", + "model.layers.45.self_attn.q_proj.weight": "model-00035-of-00062.safetensors", + "model.layers.45.self_attn.v_proj.weight": "model-00035-of-00062.safetensors", + "model.layers.46.input_layernorm.weight": "model-00037-of-00062.safetensors", + "model.layers.46.mlp.down_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.46.mlp.gate_proj.weight": "model-00036-of-00062.safetensors", + "model.layers.46.mlp.up_proj.weight": "model-00036-of-00062.safetensors", + "model.layers.46.post_attention_layernorm.weight": "model-00037-of-00062.safetensors", + "model.layers.46.self_attn.k_proj.weight": "model-00036-of-00062.safetensors", + "model.layers.46.self_attn.o_proj.weight": "model-00036-of-00062.safetensors", + "model.layers.46.self_attn.q_proj.weight": "model-00036-of-00062.safetensors", + "model.layers.46.self_attn.v_proj.weight": "model-00036-of-00062.safetensors", + "model.layers.47.input_layernorm.weight": "model-00037-of-00062.safetensors", + "model.layers.47.mlp.down_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.47.mlp.gate_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.47.mlp.up_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.47.post_attention_layernorm.weight": "model-00037-of-00062.safetensors", + "model.layers.47.self_attn.k_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.47.self_attn.o_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.47.self_attn.q_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.47.self_attn.v_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.48.input_layernorm.weight": "model-00038-of-00062.safetensors", + "model.layers.48.mlp.down_proj.weight": "model-00038-of-00062.safetensors", + "model.layers.48.mlp.gate_proj.weight": "model-00038-of-00062.safetensors", + "model.layers.48.mlp.up_proj.weight": "model-00038-of-00062.safetensors", + "model.layers.48.post_attention_layernorm.weight": "model-00038-of-00062.safetensors", + "model.layers.48.self_attn.k_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.48.self_attn.o_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.48.self_attn.q_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.48.self_attn.v_proj.weight": "model-00037-of-00062.safetensors", + "model.layers.49.input_layernorm.weight": "model-00039-of-00062.safetensors", + "model.layers.49.mlp.down_proj.weight": "model-00039-of-00062.safetensors", + "model.layers.49.mlp.gate_proj.weight": "model-00038-of-00062.safetensors", + "model.layers.49.mlp.up_proj.weight": "model-00039-of-00062.safetensors", + "model.layers.49.post_attention_layernorm.weight": "model-00039-of-00062.safetensors", + "model.layers.49.self_attn.k_proj.weight": "model-00038-of-00062.safetensors", + "model.layers.49.self_attn.o_proj.weight": "model-00038-of-00062.safetensors", + "model.layers.49.self_attn.q_proj.weight": "model-00038-of-00062.safetensors", + "model.layers.49.self_attn.v_proj.weight": "model-00038-of-00062.safetensors", + "model.layers.5.input_layernorm.weight": "model-00006-of-00062.safetensors", + "model.layers.5.mlp.down_proj.weight": "model-00006-of-00062.safetensors", + "model.layers.5.mlp.gate_proj.weight": "model-00005-of-00062.safetensors", + "model.layers.5.mlp.up_proj.weight": "model-00006-of-00062.safetensors", + "model.layers.5.post_attention_layernorm.weight": "model-00006-of-00062.safetensors", + "model.layers.5.self_attn.k_proj.weight": "model-00005-of-00062.safetensors", + "model.layers.5.self_attn.o_proj.weight": "model-00005-of-00062.safetensors", + "model.layers.5.self_attn.q_proj.weight": "model-00005-of-00062.safetensors", + "model.layers.5.self_attn.v_proj.weight": "model-00005-of-00062.safetensors", + "model.layers.50.input_layernorm.weight": "model-00040-of-00062.safetensors", + "model.layers.50.mlp.down_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.50.mlp.gate_proj.weight": "model-00039-of-00062.safetensors", + "model.layers.50.mlp.up_proj.weight": "model-00039-of-00062.safetensors", + "model.layers.50.post_attention_layernorm.weight": "model-00040-of-00062.safetensors", + "model.layers.50.self_attn.k_proj.weight": "model-00039-of-00062.safetensors", + "model.layers.50.self_attn.o_proj.weight": "model-00039-of-00062.safetensors", + "model.layers.50.self_attn.q_proj.weight": "model-00039-of-00062.safetensors", + "model.layers.50.self_attn.v_proj.weight": "model-00039-of-00062.safetensors", + "model.layers.51.input_layernorm.weight": "model-00040-of-00062.safetensors", + "model.layers.51.mlp.down_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.51.mlp.gate_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.51.mlp.up_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.51.post_attention_layernorm.weight": "model-00040-of-00062.safetensors", + "model.layers.51.self_attn.k_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.51.self_attn.o_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.51.self_attn.q_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.51.self_attn.v_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.52.input_layernorm.weight": "model-00041-of-00062.safetensors", + "model.layers.52.mlp.down_proj.weight": "model-00041-of-00062.safetensors", + "model.layers.52.mlp.gate_proj.weight": "model-00041-of-00062.safetensors", + "model.layers.52.mlp.up_proj.weight": "model-00041-of-00062.safetensors", + "model.layers.52.post_attention_layernorm.weight": "model-00041-of-00062.safetensors", + "model.layers.52.self_attn.k_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.52.self_attn.o_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.52.self_attn.q_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.52.self_attn.v_proj.weight": "model-00040-of-00062.safetensors", + "model.layers.53.input_layernorm.weight": "model-00042-of-00062.safetensors", + "model.layers.53.mlp.down_proj.weight": "model-00042-of-00062.safetensors", + "model.layers.53.mlp.gate_proj.weight": "model-00041-of-00062.safetensors", + "model.layers.53.mlp.up_proj.weight": "model-00042-of-00062.safetensors", + "model.layers.53.post_attention_layernorm.weight": "model-00042-of-00062.safetensors", + "model.layers.53.self_attn.k_proj.weight": "model-00041-of-00062.safetensors", + "model.layers.53.self_attn.o_proj.weight": "model-00041-of-00062.safetensors", + "model.layers.53.self_attn.q_proj.weight": "model-00041-of-00062.safetensors", + "model.layers.53.self_attn.v_proj.weight": "model-00041-of-00062.safetensors", + "model.layers.54.input_layernorm.weight": "model-00043-of-00062.safetensors", + "model.layers.54.mlp.down_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.54.mlp.gate_proj.weight": "model-00042-of-00062.safetensors", + "model.layers.54.mlp.up_proj.weight": "model-00042-of-00062.safetensors", + "model.layers.54.post_attention_layernorm.weight": "model-00043-of-00062.safetensors", + "model.layers.54.self_attn.k_proj.weight": "model-00042-of-00062.safetensors", + "model.layers.54.self_attn.o_proj.weight": "model-00042-of-00062.safetensors", + "model.layers.54.self_attn.q_proj.weight": "model-00042-of-00062.safetensors", + "model.layers.54.self_attn.v_proj.weight": "model-00042-of-00062.safetensors", + "model.layers.55.input_layernorm.weight": "model-00043-of-00062.safetensors", + "model.layers.55.mlp.down_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.55.mlp.gate_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.55.mlp.up_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.55.post_attention_layernorm.weight": "model-00043-of-00062.safetensors", + "model.layers.55.self_attn.k_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.55.self_attn.o_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.55.self_attn.q_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.55.self_attn.v_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.56.input_layernorm.weight": "model-00044-of-00062.safetensors", + "model.layers.56.mlp.down_proj.weight": "model-00044-of-00062.safetensors", + "model.layers.56.mlp.gate_proj.weight": "model-00044-of-00062.safetensors", + "model.layers.56.mlp.up_proj.weight": "model-00044-of-00062.safetensors", + "model.layers.56.post_attention_layernorm.weight": "model-00044-of-00062.safetensors", + "model.layers.56.self_attn.k_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.56.self_attn.o_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.56.self_attn.q_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.56.self_attn.v_proj.weight": "model-00043-of-00062.safetensors", + "model.layers.57.input_layernorm.weight": "model-00045-of-00062.safetensors", + "model.layers.57.mlp.down_proj.weight": "model-00045-of-00062.safetensors", + "model.layers.57.mlp.gate_proj.weight": "model-00044-of-00062.safetensors", + "model.layers.57.mlp.up_proj.weight": "model-00045-of-00062.safetensors", + "model.layers.57.post_attention_layernorm.weight": "model-00045-of-00062.safetensors", + "model.layers.57.self_attn.k_proj.weight": "model-00044-of-00062.safetensors", + "model.layers.57.self_attn.o_proj.weight": "model-00044-of-00062.safetensors", + "model.layers.57.self_attn.q_proj.weight": "model-00044-of-00062.safetensors", + "model.layers.57.self_attn.v_proj.weight": "model-00044-of-00062.safetensors", + "model.layers.58.input_layernorm.weight": "model-00046-of-00062.safetensors", + "model.layers.58.mlp.down_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.58.mlp.gate_proj.weight": "model-00045-of-00062.safetensors", + "model.layers.58.mlp.up_proj.weight": "model-00045-of-00062.safetensors", + "model.layers.58.post_attention_layernorm.weight": "model-00046-of-00062.safetensors", + "model.layers.58.self_attn.k_proj.weight": "model-00045-of-00062.safetensors", + "model.layers.58.self_attn.o_proj.weight": "model-00045-of-00062.safetensors", + "model.layers.58.self_attn.q_proj.weight": "model-00045-of-00062.safetensors", + "model.layers.58.self_attn.v_proj.weight": "model-00045-of-00062.safetensors", + "model.layers.59.input_layernorm.weight": "model-00046-of-00062.safetensors", + "model.layers.59.mlp.down_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.59.mlp.gate_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.59.mlp.up_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.59.post_attention_layernorm.weight": "model-00046-of-00062.safetensors", + "model.layers.59.self_attn.k_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.59.self_attn.o_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.59.self_attn.q_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.59.self_attn.v_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.6.input_layernorm.weight": "model-00007-of-00062.safetensors", + "model.layers.6.mlp.down_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.6.mlp.gate_proj.weight": "model-00006-of-00062.safetensors", + "model.layers.6.mlp.up_proj.weight": "model-00006-of-00062.safetensors", + "model.layers.6.post_attention_layernorm.weight": "model-00007-of-00062.safetensors", + "model.layers.6.self_attn.k_proj.weight": "model-00006-of-00062.safetensors", + "model.layers.6.self_attn.o_proj.weight": "model-00006-of-00062.safetensors", + "model.layers.6.self_attn.q_proj.weight": "model-00006-of-00062.safetensors", + "model.layers.6.self_attn.v_proj.weight": "model-00006-of-00062.safetensors", + "model.layers.60.input_layernorm.weight": "model-00047-of-00062.safetensors", + "model.layers.60.mlp.down_proj.weight": "model-00047-of-00062.safetensors", + "model.layers.60.mlp.gate_proj.weight": "model-00047-of-00062.safetensors", + "model.layers.60.mlp.up_proj.weight": "model-00047-of-00062.safetensors", + "model.layers.60.post_attention_layernorm.weight": "model-00047-of-00062.safetensors", + "model.layers.60.self_attn.k_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.60.self_attn.o_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.60.self_attn.q_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.60.self_attn.v_proj.weight": "model-00046-of-00062.safetensors", + "model.layers.61.input_layernorm.weight": "model-00048-of-00062.safetensors", + "model.layers.61.mlp.down_proj.weight": "model-00048-of-00062.safetensors", + "model.layers.61.mlp.gate_proj.weight": "model-00047-of-00062.safetensors", + "model.layers.61.mlp.up_proj.weight": "model-00048-of-00062.safetensors", + "model.layers.61.post_attention_layernorm.weight": "model-00048-of-00062.safetensors", + "model.layers.61.self_attn.k_proj.weight": "model-00047-of-00062.safetensors", + "model.layers.61.self_attn.o_proj.weight": "model-00047-of-00062.safetensors", + "model.layers.61.self_attn.q_proj.weight": "model-00047-of-00062.safetensors", + "model.layers.61.self_attn.v_proj.weight": "model-00047-of-00062.safetensors", + "model.layers.62.input_layernorm.weight": "model-00049-of-00062.safetensors", + "model.layers.62.mlp.down_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.62.mlp.gate_proj.weight": "model-00048-of-00062.safetensors", + "model.layers.62.mlp.up_proj.weight": "model-00048-of-00062.safetensors", + "model.layers.62.post_attention_layernorm.weight": "model-00049-of-00062.safetensors", + "model.layers.62.self_attn.k_proj.weight": "model-00048-of-00062.safetensors", + "model.layers.62.self_attn.o_proj.weight": "model-00048-of-00062.safetensors", + "model.layers.62.self_attn.q_proj.weight": "model-00048-of-00062.safetensors", + "model.layers.62.self_attn.v_proj.weight": "model-00048-of-00062.safetensors", + "model.layers.63.input_layernorm.weight": "model-00049-of-00062.safetensors", + "model.layers.63.mlp.down_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.63.mlp.gate_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.63.mlp.up_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.63.post_attention_layernorm.weight": "model-00049-of-00062.safetensors", + "model.layers.63.self_attn.k_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.63.self_attn.o_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.63.self_attn.q_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.63.self_attn.v_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.64.input_layernorm.weight": "model-00050-of-00062.safetensors", + "model.layers.64.mlp.down_proj.weight": "model-00050-of-00062.safetensors", + "model.layers.64.mlp.gate_proj.weight": "model-00050-of-00062.safetensors", + "model.layers.64.mlp.up_proj.weight": "model-00050-of-00062.safetensors", + "model.layers.64.post_attention_layernorm.weight": "model-00050-of-00062.safetensors", + "model.layers.64.self_attn.k_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.64.self_attn.o_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.64.self_attn.q_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.64.self_attn.v_proj.weight": "model-00049-of-00062.safetensors", + "model.layers.65.input_layernorm.weight": "model-00051-of-00062.safetensors", + "model.layers.65.mlp.down_proj.weight": "model-00051-of-00062.safetensors", + "model.layers.65.mlp.gate_proj.weight": "model-00050-of-00062.safetensors", + "model.layers.65.mlp.up_proj.weight": "model-00051-of-00062.safetensors", + "model.layers.65.post_attention_layernorm.weight": "model-00051-of-00062.safetensors", + "model.layers.65.self_attn.k_proj.weight": "model-00050-of-00062.safetensors", + "model.layers.65.self_attn.o_proj.weight": "model-00050-of-00062.safetensors", + "model.layers.65.self_attn.q_proj.weight": "model-00050-of-00062.safetensors", + "model.layers.65.self_attn.v_proj.weight": "model-00050-of-00062.safetensors", + "model.layers.66.input_layernorm.weight": "model-00052-of-00062.safetensors", + "model.layers.66.mlp.down_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.66.mlp.gate_proj.weight": "model-00051-of-00062.safetensors", + "model.layers.66.mlp.up_proj.weight": "model-00051-of-00062.safetensors", + "model.layers.66.post_attention_layernorm.weight": "model-00052-of-00062.safetensors", + "model.layers.66.self_attn.k_proj.weight": "model-00051-of-00062.safetensors", + "model.layers.66.self_attn.o_proj.weight": "model-00051-of-00062.safetensors", + "model.layers.66.self_attn.q_proj.weight": "model-00051-of-00062.safetensors", + "model.layers.66.self_attn.v_proj.weight": "model-00051-of-00062.safetensors", + "model.layers.67.input_layernorm.weight": "model-00052-of-00062.safetensors", + "model.layers.67.mlp.down_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.67.mlp.gate_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.67.mlp.up_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.67.post_attention_layernorm.weight": "model-00052-of-00062.safetensors", + "model.layers.67.self_attn.k_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.67.self_attn.o_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.67.self_attn.q_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.67.self_attn.v_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.68.input_layernorm.weight": "model-00053-of-00062.safetensors", + "model.layers.68.mlp.down_proj.weight": "model-00053-of-00062.safetensors", + "model.layers.68.mlp.gate_proj.weight": "model-00053-of-00062.safetensors", + "model.layers.68.mlp.up_proj.weight": "model-00053-of-00062.safetensors", + "model.layers.68.post_attention_layernorm.weight": "model-00053-of-00062.safetensors", + "model.layers.68.self_attn.k_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.68.self_attn.o_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.68.self_attn.q_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.68.self_attn.v_proj.weight": "model-00052-of-00062.safetensors", + "model.layers.69.input_layernorm.weight": "model-00054-of-00062.safetensors", + "model.layers.69.mlp.down_proj.weight": "model-00054-of-00062.safetensors", + "model.layers.69.mlp.gate_proj.weight": "model-00053-of-00062.safetensors", + "model.layers.69.mlp.up_proj.weight": "model-00054-of-00062.safetensors", + "model.layers.69.post_attention_layernorm.weight": "model-00054-of-00062.safetensors", + "model.layers.69.self_attn.k_proj.weight": "model-00053-of-00062.safetensors", + "model.layers.69.self_attn.o_proj.weight": "model-00053-of-00062.safetensors", + "model.layers.69.self_attn.q_proj.weight": "model-00053-of-00062.safetensors", + "model.layers.69.self_attn.v_proj.weight": "model-00053-of-00062.safetensors", + "model.layers.7.input_layernorm.weight": "model-00007-of-00062.safetensors", + "model.layers.7.mlp.down_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.7.mlp.gate_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.7.mlp.up_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.7.post_attention_layernorm.weight": "model-00007-of-00062.safetensors", + "model.layers.7.self_attn.k_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.7.self_attn.o_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.7.self_attn.q_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.7.self_attn.v_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.70.input_layernorm.weight": "model-00055-of-00062.safetensors", + "model.layers.70.mlp.down_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.70.mlp.gate_proj.weight": "model-00054-of-00062.safetensors", + "model.layers.70.mlp.up_proj.weight": "model-00054-of-00062.safetensors", + "model.layers.70.post_attention_layernorm.weight": "model-00055-of-00062.safetensors", + "model.layers.70.self_attn.k_proj.weight": "model-00054-of-00062.safetensors", + "model.layers.70.self_attn.o_proj.weight": "model-00054-of-00062.safetensors", + "model.layers.70.self_attn.q_proj.weight": "model-00054-of-00062.safetensors", + "model.layers.70.self_attn.v_proj.weight": "model-00054-of-00062.safetensors", + "model.layers.71.input_layernorm.weight": "model-00055-of-00062.safetensors", + "model.layers.71.mlp.down_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.71.mlp.gate_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.71.mlp.up_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.71.post_attention_layernorm.weight": "model-00055-of-00062.safetensors", + "model.layers.71.self_attn.k_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.71.self_attn.o_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.71.self_attn.q_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.71.self_attn.v_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.72.input_layernorm.weight": "model-00056-of-00062.safetensors", + "model.layers.72.mlp.down_proj.weight": "model-00056-of-00062.safetensors", + "model.layers.72.mlp.gate_proj.weight": "model-00056-of-00062.safetensors", + "model.layers.72.mlp.up_proj.weight": "model-00056-of-00062.safetensors", + "model.layers.72.post_attention_layernorm.weight": "model-00056-of-00062.safetensors", + "model.layers.72.self_attn.k_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.72.self_attn.o_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.72.self_attn.q_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.72.self_attn.v_proj.weight": "model-00055-of-00062.safetensors", + "model.layers.73.input_layernorm.weight": "model-00057-of-00062.safetensors", + "model.layers.73.mlp.down_proj.weight": "model-00057-of-00062.safetensors", + "model.layers.73.mlp.gate_proj.weight": "model-00056-of-00062.safetensors", + "model.layers.73.mlp.up_proj.weight": "model-00057-of-00062.safetensors", + "model.layers.73.post_attention_layernorm.weight": "model-00057-of-00062.safetensors", + "model.layers.73.self_attn.k_proj.weight": "model-00056-of-00062.safetensors", + "model.layers.73.self_attn.o_proj.weight": "model-00056-of-00062.safetensors", + "model.layers.73.self_attn.q_proj.weight": "model-00056-of-00062.safetensors", + "model.layers.73.self_attn.v_proj.weight": "model-00056-of-00062.safetensors", + "model.layers.74.input_layernorm.weight": "model-00058-of-00062.safetensors", + "model.layers.74.mlp.down_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.74.mlp.gate_proj.weight": "model-00057-of-00062.safetensors", + "model.layers.74.mlp.up_proj.weight": "model-00057-of-00062.safetensors", + "model.layers.74.post_attention_layernorm.weight": "model-00058-of-00062.safetensors", + "model.layers.74.self_attn.k_proj.weight": "model-00057-of-00062.safetensors", + "model.layers.74.self_attn.o_proj.weight": "model-00057-of-00062.safetensors", + "model.layers.74.self_attn.q_proj.weight": "model-00057-of-00062.safetensors", + "model.layers.74.self_attn.v_proj.weight": "model-00057-of-00062.safetensors", + "model.layers.75.input_layernorm.weight": "model-00058-of-00062.safetensors", + "model.layers.75.mlp.down_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.75.mlp.gate_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.75.mlp.up_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.75.post_attention_layernorm.weight": "model-00058-of-00062.safetensors", + "model.layers.75.self_attn.k_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.75.self_attn.o_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.75.self_attn.q_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.75.self_attn.v_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.76.input_layernorm.weight": "model-00059-of-00062.safetensors", + "model.layers.76.mlp.down_proj.weight": "model-00059-of-00062.safetensors", + "model.layers.76.mlp.gate_proj.weight": "model-00059-of-00062.safetensors", + "model.layers.76.mlp.up_proj.weight": "model-00059-of-00062.safetensors", + "model.layers.76.post_attention_layernorm.weight": "model-00059-of-00062.safetensors", + "model.layers.76.self_attn.k_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.76.self_attn.o_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.76.self_attn.q_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.76.self_attn.v_proj.weight": "model-00058-of-00062.safetensors", + "model.layers.77.input_layernorm.weight": "model-00060-of-00062.safetensors", + "model.layers.77.mlp.down_proj.weight": "model-00060-of-00062.safetensors", + "model.layers.77.mlp.gate_proj.weight": "model-00059-of-00062.safetensors", + "model.layers.77.mlp.up_proj.weight": "model-00060-of-00062.safetensors", + "model.layers.77.post_attention_layernorm.weight": "model-00060-of-00062.safetensors", + "model.layers.77.self_attn.k_proj.weight": "model-00059-of-00062.safetensors", + "model.layers.77.self_attn.o_proj.weight": "model-00059-of-00062.safetensors", + "model.layers.77.self_attn.q_proj.weight": "model-00059-of-00062.safetensors", + "model.layers.77.self_attn.v_proj.weight": "model-00059-of-00062.safetensors", + "model.layers.78.input_layernorm.weight": "model-00061-of-00062.safetensors", + "model.layers.78.mlp.down_proj.weight": "model-00061-of-00062.safetensors", + "model.layers.78.mlp.gate_proj.weight": "model-00060-of-00062.safetensors", + "model.layers.78.mlp.up_proj.weight": "model-00060-of-00062.safetensors", + "model.layers.78.post_attention_layernorm.weight": "model-00061-of-00062.safetensors", + "model.layers.78.self_attn.k_proj.weight": "model-00060-of-00062.safetensors", + "model.layers.78.self_attn.o_proj.weight": "model-00060-of-00062.safetensors", + "model.layers.78.self_attn.q_proj.weight": "model-00060-of-00062.safetensors", + "model.layers.78.self_attn.v_proj.weight": "model-00060-of-00062.safetensors", + "model.layers.79.input_layernorm.weight": "model-00061-of-00062.safetensors", + "model.layers.79.mlp.down_proj.weight": "model-00061-of-00062.safetensors", + "model.layers.79.mlp.gate_proj.weight": "model-00061-of-00062.safetensors", + "model.layers.79.mlp.up_proj.weight": "model-00061-of-00062.safetensors", + "model.layers.79.post_attention_layernorm.weight": "model-00061-of-00062.safetensors", + "model.layers.79.self_attn.k_proj.weight": "model-00061-of-00062.safetensors", + "model.layers.79.self_attn.o_proj.weight": "model-00061-of-00062.safetensors", + "model.layers.79.self_attn.q_proj.weight": "model-00061-of-00062.safetensors", + "model.layers.79.self_attn.v_proj.weight": "model-00061-of-00062.safetensors", + "model.layers.8.input_layernorm.weight": "model-00008-of-00062.safetensors", + "model.layers.8.mlp.down_proj.weight": "model-00008-of-00062.safetensors", + "model.layers.8.mlp.gate_proj.weight": "model-00008-of-00062.safetensors", + "model.layers.8.mlp.up_proj.weight": "model-00008-of-00062.safetensors", + "model.layers.8.post_attention_layernorm.weight": "model-00008-of-00062.safetensors", + "model.layers.8.self_attn.k_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.8.self_attn.o_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.8.self_attn.q_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.8.self_attn.v_proj.weight": "model-00007-of-00062.safetensors", + "model.layers.9.input_layernorm.weight": "model-00009-of-00062.safetensors", + "model.layers.9.mlp.down_proj.weight": "model-00009-of-00062.safetensors", + "model.layers.9.mlp.gate_proj.weight": "model-00008-of-00062.safetensors", + "model.layers.9.mlp.up_proj.weight": "model-00009-of-00062.safetensors", + "model.layers.9.post_attention_layernorm.weight": "model-00009-of-00062.safetensors", + "model.layers.9.self_attn.k_proj.weight": "model-00008-of-00062.safetensors", + "model.layers.9.self_attn.o_proj.weight": "model-00008-of-00062.safetensors", + "model.layers.9.self_attn.q_proj.weight": "model-00008-of-00062.safetensors", + "model.layers.9.self_attn.v_proj.weight": "model-00008-of-00062.safetensors", + "model.norm.weight": "model-00061-of-00062.safetensors" + } +}