KitsuVp commited on
Commit
91342c4
·
verified ·
1 Parent(s): 804ee22

Model save

Browse files
README.md CHANGED
@@ -14,7 +14,7 @@ should probably proofread and complete it, then remove this comment. -->
14
 
15
  This model is a fine-tuned version of [](https://huggingface.co/) on an unknown dataset.
16
  It achieves the following results on the evaluation set:
17
- - Loss: 2.7250
18
 
19
  ## Model description
20
 
@@ -39,27 +39,21 @@ The following hyperparameters were used during training:
39
  - seed: 42
40
  - optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
41
  - lr_scheduler_type: linear
42
- - lr_scheduler_warmup_ratio: 0.1
43
  - num_epochs: 1
44
 
45
  ### Training results
46
 
47
- | Training Loss | Epoch | Step | Validation Loss |
48
- |:-------------:|:------:|:-----:|:---------------:|
49
- | 3.7833 | 0.1067 | 5000 | 3.6697 |
50
- | 3.4271 | 0.2133 | 10000 | 3.2796 |
51
- | 3.2851 | 0.32 | 15000 | 3.1214 |
52
- | 3.2115 | 0.4267 | 20000 | 3.0228 |
53
- | 3.166 | 0.5333 | 25000 | 2.9708 |
54
- | 3.118 | 0.64 | 30000 | 2.9159 |
55
- | 3.0656 | 0.7467 | 35000 | 2.8546 |
56
- | 2.9881 | 0.8533 | 40000 | 2.7819 |
57
- | 2.9405 | 0.96 | 45000 | 2.7250 |
58
 
59
 
60
  ### Framework versions
61
 
62
- - Transformers 4.57.6
63
  - Pytorch 2.10.0+cu130
64
  - Datasets 4.5.0
65
  - Tokenizers 0.22.2
 
14
 
15
  This model is a fine-tuned version of [](https://huggingface.co/) on an unknown dataset.
16
  It achieves the following results on the evaluation set:
17
+ - Loss: 3.0815
18
 
19
  ## Model description
20
 
 
39
  - seed: 42
40
  - optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
41
  - lr_scheduler_type: linear
42
+ - lr_scheduler_warmup_steps: 0.1
43
  - num_epochs: 1
44
 
45
  ### Training results
46
 
47
+ | Training Loss | Epoch | Step | Validation Loss |
48
+ |:-------------:|:-----:|:-----:|:---------------:|
49
+ | 3.7890 | 0.32 | 5000 | 3.7263 |
50
+ | 3.4311 | 0.64 | 10000 | 3.3235 |
51
+ | 3.2317 | 0.96 | 15000 | 3.0815 |
 
 
 
 
 
 
52
 
53
 
54
  ### Framework versions
55
 
56
+ - Transformers 5.0.0
57
  - Pytorch 2.10.0+cu130
58
  - Datasets 4.5.0
59
  - Tokenizers 0.22.2
config.json CHANGED
@@ -28,11 +28,17 @@
28
  "pad_token_id": 151643,
29
  "partial_rotary_factor": 0.25,
30
  "rms_norm_eps": 1e-06,
31
- "rope_scaling": null,
 
 
 
 
32
  "rope_theta": 10000.0,
33
  "stack_d_model": 16,
34
  "stack_slots": 24,
35
- "transformers_version": "4.57.6",
 
 
36
  "use_stack": true,
37
  "vocab_size": 151665
38
  }
 
28
  "pad_token_id": 151643,
29
  "partial_rotary_factor": 0.25,
30
  "rms_norm_eps": 1e-06,
31
+ "rope_parameters": {
32
+ "partial_rotary_factor": 0.25,
33
+ "rope_theta": 10000.0,
34
+ "rope_type": "default"
35
+ },
36
  "rope_theta": 10000.0,
37
  "stack_d_model": 16,
38
  "stack_slots": 24,
39
+ "tie_word_embeddings": true,
40
+ "transformers_version": "5.0.0",
41
+ "use_cache": false,
42
  "use_stack": true,
43
  "vocab_size": 151665
44
  }
generation_config.json CHANGED
@@ -3,6 +3,8 @@
3
  "eos_token_id": [
4
  151643
5
  ],
 
 
6
  "pad_token_id": 151643,
7
- "transformers_version": "4.57.6"
8
  }
 
3
  "eos_token_id": [
4
  151643
5
  ],
6
+ "output_attentions": false,
7
+ "output_hidden_states": false,
8
  "pad_token_id": 151643,
9
+ "transformers_version": "5.0.0"
10
  }
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cc2cbf5ebfaeea4dbc63143d14ca208f4b5531357f7f31c2a4688e3b762c510b
3
  size 251434000
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f1823eba877b831ce1f0475d38acc179ee2f723a585009a759575f80d0ae3d54
3
  size 251434000
tokenizer.json CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
- size 11421896
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
3
+ size 11421892
tokenizer_config.json CHANGED
@@ -1,185 +1,11 @@
1
  {
2
- "add_bos_token": false,
3
  "add_prefix_space": false,
4
- "added_tokens_decoder": {
5
- "151643": {
6
- "content": "<|endoftext|>",
7
- "lstrip": false,
8
- "normalized": false,
9
- "rstrip": false,
10
- "single_word": false,
11
- "special": true
12
- },
13
- "151644": {
14
- "content": "<|im_start|>",
15
- "lstrip": false,
16
- "normalized": false,
17
- "rstrip": false,
18
- "single_word": false,
19
- "special": true
20
- },
21
- "151645": {
22
- "content": "<|im_end|>",
23
- "lstrip": false,
24
- "normalized": false,
25
- "rstrip": false,
26
- "single_word": false,
27
- "special": true
28
- },
29
- "151646": {
30
- "content": "<|object_ref_start|>",
31
- "lstrip": false,
32
- "normalized": false,
33
- "rstrip": false,
34
- "single_word": false,
35
- "special": true
36
- },
37
- "151647": {
38
- "content": "<|object_ref_end|>",
39
- "lstrip": false,
40
- "normalized": false,
41
- "rstrip": false,
42
- "single_word": false,
43
- "special": true
44
- },
45
- "151648": {
46
- "content": "<|box_start|>",
47
- "lstrip": false,
48
- "normalized": false,
49
- "rstrip": false,
50
- "single_word": false,
51
- "special": true
52
- },
53
- "151649": {
54
- "content": "<|box_end|>",
55
- "lstrip": false,
56
- "normalized": false,
57
- "rstrip": false,
58
- "single_word": false,
59
- "special": true
60
- },
61
- "151650": {
62
- "content": "<|quad_start|>",
63
- "lstrip": false,
64
- "normalized": false,
65
- "rstrip": false,
66
- "single_word": false,
67
- "special": true
68
- },
69
- "151651": {
70
- "content": "<|quad_end|>",
71
- "lstrip": false,
72
- "normalized": false,
73
- "rstrip": false,
74
- "single_word": false,
75
- "special": true
76
- },
77
- "151652": {
78
- "content": "<|vision_start|>",
79
- "lstrip": false,
80
- "normalized": false,
81
- "rstrip": false,
82
- "single_word": false,
83
- "special": true
84
- },
85
- "151653": {
86
- "content": "<|vision_end|>",
87
- "lstrip": false,
88
- "normalized": false,
89
- "rstrip": false,
90
- "single_word": false,
91
- "special": true
92
- },
93
- "151654": {
94
- "content": "<|vision_pad|>",
95
- "lstrip": false,
96
- "normalized": false,
97
- "rstrip": false,
98
- "single_word": false,
99
- "special": true
100
- },
101
- "151655": {
102
- "content": "<|image_pad|>",
103
- "lstrip": false,
104
- "normalized": false,
105
- "rstrip": false,
106
- "single_word": false,
107
- "special": true
108
- },
109
- "151656": {
110
- "content": "<|video_pad|>",
111
- "lstrip": false,
112
- "normalized": false,
113
- "rstrip": false,
114
- "single_word": false,
115
- "special": true
116
- },
117
- "151657": {
118
- "content": "<tool_call>",
119
- "lstrip": false,
120
- "normalized": false,
121
- "rstrip": false,
122
- "single_word": false,
123
- "special": false
124
- },
125
- "151658": {
126
- "content": "</tool_call>",
127
- "lstrip": false,
128
- "normalized": false,
129
- "rstrip": false,
130
- "single_word": false,
131
- "special": false
132
- },
133
- "151659": {
134
- "content": "<|fim_prefix|>",
135
- "lstrip": false,
136
- "normalized": false,
137
- "rstrip": false,
138
- "single_word": false,
139
- "special": false
140
- },
141
- "151660": {
142
- "content": "<|fim_middle|>",
143
- "lstrip": false,
144
- "normalized": false,
145
- "rstrip": false,
146
- "single_word": false,
147
- "special": false
148
- },
149
- "151661": {
150
- "content": "<|fim_suffix|>",
151
- "lstrip": false,
152
- "normalized": false,
153
- "rstrip": false,
154
- "single_word": false,
155
- "special": false
156
- },
157
- "151662": {
158
- "content": "<|fim_pad|>",
159
- "lstrip": false,
160
- "normalized": false,
161
- "rstrip": false,
162
- "single_word": false,
163
- "special": false
164
- },
165
- "151663": {
166
- "content": "<|repo_name|>",
167
- "lstrip": false,
168
- "normalized": false,
169
- "rstrip": false,
170
- "single_word": false,
171
- "special": false
172
- },
173
- "151664": {
174
- "content": "<|file_sep|>",
175
- "lstrip": false,
176
- "normalized": false,
177
- "rstrip": false,
178
- "single_word": false,
179
- "special": false
180
- }
181
- },
182
- "additional_special_tokens": [
183
  "<|im_start|>",
184
  "<|im_end|>",
185
  "<|object_ref_start|>",
@@ -194,11 +20,7 @@
194
  "<|image_pad|>",
195
  "<|video_pad|>"
196
  ],
197
- "bos_token": null,
198
- "clean_up_tokenization_spaces": false,
199
- "eos_token": "<|endoftext|>",
200
- "errors": "replace",
201
- "extra_special_tokens": {},
202
  "model_max_length": 131072,
203
  "pad_token": "<|endoftext|>",
204
  "split_special_tokens": false,
 
1
  {
 
2
  "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|endoftext|>",
7
+ "errors": "replace",
8
+ "extra_special_tokens": [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9
  "<|im_start|>",
10
  "<|im_end|>",
11
  "<|object_ref_start|>",
 
20
  "<|image_pad|>",
21
  "<|video_pad|>"
22
  ],
23
+ "is_local": false,
 
 
 
 
24
  "model_max_length": 131072,
25
  "pad_token": "<|endoftext|>",
26
  "split_special_tokens": false,
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:22cc0c67d1158f3a68631cff288743112f63360559878c1007d8da0095cfb98f
3
- size 6033
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e197c0c15fa0b6f99e507c6cd1cf5239dfb11b35f9616e30f71e431e268c8392
3
+ size 5329